- schemas/mi-strategija-0.1.xsd: Stratēģija → Pīlārs → Mērķis → Rīcība - data/pppa-mis-2026-09-10.xml: 3 pīlāri, 5 mērķi, 27 rīcības (burtiski no v2.2) - tools/mi_strategija_no_html.py, tools/validate.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
187 lines
8.8 KiB
Python
187 lines
8.8 KiB
Python
#!/usr/bin/env python3
|
|
"""PPPA MI stratēģijas lapa (pppa.lv/mi-strategija) -> mašīnlasāms XML.
|
|
|
|
Lietojums:
|
|
python3 tools/mi_strategija_no_html.py source/mi-strategija-v2.2.html data/
|
|
|
|
Teksts tiek pārņemts burtiski; HTML formatējums (treknraksts, saites) tiek
|
|
noņemts, saišu adreses saglabātas atribūtā @href.
|
|
"""
|
|
import html
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
from xml.sax.saxutils import escape, quoteattr
|
|
|
|
NS = "urn:pppa:cac:mi-strategija:0.1"
|
|
ROMAN = {"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5}
|
|
|
|
|
|
def text(fragment):
|
|
"""HTML fragments -> tīrs teksts vienā rindā."""
|
|
s = re.sub(r"<svg.*?</svg>", "", fragment, flags=re.S)
|
|
s = re.sub(r"<br\s*/?>", " ", s)
|
|
s = re.sub(r"<[^>]+>", "", s)
|
|
s = html.unescape(s)
|
|
return re.sub(r"\s+", " ", s).strip()
|
|
|
|
|
|
def one(pattern, s, default=None):
|
|
m = re.search(pattern, s, re.S)
|
|
if not m:
|
|
if default is not None:
|
|
return default
|
|
raise ValueError("Nav atrasts: " + pattern)
|
|
return m.group(1)
|
|
|
|
|
|
def absurl(u):
|
|
return "https://pppa.lv" + u if u.startswith("/") else u
|
|
|
|
|
|
def date_iso(lv):
|
|
m = re.fullmatch(r"(\d{2})\.(\d{2})\.(\d{4})\.?", lv.strip())
|
|
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
|
|
|
|
|
|
def el(name, txt, indent, **attrs):
|
|
a = "".join(f" {k}={quoteattr(str(v))}" for k, v in attrs.items() if v not in (None, ""))
|
|
return f"{' ' * indent}<{name}{a}>{escape(txt)}</{name}>"
|
|
|
|
|
|
def main(src, out_dir):
|
|
h = Path(src).read_text(encoding="utf-8")
|
|
h = re.sub(r"<script.*?</script>|<style.*?</style>", "", h, flags=re.S)
|
|
|
|
# --- Galvene ---
|
|
hero = one(r'<section class="strat-hero">(.*?)</section>', h)
|
|
title = text(one(r"<h1>(.*?)</h1>", hero))
|
|
status = text(one(r'<p class="eyebrow">(.*?)</p>', hero))
|
|
slogan = [text(x) for x in re.split(r'<span class="hs-sep">.*?</span>',
|
|
one(r'<p class="hero-slogan">(.*?)</p>', hero))]
|
|
vision = text(one(r'<p class="lead">(.*?)</p>', hero))
|
|
summary = one(r'<div class="summary-box" id="kopsavilkums">(.*?)</div>', h)
|
|
upd = one(r"Pēdējoreiz atjaunots:\s*([\d.]+)", summary).rstrip(".")
|
|
version = one(r"Darba versija\s*(v[\d.]+)", summary)
|
|
intake = one(r"Priekšlikumu saņemšana uzsākta <strong[^>]*>([\d.]+)</strong>", summary).rstrip(".")
|
|
doc_id = f"pppa-mis-{date_iso(upd)}"
|
|
|
|
# --- Pīlāru kartītes (kopsavilkums) ---
|
|
cards = {}
|
|
for c in re.finditer(r'<button type="button" class="pillar-card[^"]*" data-tab="(\d)">(.*?)</button>', h, re.S):
|
|
cards[int(c.group(1))] = {
|
|
"name": text(one(r'<span class="pc-jumbo">(.*?)</span>', c.group(2))),
|
|
"short": text(one(r'<span class="pc-sub">(.*?)</span>', c.group(2))),
|
|
"summary": text(one(r'<span class="pc-text">(.*?)</span>', c.group(2))),
|
|
}
|
|
|
|
# --- Pīlāri, mērķi, rīcības ---
|
|
panels = one(r'<div class="pillar-panels">(.*?)</div><!-- /pillar-panels -->', h)
|
|
pillars = re.split(r'<section class="pillar pillar-panel[^"]*"', panels)[1:]
|
|
xml_p, refs_all, action_ids = [], [], set()
|
|
for p in pillars:
|
|
rn = one(r'<span class="pillar-rn">(\w+)</span>', p)
|
|
pn = ROMAN[rn]
|
|
pid = f"{doc_id}.p{pn}"
|
|
head = one(r'<div class="pillar-head">(.*?)</div>', p)
|
|
out = [f' <Pillar id="{pid}" number="{rn}">',
|
|
el("Name", cards[pn]["name"], 4),
|
|
el("ShortTitle", cards[pn]["short"], 4),
|
|
el("Heading", text(one(r"<h2>(.*?)</h2>", head)), 4),
|
|
el("Summary", cards[pn]["summary"], 4),
|
|
el("Description", text(one(r"</h2>\s*<p>(.*?)</p>", head)), 4)]
|
|
for g in re.split(r'<div class="goal-block">', p)[1:]:
|
|
gn = int(one(r'<span class="goal-num">(\d+)\.</span>', g))
|
|
gid = f"{doc_id}.m{gn}"
|
|
out += [f' <Goal id="{gid}" number="{gn}.">',
|
|
el("Title", text(one(r"<h3>(.*?)</h3>", g)), 6),
|
|
el("Description", text(one(r"</h3>\s*<p>(.*?)</p>", g)), 6)]
|
|
for a in re.finditer(r'<div class="action-card" id="pasakums-(\d+)">(.*?)\n\s*</div>\s*(?=<div class="action-card"|</div>\s*</div>)', g, re.S):
|
|
an, body = int(a.group(1)), a.group(2)
|
|
aid = f"{doc_id}.r{an:02d}"
|
|
action_ids.add(an)
|
|
printed = text(one(r'<span class="ac-num">(.*?)</span>', body))
|
|
assert printed == f"Rīcība {an}", (printed, an)
|
|
dl = text(one(r'<span class="deadline-badge">(.*?)</span>', re.sub(r"<svg.*?</svg>", "", body, flags=re.S)))
|
|
out.append(f' <Action id="{aid}" number="{an}">')
|
|
for t in re.finditer(r'<span class="ac-tag( rec)?">(.*?)</span>', body):
|
|
out.append(el("Tag", text(t.group(2)), 8,
|
|
kind="parvaldibas-rekomendacija" if t.group(1) else None))
|
|
out += [el("Title", text(one(r"<h4>(.*?)</h4>", body)), 8),
|
|
el("Text", text(one(r'<p class="ac-body">(.*?)</p>', body)), 8),
|
|
el("Measurement", text(re.sub(r"<strong>.*?</strong>", "",
|
|
one(r'<div class="ac-measure">(.*?)</div>', body), count=1)), 8),
|
|
el("Deadline", dl, 8, value=date_iso(dl))]
|
|
for r in re.findall(r'<p class="ac-ref">(.*?)</p>', body, re.S):
|
|
t = text(r)
|
|
kind = {"Saistīts ar:": "saistits", "Avots:": "avots", "Piemērs:": "piemers"}
|
|
k = next((v for p_, v in kind.items() if t.startswith(p_)), None)
|
|
if k:
|
|
t = t.split(":", 1)[1].strip()
|
|
hrefs = " ".join(absurl(u) for u in re.findall(r'href="([^"]+)"', r))
|
|
nums = re.findall(r"Rīcība (\d+)", t)
|
|
refs_all.append((an, nums))
|
|
acts = " ".join(f"{doc_id}.r{int(n):02d}" for n in nums)
|
|
out.append(el("Reference", t, 8, kind=k, actions=acts, href=hrefs))
|
|
out.append(" </Action>")
|
|
out.append(" </Goal>")
|
|
out.append(" </Pillar>")
|
|
xml_p += out
|
|
|
|
# --- Principi ---
|
|
pr_sec = one(r'id="principi">(.*?)<hr class="section-divider">', h)
|
|
pr_intro = text(one(r'<p class="section-intro">(.*?)</p>', pr_sec))
|
|
xml_pr = [' <Principles>', el("Heading", text(one(r"^(.*?)</h2>", pr_sec)), 4),
|
|
el("Intro", pr_intro, 4)]
|
|
for i, b in enumerate(re.findall(r'<div class="mg-box">(.*?)</div>', pr_sec, re.S), 1):
|
|
xml_pr += [f' <Principle id="{doc_id}.pr{i}">',
|
|
el("Label", text(one(r'<p class="mg-label">(.*?)</p>', b)), 6),
|
|
el("Text", text(one(r'</p>\s*<p>(.*?)</p>', b)), 6),
|
|
" </Principle>"]
|
|
xml_pr.append(" </Principles>")
|
|
|
|
# --- Saņemtie priekšlikumi ---
|
|
pk = one(r'id="priekslikumi">.*?<p class="section-intro">(.*?)</p>', h)
|
|
pk_nums = sorted({int(n) for n in re.findall(r"\((?:Rīcība )?(\d+)\)", text(pk))})
|
|
xml_pk = [el("ProposalsNote", text(pk), 2,
|
|
actions=" ".join(f"{doc_id}.r{n:02d}" for n in pk_nums))]
|
|
|
|
# --- Starptautiskā pieredze ---
|
|
xml_rs = [" <Resources>"]
|
|
for i, c in enumerate(re.findall(r'<div class="resource-card">(.*?)</div>', h, re.S), 1):
|
|
xml_rs += [f' <Resource id="{doc_id}.a{i}" href={quoteattr(one(r'href="([^"]+)"', c))}>',
|
|
el("Country", text(one(r'<p class="rc-country">(.*?)</p>', c)), 6),
|
|
el("Title", text(one(r"<h3>(.*?)</h3>", c)), 6),
|
|
el("Text", text(one(r"</h3>\s*<p>(.*?)</p>", c)), 6),
|
|
" </Resource>"]
|
|
xml_rs.append(" </Resources>")
|
|
|
|
# --- Pārbaudes ---
|
|
expect = set(range(1, max(action_ids) + 1))
|
|
missing = expect - action_ids
|
|
bad = [(a, n) for a, ns in refs_all for n in ns if int(n) not in action_ids]
|
|
print(f"{doc_id}: {len(pillars)} pīlāri, {len(action_ids)} rīcības; trūkst {sorted(missing) or '—'}; "
|
|
f"nederīgas atsauces {bad or '—'}")
|
|
|
|
head = [
|
|
'<?xml version="1.0" encoding="UTF-8"?>',
|
|
f'<AIStrategy xmlns="{NS}" id="{doc_id}" version="{version}">',
|
|
el("Title", title, 2),
|
|
el("Author", "PPP Asociācija (PPPA)", 2),
|
|
el("Status", status, 2),
|
|
el("Updated", upd + ".", 2, value=date_iso(upd)),
|
|
el("ProposalsOpened", intake + ".", 2, value=date_iso(intake)),
|
|
el("SourceUrl", "https://pppa.lv/mi-strategija", 2),
|
|
" <Slogan>" + "".join(el("Word", w, 0) for w in slogan) + "</Slogan>",
|
|
el("Vision", vision, 2),
|
|
]
|
|
xml = "\n".join(head + xml_p + xml_pr + xml_pk + xml_rs + ["</AIStrategy>", ""])
|
|
out = Path(out_dir) / f"{doc_id}.xml"
|
|
out.parent.mkdir(parents=True, exist_ok=True)
|
|
out.write_text(xml, encoding="utf-8")
|
|
print("->", out)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main(sys.argv[1], sys.argv[2] if len(sys.argv) > 2 else "data")
|