1
0

MI stratēģijas datukopa v0.1: shēma, XML (pppa-mis-2026-09-10), rīki

- schemas/mi-strategija-0.1.xsd: Stratēģija → Pīlārs → Mērķis → Rīcība
- data/pppa-mis-2026-09-10.xml: 3 pīlāri, 5 mērķi, 27 rīcības (burtiski no v2.2)
- tools/mi_strategija_no_html.py, tools/validate.py

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
2026-10-09 22:20:39 +00:00
parent b437ff53e6
commit 7a5448113c
6 changed files with 829 additions and 0 deletions

View File

@@ -0,0 +1,186 @@
#!/usr/bin/env python3
"""PPPA MI stratēģijas lapa (pppa.lv/mi-strategija) -> mašīnlasāms XML.
Lietojums:
python3 tools/mi_strategija_no_html.py source/mi-strategija-v2.2.html data/
Teksts tiek pārņemts burtiski; HTML formatējums (treknraksts, saites) tiek
noņemts, saišu adreses saglabātas atribūtā @href.
"""
import html
import re
import sys
from pathlib import Path
from xml.sax.saxutils import escape, quoteattr
NS = "urn:pppa:cac:mi-strategija:0.1"
ROMAN = {"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5}
def text(fragment):
"""HTML fragments -> tīrs teksts vienā rindā."""
s = re.sub(r"<svg.*?</svg>", "", fragment, flags=re.S)
s = re.sub(r"<br\s*/?>", " ", s)
s = re.sub(r"<[^>]+>", "", s)
s = html.unescape(s)
return re.sub(r"\s+", " ", s).strip()
def one(pattern, s, default=None):
m = re.search(pattern, s, re.S)
if not m:
if default is not None:
return default
raise ValueError("Nav atrasts: " + pattern)
return m.group(1)
def absurl(u):
return "https://pppa.lv" + u if u.startswith("/") else u
def date_iso(lv):
m = re.fullmatch(r"(\d{2})\.(\d{2})\.(\d{4})\.?", lv.strip())
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
def el(name, txt, indent, **attrs):
a = "".join(f" {k}={quoteattr(str(v))}" for k, v in attrs.items() if v not in (None, ""))
return f"{' ' * indent}<{name}{a}>{escape(txt)}</{name}>"
def main(src, out_dir):
h = Path(src).read_text(encoding="utf-8")
h = re.sub(r"<script.*?</script>|<style.*?</style>", "", h, flags=re.S)
# --- Galvene ---
hero = one(r'<section class="strat-hero">(.*?)</section>', h)
title = text(one(r"<h1>(.*?)</h1>", hero))
status = text(one(r'<p class="eyebrow">(.*?)</p>', hero))
slogan = [text(x) for x in re.split(r'<span class="hs-sep">.*?</span>',
one(r'<p class="hero-slogan">(.*?)</p>', hero))]
vision = text(one(r'<p class="lead">(.*?)</p>', hero))
summary = one(r'<div class="summary-box" id="kopsavilkums">(.*?)</div>', h)
upd = one(r"Pēdējoreiz atjaunots:\s*([\d.]+)", summary).rstrip(".")
version = one(r"Darba versija\s*(v[\d.]+)", summary)
intake = one(r"Priekšlikumu saņemšana uzsākta <strong[^>]*>([\d.]+)</strong>", summary).rstrip(".")
doc_id = f"pppa-mis-{date_iso(upd)}"
# --- Pīlāru kartītes (kopsavilkums) ---
cards = {}
for c in re.finditer(r'<button type="button" class="pillar-card[^"]*" data-tab="(\d)">(.*?)</button>', h, re.S):
cards[int(c.group(1))] = {
"name": text(one(r'<span class="pc-jumbo">(.*?)</span>', c.group(2))),
"short": text(one(r'<span class="pc-sub">(.*?)</span>', c.group(2))),
"summary": text(one(r'<span class="pc-text">(.*?)</span>', c.group(2))),
}
# --- Pīlāri, mērķi, rīcības ---
panels = one(r'<div class="pillar-panels">(.*?)</div><!-- /pillar-panels -->', h)
pillars = re.split(r'<section class="pillar pillar-panel[^"]*"', panels)[1:]
xml_p, refs_all, action_ids = [], [], set()
for p in pillars:
rn = one(r'<span class="pillar-rn">(\w+)</span>', p)
pn = ROMAN[rn]
pid = f"{doc_id}.p{pn}"
head = one(r'<div class="pillar-head">(.*?)</div>', p)
out = [f' <Pillar id="{pid}" number="{rn}">',
el("Name", cards[pn]["name"], 4),
el("ShortTitle", cards[pn]["short"], 4),
el("Heading", text(one(r"<h2>(.*?)</h2>", head)), 4),
el("Summary", cards[pn]["summary"], 4),
el("Description", text(one(r"</h2>\s*<p>(.*?)</p>", head)), 4)]
for g in re.split(r'<div class="goal-block">', p)[1:]:
gn = int(one(r'<span class="goal-num">(\d+)\.</span>', g))
gid = f"{doc_id}.m{gn}"
out += [f' <Goal id="{gid}" number="{gn}.">',
el("Title", text(one(r"<h3>(.*?)</h3>", g)), 6),
el("Description", text(one(r"</h3>\s*<p>(.*?)</p>", g)), 6)]
for a in re.finditer(r'<div class="action-card" id="pasakums-(\d+)">(.*?)\n\s*</div>\s*(?=<div class="action-card"|</div>\s*</div>)', g, re.S):
an, body = int(a.group(1)), a.group(2)
aid = f"{doc_id}.r{an:02d}"
action_ids.add(an)
printed = text(one(r'<span class="ac-num">(.*?)</span>', body))
assert printed == f"Rīcība {an}", (printed, an)
dl = text(one(r'<span class="deadline-badge">(.*?)</span>', re.sub(r"<svg.*?</svg>", "", body, flags=re.S)))
out.append(f' <Action id="{aid}" number="{an}">')
for t in re.finditer(r'<span class="ac-tag( rec)?">(.*?)</span>', body):
out.append(el("Tag", text(t.group(2)), 8,
kind="parvaldibas-rekomendacija" if t.group(1) else None))
out += [el("Title", text(one(r"<h4>(.*?)</h4>", body)), 8),
el("Text", text(one(r'<p class="ac-body">(.*?)</p>', body)), 8),
el("Measurement", text(re.sub(r"<strong>.*?</strong>", "",
one(r'<div class="ac-measure">(.*?)</div>', body), count=1)), 8),
el("Deadline", dl, 8, value=date_iso(dl))]
for r in re.findall(r'<p class="ac-ref">(.*?)</p>', body, re.S):
t = text(r)
kind = {"Saistīts ar:": "saistits", "Avots:": "avots", "Piemērs:": "piemers"}
k = next((v for p_, v in kind.items() if t.startswith(p_)), None)
if k:
t = t.split(":", 1)[1].strip()
hrefs = " ".join(absurl(u) for u in re.findall(r'href="([^"]+)"', r))
nums = re.findall(r"Rīcība (\d+)", t)
refs_all.append((an, nums))
acts = " ".join(f"{doc_id}.r{int(n):02d}" for n in nums)
out.append(el("Reference", t, 8, kind=k, actions=acts, href=hrefs))
out.append(" </Action>")
out.append(" </Goal>")
out.append(" </Pillar>")
xml_p += out
# --- Principi ---
pr_sec = one(r'id="principi">(.*?)<hr class="section-divider">', h)
pr_intro = text(one(r'<p class="section-intro">(.*?)</p>', pr_sec))
xml_pr = [' <Principles>', el("Heading", text(one(r"^(.*?)</h2>", pr_sec)), 4),
el("Intro", pr_intro, 4)]
for i, b in enumerate(re.findall(r'<div class="mg-box">(.*?)</div>', pr_sec, re.S), 1):
xml_pr += [f' <Principle id="{doc_id}.pr{i}">',
el("Label", text(one(r'<p class="mg-label">(.*?)</p>', b)), 6),
el("Text", text(one(r'</p>\s*<p>(.*?)</p>', b)), 6),
" </Principle>"]
xml_pr.append(" </Principles>")
# --- Saņemtie priekšlikumi ---
pk = one(r'id="priekslikumi">.*?<p class="section-intro">(.*?)</p>', h)
pk_nums = sorted({int(n) for n in re.findall(r"\((?:Rīcība )?(\d+)\)", text(pk))})
xml_pk = [el("ProposalsNote", text(pk), 2,
actions=" ".join(f"{doc_id}.r{n:02d}" for n in pk_nums))]
# --- Starptautiskā pieredze ---
xml_rs = [" <Resources>"]
for i, c in enumerate(re.findall(r'<div class="resource-card">(.*?)</div>', h, re.S), 1):
xml_rs += [f' <Resource id="{doc_id}.a{i}" href={quoteattr(one(r'href="([^"]+)"', c))}>',
el("Country", text(one(r'<p class="rc-country">(.*?)</p>', c)), 6),
el("Title", text(one(r"<h3>(.*?)</h3>", c)), 6),
el("Text", text(one(r"</h3>\s*<p>(.*?)</p>", c)), 6),
" </Resource>"]
xml_rs.append(" </Resources>")
# --- Pārbaudes ---
expect = set(range(1, max(action_ids) + 1))
missing = expect - action_ids
bad = [(a, n) for a, ns in refs_all for n in ns if int(n) not in action_ids]
print(f"{doc_id}: {len(pillars)} pīlāri, {len(action_ids)} rīcības; trūkst {sorted(missing) or '—'}; "
f"nederīgas atsauces {bad or '—'}")
head = [
'<?xml version="1.0" encoding="UTF-8"?>',
f'<AIStrategy xmlns="{NS}" id="{doc_id}" version="{version}">',
el("Title", title, 2),
el("Author", "PPP Asociācija (PPPA)", 2),
el("Status", status, 2),
el("Updated", upd + ".", 2, value=date_iso(upd)),
el("ProposalsOpened", intake + ".", 2, value=date_iso(intake)),
el("SourceUrl", "https://pppa.lv/mi-strategija", 2),
" <Slogan>" + "".join(el("Word", w, 0) for w in slogan) + "</Slogan>",
el("Vision", vision, 2),
]
xml = "\n".join(head + xml_p + xml_pr + xml_pk + xml_rs + ["</AIStrategy>", ""])
out = Path(out_dir) / f"{doc_id}.xml"
out.parent.mkdir(parents=True, exist_ok=True)
out.write_text(xml, encoding="utf-8")
print("->", out)
if __name__ == "__main__":
main(sys.argv[1], sys.argv[2] if len(sys.argv) > 2 else "data")

60
tools/validate.py Normal file
View File

@@ -0,0 +1,60 @@
#!/usr/bin/env python3
"""Pārbauda MI stratēģijas datukopu: XSD, faila nosaukums = ID, numerācija.
Lietojums: python3 tools/validate.py [data/]
Prasības: pip install lxml
"""
import sys
from pathlib import Path
from lxml import etree
ROOT = Path(__file__).resolve().parent.parent
XSD = etree.XMLSchema(etree.parse(str(ROOT / "schemas" / "mi-strategija-0.1.xsd")))
NS = {"s": "urn:pppa:cac:mi-strategija:0.1"}
def check(path):
errs = []
doc = etree.parse(str(path))
if not XSD.validate(doc):
errs += [f"XSD {e.line}: {e.message}" for e in XSD.error_log]
return errs
r = doc.getroot()
did = r.get("id")
if path.stem != did:
errs.append(f"faila nosaukums {path.name} ≠ ID {did}.xml")
if r.find("s:Updated", NS).get("value") != did[-10:]:
errs.append("ID datums nesakrīt ar Updated/@value")
for p in r.iterfind("s:Pillar", NS):
if not p.get("id").startswith(did + ".p"):
errs.append(f"pīlāra ID {p.get('id')}")
nums = []
for a in r.iterfind(".//s:Action", NS):
n = int(a.get("number"))
nums.append(n)
if a.get("id") != f"{did}.r{n:02d}":
errs.append(f"rīcības {n} ID {a.get('id')} neatbilst numuram")
if sorted(nums) != list(range(1, len(nums) + 1)):
errs.append(f"rīcību numerācija nav 1..{len(nums)}: {sorted(nums)}")
goals = [int(g.get("number").rstrip(".")) for g in r.iterfind(".//s:Goal", NS)]
if goals != list(range(1, len(goals) + 1)):
errs.append(f"mērķu numerācija {goals}")
nores = [a.get("number") for a in r.iterfind(".//s:Action", NS) if a.find("s:Responsible", NS) is None]
print(f" {path.name}: {len(list(r.iterfind('s:Pillar', NS)))} pīlāri, {len(goals)} mērķi, "
f"{len(nums)} rīcības; bez atbildīgās institūcijas: {len(nores)}")
return errs
def main(folder):
bad = 0
for f in sorted(Path(folder).glob("pppa-mis-*.xml")):
e = check(f)
for x in e:
print(" KĻŪDA", f.name, x)
bad += bool(e)
print("OK" if not bad else f"{bad} faili ar kļūdām")
return 1 if bad else 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1] if len(sys.argv) > 1 else ROOT / "data"))