#!/usr/bin/env python3 """PPPA MI stratēģijas lapa (pppa.lv/mi-strategija) -> mašīnlasāms XML. Lietojums: python3 tools/mi_strategija_no_html.py source/mi-strategija-v2.2.html data/ Teksts tiek pārņemts burtiski; HTML formatējums (treknraksts, saites) tiek noņemts, saišu adreses saglabātas atribūtā @href. """ import html import re import sys from pathlib import Path from xml.sax.saxutils import escape, quoteattr NS = "urn:pppa:cac:mi-strategija:0.1" ROMAN = {"I": 1, "II": 2, "III": 3, "IV": 4, "V": 5} def text(fragment): """HTML fragments -> tīrs teksts vienā rindā.""" s = re.sub(r"", "", fragment, flags=re.S) s = re.sub(r"", " ", s) s = re.sub(r"<[^>]+>", "", s) s = html.unescape(s) return re.sub(r"\s+", " ", s).strip() def one(pattern, s, default=None): m = re.search(pattern, s, re.S) if not m: if default is not None: return default raise ValueError("Nav atrasts: " + pattern) return m.group(1) def absurl(u): return "https://pppa.lv" + u if u.startswith("/") else u def date_iso(lv): m = re.fullmatch(r"(\d{2})\.(\d{2})\.(\d{4})\.?", lv.strip()) return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None def el(name, txt, indent, **attrs): a = "".join(f" {k}={quoteattr(str(v))}" for k, v in attrs.items() if v not in (None, "")) return f"{' ' * indent}<{name}{a}>{escape(txt)}" def main(src, out_dir): h = Path(src).read_text(encoding="utf-8") h = re.sub(r"|", "", h, flags=re.S) # --- Galvene --- hero = one(r'
(.*?)
', h) title = text(one(r"

(.*?)

", hero)) status = text(one(r'

(.*?)

', hero)) slogan = [text(x) for x in re.split(r'.*?', one(r'

(.*?)

', hero))] vision = text(one(r'

(.*?)

', hero)) summary = one(r'
(.*?)
', h) upd = one(r"Pēdējoreiz atjaunots:\s*([\d.]+)", summary).rstrip(".") version = one(r"Darba versija\s*(v[\d.]+)", summary) intake = one(r"Priekšlikumu saņemšana uzsākta ]*>([\d.]+)", summary).rstrip(".") doc_id = f"pppa-mis-{date_iso(upd)}" # --- Pīlāru kartītes (kopsavilkums) --- cards = {} for c in re.finditer(r'', h, re.S): cards[int(c.group(1))] = { "name": text(one(r'(.*?)', c.group(2))), "short": text(one(r'(.*?)', c.group(2))), "summary": text(one(r'(.*?)', c.group(2))), } # --- Pīlāri, mērķi, rīcības --- panels = one(r'
(.*?)
', h) pillars = re.split(r'
(\w+)', p) pn = ROMAN[rn] pid = f"{doc_id}.p{pn}" head = one(r'
(.*?)
', p) out = [f' ', el("Name", cards[pn]["name"], 4), el("ShortTitle", cards[pn]["short"], 4), el("Heading", text(one(r"

(.*?)

", head)), 4), el("Summary", cards[pn]["summary"], 4), el("Description", text(one(r"\s*

(.*?)

", head)), 4)] for g in re.split(r'
', p)[1:]: gn = int(one(r'(\d+)\.', g)) gid = f"{doc_id}.m{gn}" out += [f' ', el("Title", text(one(r"

(.*?)

", g)), 6), el("Description", text(one(r"\s*

(.*?)

", g)), 6)] for a in re.finditer(r'
(.*?)\n\s*
\s*(?=
\s*
)', g, re.S): an, body = int(a.group(1)), a.group(2) aid = f"{doc_id}.r{an:02d}" action_ids.add(an) printed = text(one(r'(.*?)', body)) assert printed == f"Rīcība {an}", (printed, an) dl = text(one(r'(.*?)', re.sub(r"", "", body, flags=re.S))) out.append(f' ') for t in re.finditer(r'(.*?)', body): out.append(el("Tag", text(t.group(2)), 8, kind="parvaldibas-rekomendacija" if t.group(1) else None)) out += [el("Title", text(one(r"

(.*?)

", body)), 8), el("Text", text(one(r'

(.*?)

', body)), 8), el("Measurement", text(re.sub(r".*?", "", one(r'
(.*?)
', body), count=1)), 8), el("Deadline", dl, 8, value=date_iso(dl))] for r in re.findall(r'

(.*?)

', body, re.S): t = text(r) kind = {"Saistīts ar:": "saistits", "Avots:": "avots", "Piemērs:": "piemers"} k = next((v for p_, v in kind.items() if t.startswith(p_)), None) if k: t = t.split(":", 1)[1].strip() hrefs = " ".join(absurl(u) for u in re.findall(r'href="([^"]+)"', r)) nums = re.findall(r"Rīcība (\d+)", t) refs_all.append((an, nums)) acts = " ".join(f"{doc_id}.r{int(n):02d}" for n in nums) out.append(el("Reference", t, 8, kind=k, actions=acts, href=hrefs)) out.append("
") out.append("
") out.append(" ") xml_p += out # --- Principi --- pr_sec = one(r'id="principi">(.*?)
', h) pr_intro = text(one(r'

(.*?)

', pr_sec)) xml_pr = [' ', el("Heading", text(one(r"^(.*?)", pr_sec)), 4), el("Intro", pr_intro, 4)] for i, b in enumerate(re.findall(r'
(.*?)
', pr_sec, re.S), 1): xml_pr += [f' ', el("Label", text(one(r'

(.*?)

', b)), 6), el("Text", text(one(r'

\s*

(.*?)

', b)), 6), "
"] xml_pr.append("
") # --- Saņemtie priekšlikumi --- pk = one(r'id="priekslikumi">.*?

(.*?)

', h) pk_nums = sorted({int(n) for n in re.findall(r"\((?:Rīcība )?(\d+)\)", text(pk))}) xml_pk = [el("ProposalsNote", text(pk), 2, actions=" ".join(f"{doc_id}.r{n:02d}" for n in pk_nums))] # --- Starptautiskā pieredze --- xml_rs = [" "] for i, c in enumerate(re.findall(r'
(.*?)
', h, re.S), 1): xml_rs += [f' ', el("Country", text(one(r'

(.*?)

', c)), 6), el("Title", text(one(r"

(.*?)

", c)), 6), el("Text", text(one(r"\s*

(.*?)

", c)), 6), "
"] xml_rs.append("
") # --- Pārbaudes --- expect = set(range(1, max(action_ids) + 1)) missing = expect - action_ids bad = [(a, n) for a, ns in refs_all for n in ns if int(n) not in action_ids] print(f"{doc_id}: {len(pillars)} pīlāri, {len(action_ids)} rīcības; trūkst {sorted(missing) or '—'}; " f"nederīgas atsauces {bad or '—'}") head = [ '', f'', el("Title", title, 2), el("Author", "PPP Asociācija (PPPA)", 2), el("Status", status, 2), el("Updated", upd + ".", 2, value=date_iso(upd)), el("ProposalsOpened", intake + ".", 2, value=date_iso(intake)), el("SourceUrl", "https://pppa.lv/mi-strategija", 2), " " + "".join(el("Word", w, 0) for w in slogan) + "", el("Vision", vision, 2), ] xml = "\n".join(head + xml_p + xml_pr + xml_pk + xml_rs + ["", ""]) out = Path(out_dir) / f"{doc_id}.xml" out.parent.mkdir(parents=True, exist_ok=True) out.write_text(xml, encoding="utf-8") print("->", out) if __name__ == "__main__": main(sys.argv[1], sys.argv[2] if len(sys.argv) > 2 else "data")