1
0
Files
Budget-as-Code/tools/rd.py
Rihards Gailums 71250e7b0e Valsts budžets kā kods v0.1: shēma, 2025. un 2026. gada budžeta likumi kā dati
Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation),
nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi.
Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML.
Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
2026-10-10 22:34:28 +00:00

56 lines
2.4 KiB
Python

"""Shared readers for budget-law annex files (xlsx via openpyxl, xls via xlrd, docx via stdlib)."""
import zipfile, re, xml.etree.ElementTree as ET
def sheets(path):
"""Yield (sheet_name, rows) with rows as lists of stripped strings/numbers; hidden BEx sheets included."""
if path.lower().endswith('.xlsx'):
import openpyxl
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
for ws in wb.worksheets:
rows = []
for r in ws.iter_rows(values_only=True):
rows.append([("" if v is None else (v.strip() if isinstance(v, str) else v)) for v in r])
yield ws.title, rows
elif path.lower().endswith('.xls'):
import xlrd
wb = xlrd.open_workbook(path)
for ws in wb.sheets():
rows = []
for i in range(ws.nrows):
rows.append([(c.strip() if isinstance(c, str) else c) for c in ws.row_values(i)])
yield ws.name, rows
def main_sheet(path):
"""The printed annex sheet: the one whose name ends with 'piel' or is the first non-BEx sheet with most rows."""
best = None
for name, rows in sheets(path):
if name.lower().endswith('piel') or re.match(r'^\d+\.?\s*piel', name.lower()):
return name, rows
if name.startswith(('BEx', 'HEADER', 'FOOTER', 'ZQZ', 'var', 'list', 'parms')):
continue
if best is None or len(rows) > len(best[1]):
best = (name, rows)
return best
W = '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}'
def docx_tables(path):
"""List of tables, each a list of rows of cell texts."""
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
out = []
for t in x.iter(W + 'tbl'):
rows = []
for tr in t.findall(W + 'tr'):
rows.append([''.join(n.text or '' for n in tc.iter(W + 't')).strip() for tc in tr.findall(W + 'tc')])
out.append(rows)
return out
def docx_paras(path):
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
return [''.join(n.text or '' for n in p.iter(W + 't')).strip() for p in x.iter(W + 'p')]
def num(v):
"""Parse an amount cell: numbers, '1 234 567', '-1 234', '–'. Returns float or None."""
if isinstance(v, (int, float)): return float(v)
s = str(v).replace(' ', ' ').replace(' ', '').replace('–', '-').replace('−', '-').replace(',', '.')
return float(s) if re.fullmatch(r'-?\d+(\.\d+)?', s) else None