Valsts budžets kā kods v0.1: shēma, 2025. un 2026. gada budžeta likumi kā dati
Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation), nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi. Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML. Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
90
tools/calc.py
Normal file
90
tools/calc.py
Normal file
@@ -0,0 +1,90 @@
|
||||
"""Shared summation rules for the budget model (used by the parser for residuals and by the verifier).
|
||||
|
||||
A printed number in the law is reproduced by summing Allocation records that match its selector:
|
||||
fund, flow, year, nature, untilEnd, purpose subtree, block, commitment kind, and classification codes.
|
||||
Consolidation: a node's total includes an intra-fund transfer only if that node itself prints the transfer's code;
|
||||
otherwise intra-fund transfers are eliminated (they cancel out inside the node).
|
||||
"""
|
||||
import collections
|
||||
|
||||
|
||||
def kind_stem(k):
|
||||
"""Commitment-kind codes are hierarchical in 2-digit groups: 0101100000 -> 010110."""
|
||||
s = k or ''
|
||||
while len(s) > 2 and s.endswith('00'):
|
||||
s = s[:-2]
|
||||
return s
|
||||
|
||||
|
||||
class Calc:
|
||||
def __init__(self, allocs, purpose_parent, cls_parent, intra_codes, blockcodes, purpose_holder=None):
|
||||
self.purpose_parent = purpose_parent
|
||||
self.purpose_holder = purpose_holder or {}
|
||||
self.cls_parent = cls_parent
|
||||
self.intra = intra_codes
|
||||
self.blockcodes = {k: {f: set(v) for f, v in d.items()} for k, d in blockcodes.items()}
|
||||
self.idx = collections.defaultdict(list)
|
||||
for a in allocs:
|
||||
self.add(a)
|
||||
|
||||
def chain(self, p):
|
||||
out = set()
|
||||
while p and p not in out:
|
||||
out.add(p)
|
||||
p = self.purpose_parent.get(p)
|
||||
return out
|
||||
|
||||
def code_chain(self, k):
|
||||
out = set()
|
||||
while k and k not in out:
|
||||
out.add(k)
|
||||
k = self.cls_parent.get(k)
|
||||
return out
|
||||
|
||||
def add(self, a):
|
||||
a['_pchain'] = self.chain(a.get('purpose'))
|
||||
a['_ck'] = f"{a['scheme']}:{a['code']}"
|
||||
a['_cchain'] = self.code_chain(a['_ck'])
|
||||
a['_intra'] = bool(a['_cchain'] & self.intra)
|
||||
self.idx[(a['fund'], a['flow'], int(a['year']), a['nature'])].append(a)
|
||||
|
||||
def total(self, fund, flow, year, nature, purpose=None, block=None, codes=None, ckind=None, untilEnd=False,
|
||||
blk=None, ekk_prefix=None, consolidated=None, holder_view=False):
|
||||
"""holder_view: also count budget-level amounts (no purpose) whose holder administers the selected purpose."""
|
||||
printed = self.blockcodes.get(blk, {}).get(flow) if blk else None
|
||||
if consolidated is None:
|
||||
consolidated = purpose is None
|
||||
codes = set(codes) if codes else None
|
||||
s = 0.0
|
||||
for f in ([fund] if fund else ['basic', 'special']):
|
||||
for a in self.idx.get((f, flow, int(year), nature), ()):
|
||||
if a.get('partOf'):
|
||||
continue
|
||||
if bool(a.get('untilEnd')) != bool(untilEnd):
|
||||
continue
|
||||
if purpose and purpose not in a['_pchain']:
|
||||
if not (holder_view and not a.get('purpose') and a.get('holder') and a.get('holder') == self.purpose_holder.get(purpose)):
|
||||
continue
|
||||
if block and a.get('block') != block:
|
||||
continue
|
||||
if ckind and not (a.get('commitmentKind') or '').startswith(kind_stem(ckind)):
|
||||
continue
|
||||
if ekk_prefix and not (a['scheme'] == 'ekk' and a['code'].startswith(ekk_prefix)):
|
||||
continue
|
||||
ck, cc = a['_ck'], a['_cchain']
|
||||
if printed is not None:
|
||||
shown = ck in printed
|
||||
if not shown and a['_intra']:
|
||||
continue
|
||||
if codes is not None:
|
||||
if not (ck in codes or (not shown and cc & codes)):
|
||||
continue
|
||||
elif codes is not None:
|
||||
if not (cc & codes):
|
||||
continue
|
||||
if consolidated and a['_intra'] and ck not in codes:
|
||||
continue
|
||||
elif consolidated and a['_intra']:
|
||||
continue
|
||||
s += float(a['amount'])
|
||||
return s
|
||||
26
tools/klasifikacijas.py
Normal file
26
tools/klasifikacijas.py
Normal file
@@ -0,0 +1,26 @@
|
||||
"""Classification code lists from the regulation tables on likumi.lv:
|
||||
MK 27.12.2005. noteikumi Nr. 1031 (expenditure, EKK), MK noteikumi par budžetu ieņēmumu klasifikāciju (revenue), MK 22.11.2005. noteikumi Nr. 875 (financing).
|
||||
Input: source/klasifikacijas/<likumi.lv id>.html Output: source/kodi/klasdict.json (name -> [[scheme, code], ...])
|
||||
"""
|
||||
import re, html, json, os
|
||||
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
out = {}
|
||||
CODE = re.compile(r'(?:[A-Z]{1,2})?\d{1,2}(?:\.\d{1,2}){1,5}\.?|\d{4,5}|F\d{8}|[A-Z]\d{1,2}(?:\.\d+)*\.?')
|
||||
for act, kind in (('124833', 'ekk'), ('124831', 'revenue'), ('122159', 'financing')):
|
||||
s = open(os.path.join(ROOT, 'source', 'klasifikacijas', f'{act}.html'), encoding='utf-8', errors='replace').read()
|
||||
n = 0
|
||||
for tr in re.findall(r'(?is)<tr[^>]*>(.*?)</tr>', s):
|
||||
cells = [re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', c))).strip() for c in re.findall(r'(?is)<td[^>]*>(.*?)</td>', tr)]
|
||||
for i, c in enumerate(cells[:-1]):
|
||||
if CODE.fullmatch(c) and cells[i + 1] and not CODE.fullmatch(cells[i + 1]) and not cells[i + 1].startswith('Kodā'):
|
||||
code = c.rstrip('.')
|
||||
if kind == 'revenue' and '.' in code:
|
||||
p = code.split('.'); code = (p[0].zfill(2) + ''.join(p[1:])).ljust(5, '0')
|
||||
if kind == 'financing' and '.' in code:
|
||||
code = 'F' + code.replace('.', '')
|
||||
out.setdefault(cells[i + 1].strip(' .;:'), set()).add((kind, code)); n += 1
|
||||
break
|
||||
print(act, kind, 'rows', n)
|
||||
for k in [k for k in out if re.search(r'Dotācija no vispār|savstarpējie|Kapitālo izdevumu transferti$|7200|uz valsts pamatbudžetu', k)][:12]:
|
||||
print(' ', sorted(out[k])[:3], k[:100])
|
||||
json.dump({k: sorted(v) for k, v in out.items()}, open(os.path.join(ROOT, 'source', 'kodi', 'klasdict.json'), 'w'), ensure_ascii=False)
|
||||
23
tools/kodi_no_sap.py
Normal file
23
tools/kodi_no_sap.py
Normal file
@@ -0,0 +1,23 @@
|
||||
"""Code-name pairs from the hidden SAP BW (BEx) sheets inside the budget law annex files.
|
||||
The printed annex tables show line names only; the hidden ZQZBC_* sheets carry the code next to the name.
|
||||
Input: source/<year>/P01.XLSX, P02.XLSX, P03.XLSX, P05.XLSX, P11.XLSX Output: source/kodi/codepairs.json (name -> [codes])
|
||||
Usage: python tools/kodi_no_sap.py [year] (default 2026)
|
||||
"""
|
||||
import sys, os, re, json, collections
|
||||
TOOLS = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
|
||||
sys.path.insert(0, TOOLS)
|
||||
from rd import sheets
|
||||
year = sys.argv[1] if len(sys.argv) > 1 else '2026'
|
||||
pairs = collections.OrderedDict()
|
||||
for f in ('P01.XLSX', 'P02.XLSX', 'P05.XLSX', 'P03.XLSX', 'P11.XLSX'):
|
||||
for name, rows in sheets(os.path.join(ROOT, 'source', year, f)):
|
||||
if not name.startswith('ZQZ'):
|
||||
continue
|
||||
for r in rows:
|
||||
for i in range(len(r) - 1):
|
||||
c, nm = str(r[i]), r[i + 1]
|
||||
if re.fullmatch(r'[A-Z]{0,3}\d[\dA-Z]*|[A-Z]{1,5}\d?T?|\d{4,5}', c) and isinstance(nm, str) and len(nm) > 3 and not nm.startswith('Nosaukums'):
|
||||
pairs.setdefault(nm.strip(), set()).add(c)
|
||||
json.dump({k: sorted(v) for k, v in pairs.items()}, open(os.path.join(ROOT, 'source', 'kodi', 'codepairs.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=0)
|
||||
print('pairs', len(pairs))
|
||||
32
tools/odcs-config.yaml
Normal file
32
tools/odcs-config.yaml
Normal file
@@ -0,0 +1,32 @@
|
||||
# B15 datu līguma konfigurācija lakehouse rīkam tools/xsd_to_odcs.py (tas pats rīks, ar ko ģenerēti B01–B14).
|
||||
# python3 tools/xsd_to_odcs.py <šī datne> <repozitorija sakne> contracts Budget-as-Code
|
||||
base_url: https://processgit.org
|
||||
contracts:
|
||||
- id: B15
|
||||
file: B15-valsts-budzets.odcs.yaml
|
||||
name: Valsts budžets (budžeta likums kā dati)
|
||||
version: 0.1.0
|
||||
org: Valsts-Pirmkods
|
||||
repo: Budget-as-Code
|
||||
pattern: data/lv-vb-*.xml
|
||||
filename_is_id: true
|
||||
schema: schemas/valsts-budzets-0.1.xsd
|
||||
root: Budget
|
||||
namespace: urn:pppa:vpk:budzets:0.1
|
||||
owner: PPP Asociācija (PPPA)
|
||||
data_product: Valsts budžeta datukopa
|
||||
purpose: Pieņemtais valsts budžeta likums kā dati — panti, ieņēmumi, apropriācijas pa programmām un apakšprogrammām, nākamo gadu maksimālie apjomi, ilgtermiņa saistības un mērķdotācijas pašvaldībām; katrs likuma pielikumos drukātais skaitlis atjaunojams no datnes.
|
||||
limitations: Koncepcijas demonstrācija, nav oficiāls izdevums. Budžeta paskaidrojumi (mērķi, rezultatīvie rādītāji) un gada laikā veiktās apropriācijas izmaiņas vēl nav iekļautas.
|
||||
identifiers: "lv-vb-GGGG; pants .pNNN; resors .ac.rSS; programma .pr.SS.PP.AA.00; projekts .pj.…; naudas fakts .a.pNN.r<rinda>.y<gads>"
|
||||
objects:
|
||||
- {element: Budget, keys: [id]}
|
||||
- {element: Source, keys: [id]}
|
||||
- {element: Provision, keys: [id]}
|
||||
- {element: Parameter, keys: [id]}
|
||||
- {element: Actor, keys: [id]}
|
||||
- {element: Purpose, keys: [id]}
|
||||
- {element: ClassItem, keys: [scheme, code]}
|
||||
- {element: Allocation, keys: [id]}
|
||||
rules:
|
||||
- {rule: printed_numbers_reproduced, description: "Katrs likuma pielikumos drukātais skaitlis atjaunojams no datnes (tools/verify_budget.py)", dimension: accuracy}
|
||||
- {rule: resort_has_vpk_id, description: "Resoram, kas ir institūcija, norādīts VPK ID"}
|
||||
862
tools/parse_budget.py
Normal file
862
tools/parse_budget.py
Normal file
@@ -0,0 +1,862 @@
|
||||
"""Convert an adopted Latvian state budget law (text + 12 annexes) into one XML file
|
||||
following valsts-budzets-0.1.xsd, plus a JSON list of every printed number as a check.
|
||||
|
||||
Usage: python parse_budget.py YEAR LAW_HTML ANNEX_DIR OUT_DIR
|
||||
Annex files are found by number (P04.XLSX, 4_PIELIKUMS.XLS, ...).
|
||||
Storage rule: each fact is stored once, from its most detailed source; every other printed number becomes a check.
|
||||
"""
|
||||
import sys, os, re, json, html, hashlib, collections, datetime
|
||||
TOOLS = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
|
||||
KODI = os.path.join(ROOT, 'source', 'kodi')
|
||||
sys.path.insert(0, TOOLS)
|
||||
from rows import read_rows
|
||||
from rd import docx_tables, num
|
||||
from xml.sax.saxutils import quoteattr, escape
|
||||
from calc import Calc
|
||||
|
||||
YEAR, LAW_HTML, ADIR, OUT = int(sys.argv[1]), sys.argv[2], sys.argv[3], sys.argv[4]
|
||||
Y3 = [YEAR, YEAR + 1, YEAR + 2]
|
||||
BID = f'lv-vb-{YEAR}'
|
||||
report = collections.defaultdict(list)
|
||||
|
||||
# ------------------------------------------------------------------ helpers
|
||||
def norm(s):
|
||||
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
|
||||
s = re.sub(r'^i estādes', 'iestādes', s)
|
||||
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
|
||||
|
||||
def slug(s, n=40):
|
||||
t = norm(s).translate(str.maketrans('āčēģīķļņšūž', 'acegiklnsuz'))
|
||||
return re.sub(r'\s+', '-', t)[:n].strip('-')
|
||||
|
||||
def annex_file(n):
|
||||
for f in os.listdir(ADIR):
|
||||
m = re.match(r'^P?0*(\d+)[_.]', f, re.I)
|
||||
if m and int(m.group(1)) == n:
|
||||
return os.path.join(ADIR, f)
|
||||
raise FileNotFoundError(f'annex {n}')
|
||||
|
||||
def sha(path):
|
||||
return hashlib.sha256(open(path, 'rb').read()).hexdigest()
|
||||
|
||||
# ------------------------------------------------------------------ code book
|
||||
SCHEME_OF_FLOW = {'expenditure': 'ekk', 'resource': 'revenue', 'revenue': 'revenue', 'financing': 'financing'}
|
||||
book = collections.defaultdict(lambda: collections.defaultdict(set)) # scheme -> normname -> codes
|
||||
for nm, kc in json.load(open(os.path.join(KODI, 'klasdict.json'), encoding='utf-8')).items():
|
||||
for kind, c in kc:
|
||||
book[kind][norm(nm)].add(c)
|
||||
KL = {'Izdevumi': 'ekk', 'Ieņēmumi': 'revenue', 'Finansēšana': 'financing'}
|
||||
for line in open(os.path.join(KODI, 'vk_codes.tsv'), encoding='utf-8'):
|
||||
kl, v = line.rstrip('\n').split('\t')
|
||||
m = re.match(r'^([A-Z]{0,2}\d[\d.]*)\s+(.+)$', v)
|
||||
if m and kl in KL:
|
||||
book[KL[kl]][norm(m.group(2))].add(m.group(1))
|
||||
for nm, codes in json.load(open(os.path.join(KODI, 'codepairs.json'), encoding='utf-8')).items():
|
||||
for c in codes:
|
||||
sch = 'financing' if c.startswith('F') or re.fullmatch(r'[A-Z]{1,2}F\d+', c) else ('ekk' if re.fullmatch(r'\d{4}', c) else ('revenue' if re.fullmatch(r'\d{5}', c) else 'law'))
|
||||
book[sch][norm(nm)].add(c)
|
||||
|
||||
classitems = {} # (scheme, code) -> {'name', 'parent'}
|
||||
SECTION = {'ieņēmumi kopā': 'revenue', 'resursi izdevumu segšanai': 'resource', 'izdevumi kopā': 'expenditure',
|
||||
'finansēšana': 'financing', 'finansiālā bilance': 'balance'}
|
||||
|
||||
def canon(sch, c):
|
||||
if sch == 'financing' and re.fullmatch(r'[PKS]F\d{8}', c):
|
||||
return c[1:]
|
||||
return c
|
||||
|
||||
def code_for(label, flow):
|
||||
n = norm(label)
|
||||
pref = SCHEME_OF_FLOW.get(flow, 'law')
|
||||
for sch in (pref, 'law', 'revenue', 'ekk', 'financing'):
|
||||
cs = book[sch].get(n)
|
||||
if cs:
|
||||
cs = {canon(sch, x) for x in cs}
|
||||
c = sorted(cs, key=lambda x: (len(x), x))[0]
|
||||
if len(cs) > 1:
|
||||
report['ambiguous_code'].append(f'{label[:60]} -> {sorted(cs)} (took {c})')
|
||||
return sch, c
|
||||
c = 'L-' + slug(label)
|
||||
report['synthesised_code'].append(label)
|
||||
return 'law', c
|
||||
|
||||
INTRA = ('savstarpējie transferti', 'no valsts pamatbudžeta uz valsts pamatbudžetu', 'no valsts speciālā budžeta uz valsts speciālo budžetu',
|
||||
'atmaksām valsts pamatbudžet', 'atmaksa valsts budžetā par veiktajiem', 'valsts pamatbudžeta iestāžu saņemtie transferti no valsts pamatbudžeta',
|
||||
'pārējie valsts pamatbudžetā saņemtie transferti no valsts pamatbudžeta', 'valsts speciālā budžeta iestāžu saņemtie transferti no valsts speciālā budžeta')
|
||||
def is_intra(name):
|
||||
n = norm(name)
|
||||
return any(norm(k) in n for k in INTRA)
|
||||
|
||||
VOTE_WEIGHT = {'a2': 1, 'a3': 1, 'a4': 1, 'a5': 1, 'a11': 0}
|
||||
CUR_ANNEX = ['a4']
|
||||
def register_class(sch, code, name, parent):
|
||||
k = (sch, code)
|
||||
d = classitems.setdefault(k, {'name': name, 'parent': None, 'votes': collections.Counter()})
|
||||
w = VOTE_WEIGHT.get(CUR_ANNEX[0], 1)
|
||||
if w:
|
||||
d['votes'][parent] += w
|
||||
|
||||
def structural_ok(child, parent):
|
||||
"""Numeric classification codes carry their own hierarchy: 21200 cannot sit under 21100, 7131 can sit under 7130."""
|
||||
cs, cc = child.split(':', 1)
|
||||
ps, pc = parent.split(':', 1)
|
||||
if not (cc.isdigit() and pc.isdigit() and cs == ps and len(cc) == len(pc)):
|
||||
return True
|
||||
stem = pc.rstrip('0')
|
||||
return cc != pc and cc.startswith(stem)
|
||||
|
||||
def finalize_classes():
|
||||
"""Canonical parent = majority vote over indented annex blocks, restricted to structurally possible parents."""
|
||||
for k, d in classitems.items():
|
||||
votes = d['votes']
|
||||
ck = f'{k[0]}:{k[1]}'
|
||||
real = [(p, n) for p, n in votes.most_common() if p and structural_ok(ck, p)]
|
||||
rejected = [p for p in votes if p and not structural_ok(ck, p)]
|
||||
if rejected:
|
||||
report['parent_rejected_by_structure'].append(f'{ck} {d["name"][:40]}: not under {rejected}')
|
||||
d['parent'] = real[0][0] if real and (not votes.get(None) or real[0][1] >= votes[None] or rejected) else None
|
||||
if len(real) > 1:
|
||||
report['class_parent_votes'].append(f'{k[0]}:{k[1]} {d["name"][:50]} votes {dict(votes)} -> {d["parent"]}')
|
||||
|
||||
# ------------------------------------------------------------------ actors and purposes
|
||||
actors, purposes = {}, {}
|
||||
VPK_RESORTS = {l.split('\t')[0] for l in open(os.path.join(KODI, 'vpk_resorts.tsv'), encoding='utf-8') if l.strip()}
|
||||
def resort_actor(code, name):
|
||||
aid = f'{BID}.ac.r{code}'
|
||||
vpk = f'{code}-0000' if f'{code}-0000' in VPK_RESORTS else None
|
||||
actors.setdefault(aid, {'kind': 'resort', 'code': code, 'name': name, 'vpk': vpk})
|
||||
pid = f'{BID}.pr.{code}'
|
||||
purposes.setdefault(pid, {'kind': 'resort', 'code': code, 'name': name, 'holder': aid})
|
||||
return aid, pid
|
||||
|
||||
def prog_purpose(rcode, pcode, name, function, parent_pid):
|
||||
pid = f'{BID}.pr.{rcode}.{pcode}'
|
||||
kind = 'programme' if pcode.endswith('.00.00') else 'subprogramme'
|
||||
d = purposes.setdefault(pid, {'kind': kind, 'code': pcode, 'name': name, 'parent': parent_pid,
|
||||
'holder': f'{BID}.ac.r{rcode}', 'function': function})
|
||||
if function and not d.get('function'):
|
||||
d['function'] = function
|
||||
return pid
|
||||
|
||||
def project_purpose(rcode, pcode, projcode, name, parent_pid):
|
||||
pid = f'{BID}.pj.{rcode}.{pcode or "x"}.{slug(projcode, 60)}'
|
||||
purposes.setdefault(pid, {'kind': 'project', 'code': projcode, 'name': name, 'parent': parent_pid,
|
||||
'holder': f'{BID}.ac.r{rcode}'})
|
||||
return pid
|
||||
|
||||
def municipality(name):
|
||||
aid = f'{BID}.ac.m-{slug(name)}'
|
||||
actors.setdefault(aid, {'kind': 'municipality', 'name': name})
|
||||
return aid
|
||||
|
||||
# block (core/eu) per subprogramme, from Valsts kase execution data; fallback by code range
|
||||
blockmap = {}
|
||||
for line in open(os.path.join(KODI, 'vk_blocks.tsv'), encoding='utf-8'):
|
||||
g, m, p, sp, blk, bt, n = line.rstrip('\n').split('\t')
|
||||
if int(g) in (YEAR, YEAR - 1):
|
||||
blockmap.setdefault((m, sp), 'eu' if blk.startswith('Ārvalstu') else 'core')
|
||||
def block_of(rcode, pcode):
|
||||
b = blockmap.get((rcode, pcode))
|
||||
if b:
|
||||
return b
|
||||
report['block_fallback'].append(f'{rcode} {pcode}')
|
||||
return 'eu' if pcode[:2] in ('60', '61', '62', '63', '64', '65', '66', '67', '68', '69', '70', '71', '72', '73', '74', '75', '76', '77', '78', '79', '80', '81', '82', '83', '84', '85') else 'core'
|
||||
|
||||
# ------------------------------------------------------------------ outputs
|
||||
allocs, checks = [], []
|
||||
blockcodes = collections.defaultdict(lambda: collections.defaultdict(set)) # block id -> flow -> printed codes
|
||||
def register_block(bid, lines):
|
||||
for ln in lines:
|
||||
if not ln.get('skip') and not ln.get('section') and ln.get('flow') not in (None, 'balance'):
|
||||
blockcodes[bid][ln['flow']].add(f"{ln['scheme']}:{ln['code']}")
|
||||
_alloc_ids = collections.Counter()
|
||||
def add_alloc(**a):
|
||||
"""Allocation id = source annex + source row + year (+ 'l' for 'later years' column, + -N if the same cell yields several facts).
|
||||
Derived from the source, so re-parsing the same law gives the same ids."""
|
||||
base = f'{BID}.a.{a.pop("srcAnnex")}.r{a.get("srcRow", 0)}.y{a["year"]}{"l" if a.get("untilEnd") else ""}'
|
||||
_alloc_ids[base] += 1
|
||||
a['id'] = base if _alloc_ids[base] == 1 else f'{base}-{_alloc_ids[base]}'
|
||||
allocs.append(a)
|
||||
return a['id']
|
||||
|
||||
def add_check(annex, row, value, **sel):
|
||||
checks.append({'annex': annex, 'row': row, 'value': value, 'sel': sel})
|
||||
|
||||
pending = [] # (bid, flow, ck, alloc kwargs) for lines of detailed blocks; leafness decided on the canonical tree
|
||||
def queue_alloc(_bid, _flow, _ck, **kw):
|
||||
pending.append((_bid, _flow, _ck, kw))
|
||||
|
||||
def canon_chain(ck, parent):
|
||||
out, k = [], ck
|
||||
while k and k not in out:
|
||||
out.append(k)
|
||||
k = parent.get(k)
|
||||
return out
|
||||
|
||||
def finalize_lines():
|
||||
parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
|
||||
below = {} # (bid, flow, ck) -> printed codes in that block whose canonical chain contains ck
|
||||
for bid, fl in blockcodes.items():
|
||||
for flow, codes in fl.items():
|
||||
for c in codes:
|
||||
for anc in canon_chain(c, parent):
|
||||
if anc in codes:
|
||||
below.setdefault((bid, flow, anc), []).append(c)
|
||||
for c in checks:
|
||||
r = c['sel'].get('codes')
|
||||
if isinstance(r, dict) and 'resolve' in r:
|
||||
bid, flow, ck = r['resolve']
|
||||
c['sel']['codes'] = sorted(set(below.get((bid, flow, ck), [ck])) | {ck})
|
||||
for bid, flow, ck, kw in pending:
|
||||
if len(set(below.get((bid, flow, ck), [ck])) - {ck}) == 0:
|
||||
add_alloc(**kw)
|
||||
|
||||
# ------------------------------------------------------------------ generic tree-annex reader
|
||||
def tree_blocks(rows, label_col, amount_cols, header_fn, level_fn):
|
||||
"""Split rows into blocks keyed by header path; each block has lines with level, flow, amounts."""
|
||||
blocks, path, cur = [], {}, None
|
||||
for r in rows:
|
||||
cells = r['cells']
|
||||
h = header_fn(r, path)
|
||||
if h:
|
||||
path = h
|
||||
cur = {'path': dict(path), 'lines': []}
|
||||
blocks.append(cur)
|
||||
continue
|
||||
label = str(cells[label_col]) if len(cells) > label_col else ''
|
||||
amts = {}
|
||||
for y, ci in amount_cols.items():
|
||||
v = num(cells[ci]) if ci < len(cells) else None
|
||||
if v is not None:
|
||||
amts[y] = v
|
||||
if not label or not amts:
|
||||
continue
|
||||
if cur is None:
|
||||
cur = {'path': dict(path), 'lines': []}
|
||||
blocks.append(cur)
|
||||
cur['lines'].append({'row': r['row'], 'label': label, 'level': level_fn(r, label), 'amounts': amts})
|
||||
return blocks
|
||||
|
||||
def resolve_lines(block):
|
||||
"""Assign flow, code and parent code to every line; mark leaves (no deeper line follows in the same section)."""
|
||||
lines, flow, stack = block['lines'], None, []
|
||||
for i, ln in enumerate(lines):
|
||||
n = norm(ln['label'])
|
||||
if n in SECTION:
|
||||
flow = SECTION[n]
|
||||
ln.update(flow=flow, section=True, scheme='law', code='S-' + flow, leaf=False)
|
||||
stack = [(ln['level'], ln)]
|
||||
continue
|
||||
if flow is None:
|
||||
ln.update(flow=None, skip=True)
|
||||
continue
|
||||
while stack and stack[-1][0] >= ln['level']:
|
||||
stack.pop()
|
||||
parent = stack[-1][1] if stack else None
|
||||
sch, code = code_for(ln['label'], flow)
|
||||
ln.update(flow=flow, scheme=sch, code=code, section=False)
|
||||
if parent is not None and not parent.get('section'):
|
||||
register_class(sch, code, ln['label'], f'{parent["scheme"]}:{parent["code"]}')
|
||||
else:
|
||||
register_class(sch, code, ln['label'], None)
|
||||
stack.append((ln['level'], ln))
|
||||
for i, ln in enumerate(lines):
|
||||
if ln.get('skip'):
|
||||
continue
|
||||
desc = [] if ln.get('section') else [f"{ln['scheme']}:{ln['code']}"]
|
||||
for x in lines[i + 1:]:
|
||||
if x.get('skip'):
|
||||
continue
|
||||
if x.get('section') or x['level'] <= ln['level'] or x.get('flow') != ln.get('flow'):
|
||||
break
|
||||
desc.append(f"{x['scheme']}:{x['code']}")
|
||||
ln['codes'] = desc
|
||||
for i, ln in enumerate(lines):
|
||||
if ln.get('section') or ln.get('skip'):
|
||||
continue
|
||||
nxt = next((x for x in lines[i + 1:] if not x.get('skip')), None)
|
||||
ln['leaf'] = nxt is None or nxt.get('section') or nxt['level'] <= ln['level'] or nxt.get('flow') != ln['flow']
|
||||
return lines
|
||||
|
||||
# ------------------------------------------------------------------ annex 4 and 5 (2026 appropriations by subprogramme)
|
||||
label_levels = collections.defaultdict(collections.Counter) # (flow, normlabel) -> relative level counts, learned for annex 11
|
||||
|
||||
def parse_programmes(n, fund):
|
||||
CUR_ANNEX[0] = f'a{n}'
|
||||
path_ = annex_file(n)
|
||||
rows = read_rows(path_, label_col=2)
|
||||
progs_seen = {}
|
||||
def header(r, path):
|
||||
c = r['cells']
|
||||
lab = str(c[2]) if len(c) > 2 else ''
|
||||
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
||||
if m and num(c[3] if len(c) > 3 else '') is None:
|
||||
aid, pid = resort_actor(m.group(1), m.group(2))
|
||||
return {'resort': m.group(1), 'rpid': pid}
|
||||
code = str(c[0])
|
||||
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
|
||||
rc = path['resort']
|
||||
parent = path['rpid']
|
||||
pcode = code
|
||||
if not code.endswith('.00.00'):
|
||||
pp = code[:2] + '.00.00'
|
||||
if (rc, pp) in progs_seen:
|
||||
parent = progs_seen[(rc, pp)]
|
||||
func = str(c[1]) if re.fullmatch(r'\d{2}\.\d{3}', str(c[1])) else None
|
||||
pid = prog_purpose(rc, pcode, lab, func, parent)
|
||||
progs_seen[(rc, pcode)] = pid
|
||||
return {'resort': rc, 'rpid': path['rpid'], 'pcode': pcode, 'pid': pid, 'parent': parent, 'func': func}
|
||||
return None
|
||||
blocks = tree_blocks(rows, 2, {YEAR: 3}, header, lambda r, lab: r['indent'])
|
||||
# leaf blocks: subprogrammes, or programmes without subprogrammes
|
||||
has_child = set(b['path'].get('parent') for b in blocks if b['path'].get('pcode'))
|
||||
for bi, b in enumerate(blocks):
|
||||
lines = resolve_lines(b)
|
||||
bid = f'a{n}:{bi}'
|
||||
register_block(bid, lines)
|
||||
p = b['path']
|
||||
if 'pid' in p:
|
||||
sel_purpose = p['pid']
|
||||
elif 'rpid' in p:
|
||||
sel_purpose = p['rpid']
|
||||
else:
|
||||
sel_purpose = None
|
||||
leafblock = 'pid' in p and p['pid'] not in has_child
|
||||
blk = block_of(p['resort'], p['pcode']) if 'pcode' in p else None
|
||||
# learn relative levels for annex 11
|
||||
base = None
|
||||
for ln in lines:
|
||||
if ln.get('section'):
|
||||
base = ln['level']
|
||||
elif not ln.get('skip') and base is not None:
|
||||
label_levels[(ln['flow'], norm(ln['label']))][ln['level'] - base] += 1
|
||||
for ln in lines:
|
||||
if ln.get('skip') or ln['flow'] == 'balance':
|
||||
if ln.get('flow') == 'balance' or ln.get('section') and ln['flow'] == 'balance':
|
||||
add_check(n, ln['row'], ln['amounts'][YEAR], kind='balance', fund=fund, year=YEAR, purpose=sel_purpose, nature='appropriation', blk=bid)
|
||||
continue
|
||||
nature = 'forecast' if ln['flow'] == 'revenue' and fund == 'basic' else 'appropriation'
|
||||
if fund == 'special' and ln['flow'] == 'revenue':
|
||||
nature = 'forecast'
|
||||
add_check(n, ln['row'], ln['amounts'][YEAR], fund=fund, flow=ln['flow'], year=YEAR, purpose=sel_purpose,
|
||||
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, nature=nature, blk=bid)
|
||||
store = leafblock and not ln.get('section')
|
||||
if fund == 'basic' and ln['flow'] == 'revenue':
|
||||
store = False # basic-budget revenue is stored from annex 2
|
||||
if store:
|
||||
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex=f'p{n:02d}', flow=ln['flow'], fund=fund, nature=nature, year=YEAR,
|
||||
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['pid'], block=blk, function=p.get('func'),
|
||||
scheme=ln['scheme'], code=ln['code'], amount=ln['amounts'][YEAR],
|
||||
src=f'{BID}.src.p{n:02d}', srcRow=ln['row'])
|
||||
|
||||
# ------------------------------------------------------------------ annex 2 (revenue forecasts, 3 years)
|
||||
def parse_revenue():
|
||||
CUR_ANNEX[0] = 'a2'
|
||||
raw = read_rows(annex_file(2), label_col=0)
|
||||
# the label column differs between years (2025 .xls has an extra empty first column): find it from the header row
|
||||
lc = next((i for r in raw for i, c in enumerate(r['cells']) if str(c).startswith('Ieņēmumu avots')), 0)
|
||||
rows = read_rows(annex_file(2), label_col=lc) if lc else raw
|
||||
part, fund, mode = None, 'basic', 'tree'
|
||||
lines, fees, info = [], [], []
|
||||
cur_resort = None
|
||||
for r in rows:
|
||||
c = r['cells']
|
||||
lab = str(c[lc]) if lc < len(c) else ''
|
||||
if re.match(r'^I\.\s', lab):
|
||||
fund, mode = 'basic', 'tree'; continue
|
||||
if re.match(r'^II\.\s', lab):
|
||||
fund, mode = 'special', 'tree'; continue
|
||||
if lab.startswith('Ministrija (cita'):
|
||||
mode = 'fees'; continue
|
||||
if lab.startswith('Informatīvi'):
|
||||
mode = 'info'; continue
|
||||
amts = {y: num(c[lc + 1 + i]) for i, y in enumerate(Y3) if lc + 1 + i < len(c) and num(c[lc + 1 + i]) is not None}
|
||||
if not amts or not lab:
|
||||
continue
|
||||
if mode == 'tree':
|
||||
lines.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'amounts': amts, 'fund': fund})
|
||||
elif mode == 'fees':
|
||||
fees.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'bold': r['bold'], 'amounts': amts})
|
||||
else:
|
||||
info.append({'row': r['row'], 'label': lab, 'amounts': amts})
|
||||
stored = {}
|
||||
for fund in ('basic', 'special'):
|
||||
blk = {'lines': [dict(l) for l in lines if l['fund'] == fund]}
|
||||
resolve_lines(blk)
|
||||
register_block(f'a2:{fund}', blk['lines'])
|
||||
for ln in blk['lines']:
|
||||
if ln.get('skip'):
|
||||
continue
|
||||
for y, v in ln['amounts'].items():
|
||||
add_check(2, ln['row'], v, fund=fund, flow='revenue', year=y, nature='forecast', label=ln['label'],
|
||||
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=ln.get('codes'), blk=f'a2:{fund}')
|
||||
if ln.get('leaf') and not ln.get('section'):
|
||||
if fund == 'special' and y == YEAR:
|
||||
continue # 2026 special revenue is stored from annex 5 (by subprogramme)
|
||||
aid = add_alloc(srcAnnex='p02', flow='revenue', fund=fund, nature='forecast', year=y,
|
||||
holder=None, purpose=None, scheme=ln['scheme'], code=ln['code'], amount=v,
|
||||
src=f'{BID}.src.p02', srcRow=ln['row'])
|
||||
stored[(fund, ln['code'], y)] = aid
|
||||
# fees by administering resort: a breakdown of stored revenue leaves -> partOf
|
||||
leaf_codes = sorted({k[1] for k in stored if k[0] == 'basic'}, key=len, reverse=True)
|
||||
for f in fees:
|
||||
n_ = norm(f['label'])
|
||||
if n_.startswith('ieņēmumi valsts pamatbudžetā kopā'):
|
||||
for y, v in f['amounts'].items():
|
||||
add_check(2, f['row'], v, kind='fees_total', year=y)
|
||||
continue
|
||||
if f['bold']:
|
||||
name = f['label']
|
||||
m = [a for a, d in actors.items() if d['kind'] == 'resort' and norm(d['name']) == n_]
|
||||
cur_resort = m[0] if m else None
|
||||
if not cur_resort:
|
||||
report['fee_resort_unmatched'].append(f['label'])
|
||||
for y, v in f['amounts'].items():
|
||||
add_check(2, f['row'], v, kind='fees_resort', holder=cur_resort, year=y)
|
||||
continue
|
||||
sch, code = code_for(f['label'], 'revenue')
|
||||
register_class(sch, code, f['label'], None)
|
||||
parent = next((stored[('basic', lc, y0)] for lc in leaf_codes for y0 in [YEAR] if code.startswith(lc.rstrip('0') or lc) and ('basic', lc, YEAR) in stored), None)
|
||||
if parent is None:
|
||||
report['fee_parent_unmatched'].append(f'{code} {f["label"][:60]}')
|
||||
for y, v in f['amounts'].items():
|
||||
par = None
|
||||
if parent:
|
||||
pc = next(k for k, a in stored.items() if a == parent)[1]
|
||||
par = stored.get(('basic', pc, y))
|
||||
add_alloc(srcAnnex='p02', flow='revenue', fund='basic', nature='forecast', year=y, holder=cur_resort,
|
||||
purpose=None, scheme=sch, code=code, amount=v, partOf=par, src=f'{BID}.src.p02', srcRow=f['row'])
|
||||
for i in info:
|
||||
for y, v in i['amounts'].items():
|
||||
params.append({'id': f'{BID}.par.p02-{i["row"]}-{y}', 'name': i['label'], 'year': y, 'value': v, 'unit': 'EUR',
|
||||
'src': f'{BID}.src.p02'})
|
||||
|
||||
# ------------------------------------------------------------------ annex 3 (resort summaries, 3 years; 2027-28 stored as ceilings)
|
||||
def parse_resort_summary():
|
||||
CUR_ANNEX[0] = 'a3'
|
||||
rows = read_rows(annex_file(3), label_col=0)
|
||||
fund_names = {'valsts pamatbudžets': 'basic', 'valsts speciālais budžets': 'special'}
|
||||
def header(r, path):
|
||||
lab = str(r['cells'][0])
|
||||
n_ = norm(lab)
|
||||
if num(r['cells'][1] if len(r['cells']) > 1 else '') is not None:
|
||||
return None
|
||||
if n_ in fund_names:
|
||||
p = {k: v for k, v in path.items() if k in ('resort', 'rpid')}
|
||||
p['fund'] = fund_names[n_]
|
||||
return p
|
||||
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
||||
if m:
|
||||
aid, pid = resort_actor(m.group(1), m.group(2))
|
||||
return {'resort': m.group(1), 'rpid': pid, 'fund': 'basic'}
|
||||
if re.match(r'^I\.\s+Valsts pamatfunkciju', lab):
|
||||
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'core'; return p
|
||||
if re.match(r'^II\.\s+ES politiku', lab):
|
||||
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'eu'; return p
|
||||
return None
|
||||
blocks = tree_blocks(rows, 0, {Y3[0]: 1, Y3[1]: 2, Y3[2]: 3}, header, lambda r, lab: r['indent'])
|
||||
keys = [(b['path'].get('resort'), b['path'].get('fund')) for b in blocks]
|
||||
for bi, b in enumerate(blocks):
|
||||
lines = resolve_lines(b)
|
||||
bid = f'a3:{bi}'
|
||||
register_block(bid, lines)
|
||||
p = b['path']
|
||||
fund = p.get('fund', 'basic')
|
||||
# a resort's special-budget part may appear without its own fund header: detect by section content later
|
||||
leafblock = 'resort' in p and 'block' in p
|
||||
sel = dict(fund=fund, purpose=p.get('rpid'), block=p.get('block'), blk=bid, holderView=True)
|
||||
for ln in lines:
|
||||
if ln.get('skip'):
|
||||
continue
|
||||
for y, v in ln['amounts'].items():
|
||||
if ln['flow'] == 'balance':
|
||||
add_check(3, ln['row'], v, kind='balance', year=y, nature='appropriation' if y == YEAR else 'ceiling', **sel)
|
||||
continue
|
||||
nature = 'forecast' if ln['flow'] == 'revenue' else ('appropriation' if y == YEAR else 'ceiling')
|
||||
add_check(3, ln['row'], v, flow=ln['flow'], year=y, nature=nature,
|
||||
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
|
||||
if leafblock and not ln.get('section') and y != YEAR and ln['flow'] != 'revenue':
|
||||
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p03', flow=ln['flow'], fund=fund, nature='ceiling', year=y,
|
||||
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['rpid'], block=p['block'],
|
||||
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p03', srcRow=ln['row'])
|
||||
|
||||
# ------------------------------------------------------------------ annex 11 (long-term commitments)
|
||||
def parse_commitments():
|
||||
CUR_ANNEX[0] = 'a11'
|
||||
rows = read_rows(annex_file(11), label_col=1)
|
||||
hdr = next(r for r in rows if 'Projekta kods' in [str(x) for x in r['cells']])
|
||||
years = {}
|
||||
for i, c in enumerate(hdr['cells']):
|
||||
m = re.match(r'^(\d{4})\.', str(c))
|
||||
if m:
|
||||
years[int(m.group(1))] = i
|
||||
elif str(c).startswith('Tālākā'):
|
||||
years['later'] = i
|
||||
last = max(y for y in years if y != 'later')
|
||||
kinds = {}
|
||||
def level_fn(r, lab):
|
||||
cnt = None
|
||||
for fl in ('expenditure', 'resource', 'financing', 'revenue'):
|
||||
c = label_levels.get((fl, norm(lab)))
|
||||
if c:
|
||||
cnt = c.most_common(1)[0][0] + 1
|
||||
break
|
||||
if cnt is None and norm(lab) not in SECTION:
|
||||
report['a11_level_unknown'].append(lab)
|
||||
return 99
|
||||
return cnt or 0
|
||||
progs_seen = {}
|
||||
state = {'prev_header': None}
|
||||
def header(r, path):
|
||||
h = _header(r, path)
|
||||
state['prev_header'] = h is not None and h.get('_kindable', False)
|
||||
if h is not None:
|
||||
h.pop('_kindable', None)
|
||||
return h
|
||||
def _header(r, path):
|
||||
c = r['cells']
|
||||
lab = str(c[1]) if len(c) > 1 else ''
|
||||
if any(num(c[i]) is not None for i in years.values() if i < len(c)):
|
||||
return None
|
||||
if lab.upper().startswith('VALSTS PAMATBUDŽETS'):
|
||||
return {'fund': 'basic'}
|
||||
if lab.upper().startswith('VALSTS SPECIĀLAIS BUDŽETS'):
|
||||
return {'fund': 'special'}
|
||||
m = re.match(r'^(\d{10})\s+(.+)$', lab)
|
||||
if m:
|
||||
kinds[m.group(1)] = m.group(2)
|
||||
register_class('commitment', m.group(1), m.group(2), None)
|
||||
if state['prev_header']: # directly under a resort or programme header
|
||||
p = dict(path); p['kind'] = m.group(1)
|
||||
if 'pcode' not in p:
|
||||
p['rkind'] = m.group(1)
|
||||
return p
|
||||
return {'fund': path.get('fund', 'basic'), 'kind': m.group(1), 'topkind': m.group(1)} # new top-level commitment-kind section
|
||||
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
||||
if m:
|
||||
aid, pid = resort_actor(m.group(1), m.group(2))
|
||||
return {'fund': path.get('fund', 'basic'), 'resort': m.group(1), 'rpid': pid, '_kindable': True,
|
||||
'topkind': path.get('topkind'), 'kind': path.get('topkind')}
|
||||
code = str(c[0])
|
||||
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
|
||||
rc = path['resort']
|
||||
pid = f'{BID}.pr.{rc}.{code}'
|
||||
if pid not in purposes:
|
||||
parent = purposes.get(f'{BID}.pr.{rc}.{code[:2]}.00.00') and f'{BID}.pr.{rc}.{code[:2]}.00.00' or path['rpid']
|
||||
prog_purpose(rc, code, lab, None, parent)
|
||||
report['a11_new_programme'].append(f'{rc} {code} {lab[:50]}')
|
||||
p = {k: v for k, v in path.items() if k in ('fund', 'resort', 'rpid', 'topkind', 'rkind')}
|
||||
p.update(pcode=code, pid=pid, _kindable=True, kind=path.get('rkind') or path.get('topkind'))
|
||||
return p
|
||||
proj = str(c[2]) if len(c) > 2 else ''
|
||||
if proj and 'resort' in path:
|
||||
parent = path.get('pid') or path['rpid']
|
||||
pj = project_purpose(path['resort'], path.get('pcode'), proj, lab, parent)
|
||||
p = dict(path); p['project'] = proj; p['pjid'] = pj
|
||||
return p
|
||||
return None
|
||||
amount_cols = {y: i for y, i in years.items()}
|
||||
blocks = tree_blocks(rows, 1, amount_cols, header, level_fn)
|
||||
|
||||
def more_specific(k2, k): # commitment kind k2 is a sub-type of k (or equal); codes nest in 2-digit groups
|
||||
from calc import kind_stem
|
||||
if not k:
|
||||
return True
|
||||
return bool(k2 and k2.startswith(kind_stem(k)))
|
||||
def refined(bi):
|
||||
"""A detailed block is refined (not a leaf) if, within the same resort+programme, a later or earlier block adds
|
||||
a project under a compatible kind, or shows the same project under a more specific kind."""
|
||||
p = blocks[bi]['path']
|
||||
for j, b2 in enumerate(blocks):
|
||||
if j == bi:
|
||||
continue
|
||||
q = b2['path']
|
||||
if q.get('resort') != p.get('resort') or q.get('fund') != p.get('fund') or not q.get('pcode'):
|
||||
continue
|
||||
if q['pcode'] != p.get('pcode'):
|
||||
# a programme block is refined by its subprogramme blocks (same resort, same programme number)
|
||||
if (p.get('pcode') or '').endswith('.00.00') and q['pcode'][:2] == p['pcode'][:2] and not q['pcode'].endswith('.00.00') \
|
||||
and not p.get('project') and more_specific(q.get('kind'), p.get('kind')):
|
||||
return True
|
||||
continue
|
||||
if not p.get('project') and q.get('project') and more_specific(q.get('kind'), p.get('kind')):
|
||||
return True
|
||||
if p.get('project') and q.get('project') == p['project'] and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
|
||||
return True
|
||||
if not p.get('project') and not q.get('project') and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
|
||||
return True
|
||||
return False
|
||||
# leaf blocks: those whose path is not extended by a later block
|
||||
def key(p):
|
||||
return (p.get('fund'), p.get('kind'), p.get('resort'), p.get('pcode'), p.get('project'))
|
||||
keys = [key(b['path']) for b in blocks]
|
||||
def extends(a, b): # b is strictly more specific than a
|
||||
return a != b and all(x is None or x == y or (i == 1 and x and y and y.startswith(x.rstrip('0'))) for i, (x, y) in enumerate(zip(a, b)))
|
||||
for bi, b in enumerate(blocks):
|
||||
lines = resolve_lines(b)
|
||||
bid = f'a11:{bi}'
|
||||
register_block(bid, lines)
|
||||
p = b['path']
|
||||
leafblock = bool(p.get('pcode')) and not refined(bi)
|
||||
sel = dict(fund=p.get('fund', 'basic'), ckind=p.get('kind'), purpose=p.get('pjid') or p.get('pid') or p.get('rpid'), blk=bid)
|
||||
for ln in lines:
|
||||
if ln.get('skip'):
|
||||
continue
|
||||
for y, v in ln['amounts'].items():
|
||||
yy, until = (last + 1, True) if y == 'later' else (y, False)
|
||||
if ln['flow'] == 'balance':
|
||||
add_check(11, ln['row'], v, kind='balance', year=yy, untilEnd=until, nature='commitment', **sel)
|
||||
continue
|
||||
add_check(11, ln['row'], v, flow=ln['flow'], year=yy, untilEnd=until, nature='commitment',
|
||||
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
|
||||
if leafblock and not ln.get('section') and v != 0:
|
||||
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p11', flow=ln['flow'], fund=sel['fund'], nature='commitment', year=yy, untilEnd=until,
|
||||
holder=f'{BID}.ac.r{p["resort"]}' if p.get('resort') else None,
|
||||
purpose=sel['purpose'], commitmentKind=p.get('kind'),
|
||||
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p11', srcRow=ln['row'])
|
||||
|
||||
# ------------------------------------------------------------------ annexes 6-10 (earmarked grants to municipalities)
|
||||
def parse_grants(n):
|
||||
path_ = annex_file(n)
|
||||
t = docx_tables(path_)[-1]
|
||||
title = next((r[0] for r in t if r and len(r[0]) > 40), f'{n}. pielikums')
|
||||
gid = f'{BID}.pr.g{n:02d}'
|
||||
purposes[gid] = {'kind': 'grant', 'code': f'P{n:02d}', 'name': title}
|
||||
period = (f'{YEAR}-01-01', f'{YEAR}-12-31')
|
||||
cols = None
|
||||
for ri, r in enumerate(t):
|
||||
first = r[0] if r else ''
|
||||
m = re.match(r'^(I{1,2})\.\s', first)
|
||||
if m:
|
||||
period = (f'{YEAR}-01-01', f'{YEAR}-08-31') if m.group(1) == 'I' else (f'{YEAR}-09-01', f'{YEAR}-12-31')
|
||||
continue
|
||||
if first.startswith('Pašvaldības'):
|
||||
cols = r; continue
|
||||
vals = [num(x) for x in r[1:]]
|
||||
if not first or not any(v is not None for v in vals):
|
||||
continue
|
||||
n_ = norm(first)
|
||||
if n_ in ('kopā', 'pavisam kopā'):
|
||||
heads_ = [norm(h) for h in (cols or [])[1:]]
|
||||
for ci, v in enumerate(vals):
|
||||
if v is not None:
|
||||
h = heads_[ci] if ci < len(heads_) else ''
|
||||
add_check(n, ri + 1, v, kind='grant_total', purpose=gid, code=f'law:G{n:02d}-{slug(h or "summa", 30) or "summa"}',
|
||||
periodFrom=None if n_ == 'pavisam kopā' else period[0], periodTo=None if n_ == 'pavisam kopā' else period[1], year=YEAR)
|
||||
continue
|
||||
recip = f'{BID}.ac.unallocated' if n_.startswith('nesadalītie') else municipality(first)
|
||||
if recip.endswith('unallocated'):
|
||||
actors[recip] = {'kind': 'unallocated', 'name': 'Nesadalītie līdzekļi'}
|
||||
heads = [norm(h) for h in (cols or [])[1:]]
|
||||
# column roles: 'pavisam kopā' = total; 'tai skaitā …' = part of previous; otherwise the main amount
|
||||
ids = {}
|
||||
total_idx = next((i for i, h in enumerate(heads) if h.startswith('pavisam kopā')), None)
|
||||
main_idx = 0
|
||||
order = ([total_idx] if total_idx is not None else []) + [i for i in range(len(vals)) if i != total_idx]
|
||||
for ci in order:
|
||||
v = vals[ci] if ci < len(vals) else None
|
||||
if v is None:
|
||||
continue
|
||||
h = heads[ci] if ci < len(heads) else ''
|
||||
part = None
|
||||
if h.startswith('tai skaitā'):
|
||||
part = ids.get(ci - 1)
|
||||
elif total_idx is not None and ci != total_idx:
|
||||
part = ids.get(total_idx)
|
||||
aid = add_alloc(srcAnnex=f'p{n:02d}', flow='expenditure', fund='basic', nature='earmarkedGrant', year=YEAR,
|
||||
periodFrom=period[0], periodTo=period[1], recipient=recip, purpose=gid,
|
||||
scheme='law', code=f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', amount=v, partOf=part,
|
||||
src=f'{BID}.src.p{n:02d}', srcRow=ri + 1)
|
||||
register_class('law', f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', (cols[ci + 1] if cols and ci + 1 < len(cols) else 'Summa') or 'Summa', None)
|
||||
ids[ci] = aid
|
||||
|
||||
# ------------------------------------------------------------------ annex 1 (consolidated; checks and GDP parameter)
|
||||
params = []
|
||||
def parse_consolidated():
|
||||
rows = read_rows(annex_file(1), label_col=0)
|
||||
hdr = next(r for r in rows if any(re.match(r'^\d{4}\.', str(c)) for c in r['cells']))
|
||||
ycols = {int(re.match(r'^(\d{4})', str(c)).group(1)): i for i, c in enumerate(hdr['cells']) if re.match(r'^\d{4}\.', str(c))}
|
||||
lc = next(i for i, c in enumerate(hdr['cells']) if str(c) == 'Nosaukums')
|
||||
pct = False
|
||||
for r in rows:
|
||||
c = r['cells']
|
||||
lab = str(c[lc]) if lc < len(c) else ''
|
||||
if lab.startswith('Procentos no IKP'):
|
||||
pct = True; continue
|
||||
for y, i in ycols.items():
|
||||
v = num(c[i]) if i < len(c) else None
|
||||
if v is None or not lab:
|
||||
continue
|
||||
if lab.startswith('IKP milj'):
|
||||
params.append({'id': f'{BID}.par.ikp-{y}', 'name': 'IKP prognoze', 'code': 'IKP', 'year': y, 'value': v,
|
||||
'unit': 'milj. EUR', 'src': f'{BID}.src.p01'})
|
||||
continue
|
||||
add_check(1, r['row'], v, kind='a1', label=lab, indent=r['indent'], pct=pct, year=y)
|
||||
|
||||
# ------------------------------------------------------------------ law text
|
||||
provisions = []
|
||||
def parse_law():
|
||||
s = open(LAW_HTML, encoding='utf-8', errors='replace').read()
|
||||
chapter = None
|
||||
for m in re.finditer(r"<div class='(TV212|TV213)'([^>]*)>(.*?)(?=<div class='TV21[23]'|<div class='TV9|$)", s, re.S):
|
||||
kind, tag, body = m.groups()
|
||||
mm = re.search(r'data-num="(\d+)"', tag)
|
||||
nr = mm.group(1) if mm else None
|
||||
text = body.replace('</p>', '\n').replace('<br />', ' ')
|
||||
text = html.unescape(re.sub(r'<[^>]+>', '', text))
|
||||
text = re.sub(r'[ \t]+', ' ', re.sub(r'\n\s*\n+', '\n', text)).strip()
|
||||
if kind == 'TV212':
|
||||
chapter = re.sub(r'\s+', ' ', text)
|
||||
continue
|
||||
if not nr:
|
||||
continue
|
||||
text = re.sub(r'\n?\d+\s*$', '', text).strip()
|
||||
provisions.append({'id': f'{BID}.p{int(nr):03d}', 'chapter': chapter, 'article': int(nr), 'text': text,
|
||||
'src': f'{BID}.src.law'})
|
||||
|
||||
# ------------------------------------------------------------------ run
|
||||
parse_law()
|
||||
parse_programmes(4, 'basic')
|
||||
parse_programmes(5, 'special')
|
||||
parse_revenue()
|
||||
parse_resort_summary()
|
||||
parse_commitments()
|
||||
for g in (6, 7, 8, 9, 10):
|
||||
parse_grants(g)
|
||||
parse_consolidated()
|
||||
finalize_classes()
|
||||
finalize_lines()
|
||||
|
||||
# special budget revenue 2026 belongs to the resort that holds the special budget; annex 2 2027+ special revenue too
|
||||
special_holders = sorted({a['holder'] for a in allocs if a['fund'] == 'special' and a.get('holder')})
|
||||
for a in allocs:
|
||||
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('holder') and len(special_holders) == 1:
|
||||
a['holder'] = special_holders[0]
|
||||
a['purpose'] = a.get('purpose') or f"{BID}.pr.{special_holders[0].rsplit('.r', 1)[1]}"
|
||||
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('block'):
|
||||
a['block'] = 'core' # the special budget (social insurance) is entirely basic functions
|
||||
|
||||
# ------------------------------------------------------------------ residuals: amounts the law shows only at an aggregate level
|
||||
purpose_parent = {k: v.get('parent') for k, v in purposes.items()}
|
||||
cls_parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
|
||||
intra_codes = {f'{k[0]}:{k[1]}' for k, v in classitems.items() if is_intra(v['name'])}
|
||||
purpose_holder = {k: v.get('holder') for k, v in purposes.items()}
|
||||
calc = Calc(allocs, purpose_parent, cls_parent, intra_codes, blockcodes, purpose_holder)
|
||||
def depth(p):
|
||||
d = 0
|
||||
while p:
|
||||
d += 1
|
||||
p = purpose_parent.get(p)
|
||||
return d
|
||||
cands = []
|
||||
for c in checks:
|
||||
s_ = c['sel']
|
||||
if c['annex'] not in (3, 4, 5, 11) or s_.get('kind') or not s_.get('code') or not s_.get('codes') or len(s_['codes']) != 1 or s_.get('flow') == 'revenue':
|
||||
continue
|
||||
if c['annex'] == 3 and s_['year'] == YEAR:
|
||||
continue
|
||||
cands.append(c)
|
||||
cands.sort(key=lambda c: (-depth(c['sel'].get('purpose')), c['sel'].get('block') is None, c['annex'] in (3,), c['annex']))
|
||||
nres = collections.Counter()
|
||||
for c in cands:
|
||||
s_ = c['sel']
|
||||
got = calc.total(s_.get('fund'), s_['flow'], s_['year'], s_['nature'], purpose=s_.get('purpose'), block=s_.get('block'),
|
||||
codes=s_['codes'], ckind=s_.get('ckind'), untilEnd=s_.get('untilEnd', False), blk=s_.get('blk'))
|
||||
diff = c['value'] - got
|
||||
if abs(diff) < 0.5:
|
||||
continue
|
||||
if not (s_['flow'] == 'financing' or abs(got) < 0.5):
|
||||
report['unexplained'].append(f"annex {c['annex']} row {c['row']} {s_['code']} {s_.get('purpose')} {s_['year']}: printed {c['value']} children {got}")
|
||||
continue
|
||||
sch, code = s_['code'].split(':', 1)
|
||||
if is_intra(classitems.get((sch, code), {}).get('name', '')) and s_.get('purpose') is None:
|
||||
continue
|
||||
pur = s_.get('purpose')
|
||||
holder = (purposes.get(pur) or {}).get('holder') if pur else None
|
||||
a = dict(srcAnnex=f"p{c['annex']:02d}", flow=s_['flow'], fund=s_.get('fund') or 'basic', nature=s_['nature'], year=s_['year'],
|
||||
untilEnd=s_.get('untilEnd', False), holder=holder, purpose=pur, block=s_.get('block'), commitmentKind=s_.get('ckind'),
|
||||
scheme=sch, code=code, amount=diff, src=f"{BID}.src.p{c['annex']:02d}", srcRow=c['row'])
|
||||
add_alloc(**a)
|
||||
calc.add(allocs[-1])
|
||||
nres[c['annex']] += 1
|
||||
splits = [c for c in checks if c['annex'] == 3 and not c['sel'].get('kind') and c['sel']['year'] == YEAR
|
||||
and c['sel'].get('flow') == 'financing' and c['sel'].get('block') and c['sel'].get('code')
|
||||
and len(c['sel'].get('codes') or []) == 1]
|
||||
splits.sort(key=lambda c: -depth(c['sel'].get('purpose')))
|
||||
for c in splits:
|
||||
s_ = c['sel']
|
||||
got = calc.total(s_.get('fund'), 'financing', YEAR, s_['nature'], purpose=s_.get('purpose'), block=s_['block'], codes=s_['codes'],
|
||||
blk=s_.get('blk'))
|
||||
diff = c['value'] - got
|
||||
if abs(diff) < 0.5:
|
||||
continue
|
||||
sch, code = s_['code'].split(':', 1)
|
||||
pur = s_.get('purpose')
|
||||
holder = (purposes.get(pur) or {}).get('holder') if pur else None
|
||||
# annex 3 attributes an amount that annex 4 shows without a block to a block: reclassify within the same purpose (totals unchanged)
|
||||
for pur_, hol_, blk_, amt in ((pur, holder, s_['block'], diff), (pur, holder, None, -diff)):
|
||||
add_alloc(srcAnnex='p03', flow='financing', fund=s_.get('fund') or 'basic', nature=s_['nature'], year=YEAR, holder=hol_,
|
||||
purpose=pur_, block=blk_, scheme=sch, code=code, amount=amt, src=f'{BID}.src.p03', srcRow=c['row'])
|
||||
calc.add(allocs[-1])
|
||||
nres['3-block-split'] += 1
|
||||
report['residual_allocations'] = [f'annex {k}: {v}' for k, v in sorted(nres.items(), key=str)]
|
||||
|
||||
# ------------------------------------------------------------------ sources
|
||||
law_title = re.search(r"<div class='TV207'[^>]*>(.*?)</div>", open(LAW_HTML, encoding='utf-8').read(), re.S)
|
||||
law_title = html.unescape(re.sub(r'<[^>]+>', '', law_title.group(1))).strip() if law_title else f'Par valsts budžetu {YEAR}. gadam'
|
||||
links = json.load(open(os.path.join(ADIR, 'links.json'), encoding='utf-8'))
|
||||
sources = [{'id': f'{BID}.src.law', 'kind': 'lawText', 'title': law_title, 'url': links['law'], 'sha256': sha(LAW_HTML)}]
|
||||
for n in range(1, 13):
|
||||
f = annex_file(n)
|
||||
sources.append({'id': f'{BID}.src.p{n:02d}', 'kind': 'form' if n == 12 else 'annex', 'annex': n,
|
||||
'title': f'{n}. pielikums', 'url': links['annex'][str(n)], 'sha256': sha(f)})
|
||||
|
||||
# ------------------------------------------------------------------ write XML
|
||||
def attrs(d, order):
|
||||
out = []
|
||||
for k in order:
|
||||
v = d.get(k)
|
||||
if v is None or v == '' or v is False:
|
||||
continue
|
||||
if isinstance(v, bool):
|
||||
v = 'true'
|
||||
if isinstance(v, float):
|
||||
v = (f'{v:.2f}'.rstrip('0').rstrip('.')) if v != int(v) else str(int(v))
|
||||
out.append(f'{k}={quoteattr(str(v))}')
|
||||
return ' '.join(out)
|
||||
|
||||
os.makedirs(OUT, exist_ok=True)
|
||||
xp = os.path.join(OUT, f'{BID}.xml')
|
||||
with open(xp, 'w', encoding='utf-8') as fo:
|
||||
fo.write('<?xml version="1.0" encoding="UTF-8"?>\n')
|
||||
fo.write(f'<Budget xmlns="urn:pppa:vpk:budzets:0.1" {attrs({"id": BID, "title": law_title, "year": YEAR, "horizonTo": YEAR + 2, "status": "adopted", "act": links["law"], "published": links.get("published"), "version": links.get("version")}, ["id", "title", "year", "horizonTo", "status", "act", "published", "version"])}>\n')
|
||||
for s_ in sources:
|
||||
fo.write(f' <Source {attrs(s_, ["id", "kind", "annex", "title", "url", "sha256"])}/>\n')
|
||||
for p in provisions:
|
||||
fo.write(f' <Provision {attrs(p, ["id", "chapter", "article", "src"])}>{escape(p["text"])}</Provision>\n')
|
||||
for p in params:
|
||||
fo.write(f' <Parameter {attrs(p, ["id", "name", "code", "year", "value", "unit", "src"])}/>\n')
|
||||
for aid, a in sorted(actors.items()):
|
||||
fo.write(f' <Actor {attrs(dict(a, id=aid), ["id", "kind", "code", "vpk", "name"])}/>\n')
|
||||
for pid, p in purposes.items():
|
||||
fo.write(f' <Purpose {attrs(dict(p, id=pid), ["id", "kind", "code", "name", "parent", "holder", "block", "function"])}/>\n')
|
||||
for (sch, code), c in sorted(classitems.items()):
|
||||
ps, par = (c['parent'].split(':', 1) if c['parent'] else (None, None))
|
||||
fo.write(f' <ClassItem {attrs({"scheme": sch, "code": code, "name": c["name"], "parent": par, "parentScheme": ps if ps != sch else None, "intraFund": is_intra(c["name"]) or None}, ["scheme", "code", "name", "parent", "parentScheme", "intraFund"])}/>\n')
|
||||
for a in allocs:
|
||||
a = {k: v for k, v in a.items() if not k.startswith('_')}
|
||||
fo.write(f' <Allocation {attrs(a, ["id", "flow", "fund", "nature", "year", "untilEnd", "periodFrom", "periodTo", "holder", "recipient", "purpose", "block", "commitmentKind", "function", "scheme", "code", "amount", "partOf", "src", "srcRow"])}/>\n')
|
||||
fo.write('</Budget>\n')
|
||||
|
||||
json.dump({'checks': checks, 'blockcodes': {k: {f: sorted(v) for f, v in d.items()} for k, d in blockcodes.items()}},
|
||||
open(os.path.join(OUT, f'{BID}.checks.json'), 'w', encoding='utf-8'), ensure_ascii=False)
|
||||
summary = {'provisions': len(provisions), 'parameters': len(params), 'actors': len(actors), 'purposes': len(purposes),
|
||||
'classitems': len(classitems), 'allocations': len(allocs), 'checks': len(checks),
|
||||
'allocations_by_annex': collections.Counter(a['src'].rsplit('.', 1)[1] for a in allocs)}
|
||||
print(json.dumps(summary, ensure_ascii=False, default=str))
|
||||
for k, v in report.items():
|
||||
print(f'REPORT {k}: {len(v)}')
|
||||
for x in v[:8]:
|
||||
print(' ', x)
|
||||
json.dump(report, open(os.path.join(OUT, f'{BID}.report.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
|
||||
55
tools/rd.py
Normal file
55
tools/rd.py
Normal file
@@ -0,0 +1,55 @@
|
||||
"""Shared readers for budget-law annex files (xlsx via openpyxl, xls via xlrd, docx via stdlib)."""
|
||||
import zipfile, re, xml.etree.ElementTree as ET
|
||||
|
||||
def sheets(path):
|
||||
"""Yield (sheet_name, rows) with rows as lists of stripped strings/numbers; hidden BEx sheets included."""
|
||||
if path.lower().endswith('.xlsx'):
|
||||
import openpyxl
|
||||
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
|
||||
for ws in wb.worksheets:
|
||||
rows = []
|
||||
for r in ws.iter_rows(values_only=True):
|
||||
rows.append([("" if v is None else (v.strip() if isinstance(v, str) else v)) for v in r])
|
||||
yield ws.title, rows
|
||||
elif path.lower().endswith('.xls'):
|
||||
import xlrd
|
||||
wb = xlrd.open_workbook(path)
|
||||
for ws in wb.sheets():
|
||||
rows = []
|
||||
for i in range(ws.nrows):
|
||||
rows.append([(c.strip() if isinstance(c, str) else c) for c in ws.row_values(i)])
|
||||
yield ws.name, rows
|
||||
|
||||
def main_sheet(path):
|
||||
"""The printed annex sheet: the one whose name ends with 'piel' or is the first non-BEx sheet with most rows."""
|
||||
best = None
|
||||
for name, rows in sheets(path):
|
||||
if name.lower().endswith('piel') or re.match(r'^\d+\.?\s*piel', name.lower()):
|
||||
return name, rows
|
||||
if name.startswith(('BEx', 'HEADER', 'FOOTER', 'ZQZ', 'var', 'list', 'parms')):
|
||||
continue
|
||||
if best is None or len(rows) > len(best[1]):
|
||||
best = (name, rows)
|
||||
return best
|
||||
|
||||
W = '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}'
|
||||
def docx_tables(path):
|
||||
"""List of tables, each a list of rows of cell texts."""
|
||||
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
|
||||
out = []
|
||||
for t in x.iter(W + 'tbl'):
|
||||
rows = []
|
||||
for tr in t.findall(W + 'tr'):
|
||||
rows.append([''.join(n.text or '' for n in tc.iter(W + 't')).strip() for tc in tr.findall(W + 'tc')])
|
||||
out.append(rows)
|
||||
return out
|
||||
|
||||
def docx_paras(path):
|
||||
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
|
||||
return [''.join(n.text or '' for n in p.iter(W + 't')).strip() for p in x.iter(W + 'p')]
|
||||
|
||||
def num(v):
|
||||
"""Parse an amount cell: numbers, '1 234 567', '-1 234', '–'. Returns float or None."""
|
||||
if isinstance(v, (int, float)): return float(v)
|
||||
s = str(v).replace(' ', ' ').replace(' ', '').replace('–', '-').replace('−', '-').replace(',', '.')
|
||||
return float(s) if re.fullmatch(r'-?\d+(\.\d+)?', s) else None
|
||||
33
tools/rows.py
Normal file
33
tools/rows.py
Normal file
@@ -0,0 +1,33 @@
|
||||
"""Uniform row reader with indent level of the label column, for xlsx (openpyxl) and xls (xlrd)."""
|
||||
import re
|
||||
def _label_col(cells):
|
||||
return next((i for i, v in enumerate(cells) if isinstance(v, str) and v and not re.fullmatch(r'[\d.]+', v)), 0)
|
||||
def read_rows(path, sheet_hint='piel', label_col=None, max_col=12):
|
||||
"""Return list of dicts: row (1-based), cells, indent (of label cell), bold. Stops after 200 empty rows."""
|
||||
out, empty = [], 0
|
||||
if path.lower().endswith('.xlsx'):
|
||||
import openpyxl
|
||||
wb = openpyxl.load_workbook(path, read_only=False, data_only=True)
|
||||
ws = next(w for w in wb.worksheets if sheet_hint in w.title.lower().replace('.', ''))
|
||||
for row in ws.iter_rows(min_row=1, max_col=max_col):
|
||||
cells = [("" if c.value is None else (c.value.strip() if isinstance(c.value, str) else c.value)) for c in row]
|
||||
if not any(str(v) for v in cells):
|
||||
empty += 1
|
||||
if empty > 200: break
|
||||
continue
|
||||
empty = 0
|
||||
lc = label_col if label_col is not None else _label_col(cells)
|
||||
c = row[lc] if lc < len(row) else row[0]
|
||||
out.append({'row': row[0].row, 'cells': cells, 'indent': int(c.alignment.indent or 0), 'bold': bool(c.font and c.font.b)})
|
||||
else:
|
||||
import xlrd
|
||||
bk = xlrd.open_workbook(path, formatting_info=True)
|
||||
sh = next(bk.sheet_by_name(n) for n in bk.sheet_names() if sheet_hint in n.lower().replace('.', ''))
|
||||
for r in range(sh.nrows):
|
||||
cells = [(c.strip() if isinstance(c, str) else c) for c in sh.row_values(r)][:max_col]
|
||||
if not any(str(v) for v in cells): continue
|
||||
lc = label_col if label_col is not None else _label_col(cells)
|
||||
xf = bk.xf_list[sh.cell_xf_index(r, lc)] if lc < sh.ncols else None
|
||||
out.append({'row': r + 1, 'cells': cells, 'indent': xf.alignment.indent_level if xf else 0,
|
||||
'bold': bool(xf and bk.font_list[xf.font_index].bold)})
|
||||
return out
|
||||
174
tools/verify_budget.py
Normal file
174
tools/verify_budget.py
Normal file
@@ -0,0 +1,174 @@
|
||||
"""Round-trip proof: recompute every printed number of the budget law annexes from the XML alone.
|
||||
|
||||
Usage: python verify_budget.py OUT_DIR YEAR
|
||||
Reads lv-vb-YEAR.xml (validated against the XSD) and lv-vb-YEAR.checks.json (each printed number with its meaning).
|
||||
A check passes when the value computed from the XML equals the printed value (difference < 0.5 EUR; < 0.005 for % of GDP).
|
||||
"""
|
||||
import sys, os, re, json, collections
|
||||
TOOLS = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
|
||||
sys.path.insert(0, TOOLS)
|
||||
from lxml import etree
|
||||
from calc import Calc
|
||||
|
||||
OUT, YEAR = sys.argv[1], int(sys.argv[2])
|
||||
BID = f'lv-vb-{YEAR}'
|
||||
NS = '{urn:pppa:vpk:budzets:0.1}'
|
||||
schema = etree.XMLSchema(etree.parse(os.path.join(ROOT, 'schemas', 'valsts-budzets-0.1.xsd')))
|
||||
doc = etree.parse(os.path.join(OUT, f'{BID}.xml'))
|
||||
valid = schema.validate(doc)
|
||||
print('XSD valid:', valid)
|
||||
for e in list(schema.error_log)[:10]:
|
||||
print(' ', e.line, e.message[:200])
|
||||
|
||||
def norm(s):
|
||||
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
|
||||
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
|
||||
|
||||
root = doc.getroot()
|
||||
purpose_parent = {p.get('id'): p.get('parent') for p in root.iter(NS + 'Purpose')}
|
||||
cls_parent, cls_by_name, intra = {}, collections.defaultdict(set), set()
|
||||
for c in root.iter(NS + 'ClassItem'):
|
||||
k = f"{c.get('scheme')}:{c.get('code')}"
|
||||
if c.get('parent'):
|
||||
cls_parent[k] = f"{c.get('parentScheme') or c.get('scheme')}:{c.get('parent')}"
|
||||
if c.get('intraFund') == 'true':
|
||||
intra.add(k)
|
||||
cls_by_name[norm(c.get('name'))].add(k)
|
||||
gdp = {int(p.get('year')): float(p.get('value')) for p in root.iter(NS + 'Parameter') if p.get('code') == 'IKP'}
|
||||
A = []
|
||||
for a in root.iter(NS + 'Allocation'):
|
||||
d = dict(a.attrib)
|
||||
d['year'] = int(d['year'])
|
||||
d['untilEnd'] = d.get('untilEnd') == 'true'
|
||||
A.append(d)
|
||||
data = json.load(open(os.path.join(OUT, f'{BID}.checks.json'), encoding='utf-8'))
|
||||
purpose_holder = {p.get('id'): p.get('holder') for p in root.iter(NS + 'Purpose')}
|
||||
calc = Calc(A, purpose_parent, cls_parent, intra, data['blockcodes'], purpose_holder)
|
||||
print('allocations:', len(A))
|
||||
|
||||
res = collections.defaultdict(lambda: {'n': 0, 'ok': 0, 'bad': [], 'skip': 0})
|
||||
def judge(c, got, tol=0.5):
|
||||
r = res[c['annex']]
|
||||
r['n'] += 1
|
||||
if got is None:
|
||||
r['skip'] += 1
|
||||
r['bad'].append(('not computed', c['row'], c['sel'].get('label', ''), c['value'], None))
|
||||
elif abs(got - c['value']) < tol:
|
||||
r['ok'] += 1
|
||||
else:
|
||||
r['bad'].append(('mismatch', c['row'], json.dumps({k: v for k, v in c['sel'].items() if k != 'codes'}, ensure_ascii=False)[:230], c['value'], round(got, 2)))
|
||||
|
||||
# ------------------------------------------------------------------ annex 1: consolidated budget by formula
|
||||
def a1_value(label, year, ctx, pct):
|
||||
n = norm(label)
|
||||
base = 'appropriation' if year == YEAR else 'ceiling'
|
||||
T = calc.total
|
||||
def rev(f): return T(f, 'revenue', year, 'forecast')
|
||||
def exp(f, pre=None): return T(f, 'expenditure', year, base, ekk_prefix=pre)
|
||||
def fin(f, codes=None): return T(f, 'financing', year, base, codes=codes)
|
||||
def cap(f): return exp(f, '5') + exp(f, '9')
|
||||
b2s_m, b2s_c, s2b_m, s2b_c = exp('basic', '712'), exp('basic', '912'), exp('special', '711'), exp('special', '911')
|
||||
PA, SA = rev('basic') - (s2b_m + s2b_c), rev('special') - (b2s_m + b2s_c)
|
||||
PB, SB = exp('basic') - (b2s_m + b2s_c), exp('special') - (s2b_m + s2b_c)
|
||||
PB2, SB2 = cap('basic') - b2s_c, cap('special') - s2b_c
|
||||
PB1, SB1 = PB - PB2, SB - SB2
|
||||
if pct:
|
||||
g = gdp.get(year)
|
||||
if not g:
|
||||
return None
|
||||
val = {'valsts budžeta ieņēmumi': PA + SA, 'valsts budžeta izdevumi': PB + SB, 'valsts budžeta finansiālā bilance': PA + SA - PB - SB}
|
||||
for k, v in val.items():
|
||||
if n.startswith(k):
|
||||
return round(v / (g * 1e6) * 100, 2)
|
||||
return None
|
||||
first = n.split(' ')[0] if n else ''
|
||||
exact = {'ka': PA + SA, 'pa': PA, 'sa': SA, 'kb': PB + SB, 'kb1': PB1 + SB1, 'kb2': PB2 + SB2,
|
||||
'pb': PB, 'pb1': PB1, 'pb2': PB2, 'sb': SB, 'sb1': SB1, 'sb2': SB2}
|
||||
if first in exact:
|
||||
return exact[first]
|
||||
prefix = [
|
||||
('valsts pamatbudžeta ieņēmumi', rev('basic')), ('valsts speciālā budžeta ieņēmumi', rev('special')),
|
||||
('valsts pamatbudžeta izdevumi', exp('basic')), ('valsts speciālā budžeta izdevumi', exp('special')),
|
||||
('valsts pamatbudžeta uzturēšanas izdevumi', exp('basic') - cap('basic')), ('valsts pamatbudžeta kapitālie izdevumi', cap('basic')),
|
||||
('valsts speciālā budžeta uzturēšanas izdevumi', exp('special') - cap('special')), ('valsts speciālā budžeta kapitālie izdevumi', cap('special')),
|
||||
('valsts budžeta finansiālā bilance', PA + SA - PB - SB),
|
||||
('valsts pamatbudžeta finansiālā bilance', rev('basic') - exp('basic')),
|
||||
('valsts speciālā budžeta finansiālā bilance', rev('special') - exp('special')),
|
||||
]
|
||||
for k, v in prefix:
|
||||
if n.startswith(k):
|
||||
return v
|
||||
if n.startswith('mīnus transferts no valsts speciālā'): return s2b_m + s2b_c
|
||||
if n.startswith('mīnus transferts no valsts pamatbudžeta'): return b2s_m + b2s_c
|
||||
if n.startswith('mīnus transferts valsts speciāl'):
|
||||
return {'gross': b2s_m + b2s_c, 'maint': b2s_m, 'cap': b2s_c}[ctx['exp_part']]
|
||||
if n.startswith('mīnus transferts valsts pamatbudžet'):
|
||||
return {'gross': s2b_m + s2b_c, 'maint': s2b_m, 'cap': s2b_c}[ctx['exp_part']]
|
||||
f = {'fin_all': None, 'fin_basic': 'basic', 'fin_special': 'special'}.get(ctx['part'])
|
||||
if n == 'finansēšana' and ctx['part'].startswith('fin'):
|
||||
return fin(f)
|
||||
if not ctx['part'].startswith('fin'):
|
||||
a2 = a2_codes.get((ctx['fund'], n))
|
||||
if a2:
|
||||
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=a2, blk=f"a2:{ctx['fund']}")
|
||||
codes = cls_by_name.get(n)
|
||||
if not codes:
|
||||
return None
|
||||
if ctx['part'].startswith('fin'):
|
||||
return fin(f, codes)
|
||||
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=codes)
|
||||
|
||||
a2_codes = {}
|
||||
for c in data['checks']:
|
||||
if c['annex'] == 2 and c['sel'].get('label') and c['sel'].get('codes'):
|
||||
a2_codes.setdefault((c['sel']['fund'], norm(c['sel']['label'])), c['sel']['codes'])
|
||||
ctx = {'fund': 'basic', 'part': 'rev', 'exp_part': 'gross'}
|
||||
for c in sorted([c for c in data['checks'] if c['annex'] == 1], key=lambda c: (c['row'], c['sel']['year'])):
|
||||
n = norm(c['sel']['label'])
|
||||
if n.startswith('valsts pamatbudžeta ieņēmumi'): ctx.update(fund='basic', part='rev')
|
||||
if n.startswith('valsts speciālā budžeta ieņēmumi'): ctx.update(fund='special', part='rev')
|
||||
if n.startswith('valsts budžeta finansiālā bilance'): ctx.update(part='fin_all')
|
||||
if n.startswith('valsts pamatbudžeta finansiālā bilance'): ctx.update(part='fin_basic')
|
||||
if n.startswith('valsts speciālā budžeta finansiālā bilance'): ctx.update(part='fin_special')
|
||||
if n.startswith(('valsts pamatbudžeta', 'valsts speciālā budžeta')) and 'izdevumi' in n:
|
||||
ctx.update(exp_part='maint' if 'uzturēšanas' in n else ('cap' if 'kapitālie' in n else 'gross'))
|
||||
judge(c, a1_value(c['sel']['label'], c['sel']['year'], ctx, c['sel']['pct']), tol=0.005 if c['sel']['pct'] else 0.5)
|
||||
|
||||
# ------------------------------------------------------------------ all other annexes
|
||||
for c in data['checks']:
|
||||
if c['annex'] == 1:
|
||||
continue
|
||||
s = c['sel']
|
||||
kind = s.get('kind')
|
||||
if kind == 'grant_total':
|
||||
got = sum(float(a['amount']) for a in A if a.get('purpose') == s['purpose'] and f"{a['scheme']}:{a['code']}" == s['code']
|
||||
and (not s.get('periodFrom') or a.get('periodFrom') == s['periodFrom']))
|
||||
judge(c, got); continue
|
||||
if kind in ('fees_resort', 'fees_total'):
|
||||
got = sum(float(a['amount']) for a in A if a.get('partOf') and a['src'].endswith('.p02') and a['year'] == s['year']
|
||||
and (kind == 'fees_total' or a.get('holder') == s['holder']))
|
||||
judge(c, got); continue
|
||||
common = dict(purpose=s.get('purpose'), block=s.get('block'), ckind=s.get('ckind'), untilEnd=s.get('untilEnd', False), blk=s.get('blk'),
|
||||
holder_view=bool(s.get('holderView')))
|
||||
if kind == 'balance':
|
||||
printed = data['blockcodes'].get(s.get('blk'), {})
|
||||
if c['annex'] != 11 and ((s.get('purpose') is None and not s.get('block')) or s.get('fund') == 'special'):
|
||||
inflow = calc.total(s.get('fund'), 'revenue', s['year'], 'forecast', **common)
|
||||
else:
|
||||
inflow = calc.total(s.get('fund'), 'resource', s['year'], s['nature'], **common)
|
||||
judge(c, inflow - calc.total(s.get('fund'), 'expenditure', s['year'], s['nature'], **common)); continue
|
||||
judge(c, calc.total(s.get('fund'), s['flow'], s['year'], s['nature'], codes=s.get('codes'), **common))
|
||||
|
||||
# ------------------------------------------------------------------ report
|
||||
tot_n = tot_ok = 0
|
||||
summary = {}
|
||||
for an in sorted(res):
|
||||
r = res[an]
|
||||
tot_n += r['n']; tot_ok += r['ok']
|
||||
summary[an] = {'checks': r['n'], 'ok': r['ok'], 'mismatch': r['n'] - r['ok'] - r['skip'], 'not_computed': r['skip']}
|
||||
print(f"annex {an:2d}: {r['n']:6d} printed numbers, {r['ok']:6d} reproduced ({100 * r['ok'] / max(r['n'], 1):.2f}%), {r['skip']} not computed")
|
||||
print(f'TOTAL: {tot_ok}/{tot_n} printed numbers reproduced from XML ({100 * tot_ok / max(tot_n, 1):.3f}%)')
|
||||
json.dump({'xsd_valid': valid, 'allocations': len(A), 'annexes': summary, 'total': tot_n, 'reproduced': tot_ok,
|
||||
'issues': {an: r['bad'][:300] for an, r in res.items()}},
|
||||
open(os.path.join(OUT, f'{BID}.verify.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
|
||||
Reference in New Issue
Block a user