Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation), nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi. Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML. Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
863 lines
45 KiB
Python
863 lines
45 KiB
Python
"""Convert an adopted Latvian state budget law (text + 12 annexes) into one XML file
|
|
following valsts-budzets-0.1.xsd, plus a JSON list of every printed number as a check.
|
|
|
|
Usage: python parse_budget.py YEAR LAW_HTML ANNEX_DIR OUT_DIR
|
|
Annex files are found by number (P04.XLSX, 4_PIELIKUMS.XLS, ...).
|
|
Storage rule: each fact is stored once, from its most detailed source; every other printed number becomes a check.
|
|
"""
|
|
import sys, os, re, json, html, hashlib, collections, datetime
|
|
TOOLS = os.path.dirname(os.path.abspath(__file__))
|
|
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
|
|
KODI = os.path.join(ROOT, 'source', 'kodi')
|
|
sys.path.insert(0, TOOLS)
|
|
from rows import read_rows
|
|
from rd import docx_tables, num
|
|
from xml.sax.saxutils import quoteattr, escape
|
|
from calc import Calc
|
|
|
|
YEAR, LAW_HTML, ADIR, OUT = int(sys.argv[1]), sys.argv[2], sys.argv[3], sys.argv[4]
|
|
Y3 = [YEAR, YEAR + 1, YEAR + 2]
|
|
BID = f'lv-vb-{YEAR}'
|
|
report = collections.defaultdict(list)
|
|
|
|
# ------------------------------------------------------------------ helpers
|
|
def norm(s):
|
|
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
|
|
s = re.sub(r'^i estādes', 'iestādes', s)
|
|
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
|
|
|
|
def slug(s, n=40):
|
|
t = norm(s).translate(str.maketrans('āčēģīķļņšūž', 'acegiklnsuz'))
|
|
return re.sub(r'\s+', '-', t)[:n].strip('-')
|
|
|
|
def annex_file(n):
|
|
for f in os.listdir(ADIR):
|
|
m = re.match(r'^P?0*(\d+)[_.]', f, re.I)
|
|
if m and int(m.group(1)) == n:
|
|
return os.path.join(ADIR, f)
|
|
raise FileNotFoundError(f'annex {n}')
|
|
|
|
def sha(path):
|
|
return hashlib.sha256(open(path, 'rb').read()).hexdigest()
|
|
|
|
# ------------------------------------------------------------------ code book
|
|
SCHEME_OF_FLOW = {'expenditure': 'ekk', 'resource': 'revenue', 'revenue': 'revenue', 'financing': 'financing'}
|
|
book = collections.defaultdict(lambda: collections.defaultdict(set)) # scheme -> normname -> codes
|
|
for nm, kc in json.load(open(os.path.join(KODI, 'klasdict.json'), encoding='utf-8')).items():
|
|
for kind, c in kc:
|
|
book[kind][norm(nm)].add(c)
|
|
KL = {'Izdevumi': 'ekk', 'Ieņēmumi': 'revenue', 'Finansēšana': 'financing'}
|
|
for line in open(os.path.join(KODI, 'vk_codes.tsv'), encoding='utf-8'):
|
|
kl, v = line.rstrip('\n').split('\t')
|
|
m = re.match(r'^([A-Z]{0,2}\d[\d.]*)\s+(.+)$', v)
|
|
if m and kl in KL:
|
|
book[KL[kl]][norm(m.group(2))].add(m.group(1))
|
|
for nm, codes in json.load(open(os.path.join(KODI, 'codepairs.json'), encoding='utf-8')).items():
|
|
for c in codes:
|
|
sch = 'financing' if c.startswith('F') or re.fullmatch(r'[A-Z]{1,2}F\d+', c) else ('ekk' if re.fullmatch(r'\d{4}', c) else ('revenue' if re.fullmatch(r'\d{5}', c) else 'law'))
|
|
book[sch][norm(nm)].add(c)
|
|
|
|
classitems = {} # (scheme, code) -> {'name', 'parent'}
|
|
SECTION = {'ieņēmumi kopā': 'revenue', 'resursi izdevumu segšanai': 'resource', 'izdevumi kopā': 'expenditure',
|
|
'finansēšana': 'financing', 'finansiālā bilance': 'balance'}
|
|
|
|
def canon(sch, c):
|
|
if sch == 'financing' and re.fullmatch(r'[PKS]F\d{8}', c):
|
|
return c[1:]
|
|
return c
|
|
|
|
def code_for(label, flow):
|
|
n = norm(label)
|
|
pref = SCHEME_OF_FLOW.get(flow, 'law')
|
|
for sch in (pref, 'law', 'revenue', 'ekk', 'financing'):
|
|
cs = book[sch].get(n)
|
|
if cs:
|
|
cs = {canon(sch, x) for x in cs}
|
|
c = sorted(cs, key=lambda x: (len(x), x))[0]
|
|
if len(cs) > 1:
|
|
report['ambiguous_code'].append(f'{label[:60]} -> {sorted(cs)} (took {c})')
|
|
return sch, c
|
|
c = 'L-' + slug(label)
|
|
report['synthesised_code'].append(label)
|
|
return 'law', c
|
|
|
|
INTRA = ('savstarpējie transferti', 'no valsts pamatbudžeta uz valsts pamatbudžetu', 'no valsts speciālā budžeta uz valsts speciālo budžetu',
|
|
'atmaksām valsts pamatbudžet', 'atmaksa valsts budžetā par veiktajiem', 'valsts pamatbudžeta iestāžu saņemtie transferti no valsts pamatbudžeta',
|
|
'pārējie valsts pamatbudžetā saņemtie transferti no valsts pamatbudžeta', 'valsts speciālā budžeta iestāžu saņemtie transferti no valsts speciālā budžeta')
|
|
def is_intra(name):
|
|
n = norm(name)
|
|
return any(norm(k) in n for k in INTRA)
|
|
|
|
VOTE_WEIGHT = {'a2': 1, 'a3': 1, 'a4': 1, 'a5': 1, 'a11': 0}
|
|
CUR_ANNEX = ['a4']
|
|
def register_class(sch, code, name, parent):
|
|
k = (sch, code)
|
|
d = classitems.setdefault(k, {'name': name, 'parent': None, 'votes': collections.Counter()})
|
|
w = VOTE_WEIGHT.get(CUR_ANNEX[0], 1)
|
|
if w:
|
|
d['votes'][parent] += w
|
|
|
|
def structural_ok(child, parent):
|
|
"""Numeric classification codes carry their own hierarchy: 21200 cannot sit under 21100, 7131 can sit under 7130."""
|
|
cs, cc = child.split(':', 1)
|
|
ps, pc = parent.split(':', 1)
|
|
if not (cc.isdigit() and pc.isdigit() and cs == ps and len(cc) == len(pc)):
|
|
return True
|
|
stem = pc.rstrip('0')
|
|
return cc != pc and cc.startswith(stem)
|
|
|
|
def finalize_classes():
|
|
"""Canonical parent = majority vote over indented annex blocks, restricted to structurally possible parents."""
|
|
for k, d in classitems.items():
|
|
votes = d['votes']
|
|
ck = f'{k[0]}:{k[1]}'
|
|
real = [(p, n) for p, n in votes.most_common() if p and structural_ok(ck, p)]
|
|
rejected = [p for p in votes if p and not structural_ok(ck, p)]
|
|
if rejected:
|
|
report['parent_rejected_by_structure'].append(f'{ck} {d["name"][:40]}: not under {rejected}')
|
|
d['parent'] = real[0][0] if real and (not votes.get(None) or real[0][1] >= votes[None] or rejected) else None
|
|
if len(real) > 1:
|
|
report['class_parent_votes'].append(f'{k[0]}:{k[1]} {d["name"][:50]} votes {dict(votes)} -> {d["parent"]}')
|
|
|
|
# ------------------------------------------------------------------ actors and purposes
|
|
actors, purposes = {}, {}
|
|
VPK_RESORTS = {l.split('\t')[0] for l in open(os.path.join(KODI, 'vpk_resorts.tsv'), encoding='utf-8') if l.strip()}
|
|
def resort_actor(code, name):
|
|
aid = f'{BID}.ac.r{code}'
|
|
vpk = f'{code}-0000' if f'{code}-0000' in VPK_RESORTS else None
|
|
actors.setdefault(aid, {'kind': 'resort', 'code': code, 'name': name, 'vpk': vpk})
|
|
pid = f'{BID}.pr.{code}'
|
|
purposes.setdefault(pid, {'kind': 'resort', 'code': code, 'name': name, 'holder': aid})
|
|
return aid, pid
|
|
|
|
def prog_purpose(rcode, pcode, name, function, parent_pid):
|
|
pid = f'{BID}.pr.{rcode}.{pcode}'
|
|
kind = 'programme' if pcode.endswith('.00.00') else 'subprogramme'
|
|
d = purposes.setdefault(pid, {'kind': kind, 'code': pcode, 'name': name, 'parent': parent_pid,
|
|
'holder': f'{BID}.ac.r{rcode}', 'function': function})
|
|
if function and not d.get('function'):
|
|
d['function'] = function
|
|
return pid
|
|
|
|
def project_purpose(rcode, pcode, projcode, name, parent_pid):
|
|
pid = f'{BID}.pj.{rcode}.{pcode or "x"}.{slug(projcode, 60)}'
|
|
purposes.setdefault(pid, {'kind': 'project', 'code': projcode, 'name': name, 'parent': parent_pid,
|
|
'holder': f'{BID}.ac.r{rcode}'})
|
|
return pid
|
|
|
|
def municipality(name):
|
|
aid = f'{BID}.ac.m-{slug(name)}'
|
|
actors.setdefault(aid, {'kind': 'municipality', 'name': name})
|
|
return aid
|
|
|
|
# block (core/eu) per subprogramme, from Valsts kase execution data; fallback by code range
|
|
blockmap = {}
|
|
for line in open(os.path.join(KODI, 'vk_blocks.tsv'), encoding='utf-8'):
|
|
g, m, p, sp, blk, bt, n = line.rstrip('\n').split('\t')
|
|
if int(g) in (YEAR, YEAR - 1):
|
|
blockmap.setdefault((m, sp), 'eu' if blk.startswith('Ārvalstu') else 'core')
|
|
def block_of(rcode, pcode):
|
|
b = blockmap.get((rcode, pcode))
|
|
if b:
|
|
return b
|
|
report['block_fallback'].append(f'{rcode} {pcode}')
|
|
return 'eu' if pcode[:2] in ('60', '61', '62', '63', '64', '65', '66', '67', '68', '69', '70', '71', '72', '73', '74', '75', '76', '77', '78', '79', '80', '81', '82', '83', '84', '85') else 'core'
|
|
|
|
# ------------------------------------------------------------------ outputs
|
|
allocs, checks = [], []
|
|
blockcodes = collections.defaultdict(lambda: collections.defaultdict(set)) # block id -> flow -> printed codes
|
|
def register_block(bid, lines):
|
|
for ln in lines:
|
|
if not ln.get('skip') and not ln.get('section') and ln.get('flow') not in (None, 'balance'):
|
|
blockcodes[bid][ln['flow']].add(f"{ln['scheme']}:{ln['code']}")
|
|
_alloc_ids = collections.Counter()
|
|
def add_alloc(**a):
|
|
"""Allocation id = source annex + source row + year (+ 'l' for 'later years' column, + -N if the same cell yields several facts).
|
|
Derived from the source, so re-parsing the same law gives the same ids."""
|
|
base = f'{BID}.a.{a.pop("srcAnnex")}.r{a.get("srcRow", 0)}.y{a["year"]}{"l" if a.get("untilEnd") else ""}'
|
|
_alloc_ids[base] += 1
|
|
a['id'] = base if _alloc_ids[base] == 1 else f'{base}-{_alloc_ids[base]}'
|
|
allocs.append(a)
|
|
return a['id']
|
|
|
|
def add_check(annex, row, value, **sel):
|
|
checks.append({'annex': annex, 'row': row, 'value': value, 'sel': sel})
|
|
|
|
pending = [] # (bid, flow, ck, alloc kwargs) for lines of detailed blocks; leafness decided on the canonical tree
|
|
def queue_alloc(_bid, _flow, _ck, **kw):
|
|
pending.append((_bid, _flow, _ck, kw))
|
|
|
|
def canon_chain(ck, parent):
|
|
out, k = [], ck
|
|
while k and k not in out:
|
|
out.append(k)
|
|
k = parent.get(k)
|
|
return out
|
|
|
|
def finalize_lines():
|
|
parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
|
|
below = {} # (bid, flow, ck) -> printed codes in that block whose canonical chain contains ck
|
|
for bid, fl in blockcodes.items():
|
|
for flow, codes in fl.items():
|
|
for c in codes:
|
|
for anc in canon_chain(c, parent):
|
|
if anc in codes:
|
|
below.setdefault((bid, flow, anc), []).append(c)
|
|
for c in checks:
|
|
r = c['sel'].get('codes')
|
|
if isinstance(r, dict) and 'resolve' in r:
|
|
bid, flow, ck = r['resolve']
|
|
c['sel']['codes'] = sorted(set(below.get((bid, flow, ck), [ck])) | {ck})
|
|
for bid, flow, ck, kw in pending:
|
|
if len(set(below.get((bid, flow, ck), [ck])) - {ck}) == 0:
|
|
add_alloc(**kw)
|
|
|
|
# ------------------------------------------------------------------ generic tree-annex reader
|
|
def tree_blocks(rows, label_col, amount_cols, header_fn, level_fn):
|
|
"""Split rows into blocks keyed by header path; each block has lines with level, flow, amounts."""
|
|
blocks, path, cur = [], {}, None
|
|
for r in rows:
|
|
cells = r['cells']
|
|
h = header_fn(r, path)
|
|
if h:
|
|
path = h
|
|
cur = {'path': dict(path), 'lines': []}
|
|
blocks.append(cur)
|
|
continue
|
|
label = str(cells[label_col]) if len(cells) > label_col else ''
|
|
amts = {}
|
|
for y, ci in amount_cols.items():
|
|
v = num(cells[ci]) if ci < len(cells) else None
|
|
if v is not None:
|
|
amts[y] = v
|
|
if not label or not amts:
|
|
continue
|
|
if cur is None:
|
|
cur = {'path': dict(path), 'lines': []}
|
|
blocks.append(cur)
|
|
cur['lines'].append({'row': r['row'], 'label': label, 'level': level_fn(r, label), 'amounts': amts})
|
|
return blocks
|
|
|
|
def resolve_lines(block):
|
|
"""Assign flow, code and parent code to every line; mark leaves (no deeper line follows in the same section)."""
|
|
lines, flow, stack = block['lines'], None, []
|
|
for i, ln in enumerate(lines):
|
|
n = norm(ln['label'])
|
|
if n in SECTION:
|
|
flow = SECTION[n]
|
|
ln.update(flow=flow, section=True, scheme='law', code='S-' + flow, leaf=False)
|
|
stack = [(ln['level'], ln)]
|
|
continue
|
|
if flow is None:
|
|
ln.update(flow=None, skip=True)
|
|
continue
|
|
while stack and stack[-1][0] >= ln['level']:
|
|
stack.pop()
|
|
parent = stack[-1][1] if stack else None
|
|
sch, code = code_for(ln['label'], flow)
|
|
ln.update(flow=flow, scheme=sch, code=code, section=False)
|
|
if parent is not None and not parent.get('section'):
|
|
register_class(sch, code, ln['label'], f'{parent["scheme"]}:{parent["code"]}')
|
|
else:
|
|
register_class(sch, code, ln['label'], None)
|
|
stack.append((ln['level'], ln))
|
|
for i, ln in enumerate(lines):
|
|
if ln.get('skip'):
|
|
continue
|
|
desc = [] if ln.get('section') else [f"{ln['scheme']}:{ln['code']}"]
|
|
for x in lines[i + 1:]:
|
|
if x.get('skip'):
|
|
continue
|
|
if x.get('section') or x['level'] <= ln['level'] or x.get('flow') != ln.get('flow'):
|
|
break
|
|
desc.append(f"{x['scheme']}:{x['code']}")
|
|
ln['codes'] = desc
|
|
for i, ln in enumerate(lines):
|
|
if ln.get('section') or ln.get('skip'):
|
|
continue
|
|
nxt = next((x for x in lines[i + 1:] if not x.get('skip')), None)
|
|
ln['leaf'] = nxt is None or nxt.get('section') or nxt['level'] <= ln['level'] or nxt.get('flow') != ln['flow']
|
|
return lines
|
|
|
|
# ------------------------------------------------------------------ annex 4 and 5 (2026 appropriations by subprogramme)
|
|
label_levels = collections.defaultdict(collections.Counter) # (flow, normlabel) -> relative level counts, learned for annex 11
|
|
|
|
def parse_programmes(n, fund):
|
|
CUR_ANNEX[0] = f'a{n}'
|
|
path_ = annex_file(n)
|
|
rows = read_rows(path_, label_col=2)
|
|
progs_seen = {}
|
|
def header(r, path):
|
|
c = r['cells']
|
|
lab = str(c[2]) if len(c) > 2 else ''
|
|
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
|
if m and num(c[3] if len(c) > 3 else '') is None:
|
|
aid, pid = resort_actor(m.group(1), m.group(2))
|
|
return {'resort': m.group(1), 'rpid': pid}
|
|
code = str(c[0])
|
|
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
|
|
rc = path['resort']
|
|
parent = path['rpid']
|
|
pcode = code
|
|
if not code.endswith('.00.00'):
|
|
pp = code[:2] + '.00.00'
|
|
if (rc, pp) in progs_seen:
|
|
parent = progs_seen[(rc, pp)]
|
|
func = str(c[1]) if re.fullmatch(r'\d{2}\.\d{3}', str(c[1])) else None
|
|
pid = prog_purpose(rc, pcode, lab, func, parent)
|
|
progs_seen[(rc, pcode)] = pid
|
|
return {'resort': rc, 'rpid': path['rpid'], 'pcode': pcode, 'pid': pid, 'parent': parent, 'func': func}
|
|
return None
|
|
blocks = tree_blocks(rows, 2, {YEAR: 3}, header, lambda r, lab: r['indent'])
|
|
# leaf blocks: subprogrammes, or programmes without subprogrammes
|
|
has_child = set(b['path'].get('parent') for b in blocks if b['path'].get('pcode'))
|
|
for bi, b in enumerate(blocks):
|
|
lines = resolve_lines(b)
|
|
bid = f'a{n}:{bi}'
|
|
register_block(bid, lines)
|
|
p = b['path']
|
|
if 'pid' in p:
|
|
sel_purpose = p['pid']
|
|
elif 'rpid' in p:
|
|
sel_purpose = p['rpid']
|
|
else:
|
|
sel_purpose = None
|
|
leafblock = 'pid' in p and p['pid'] not in has_child
|
|
blk = block_of(p['resort'], p['pcode']) if 'pcode' in p else None
|
|
# learn relative levels for annex 11
|
|
base = None
|
|
for ln in lines:
|
|
if ln.get('section'):
|
|
base = ln['level']
|
|
elif not ln.get('skip') and base is not None:
|
|
label_levels[(ln['flow'], norm(ln['label']))][ln['level'] - base] += 1
|
|
for ln in lines:
|
|
if ln.get('skip') or ln['flow'] == 'balance':
|
|
if ln.get('flow') == 'balance' or ln.get('section') and ln['flow'] == 'balance':
|
|
add_check(n, ln['row'], ln['amounts'][YEAR], kind='balance', fund=fund, year=YEAR, purpose=sel_purpose, nature='appropriation', blk=bid)
|
|
continue
|
|
nature = 'forecast' if ln['flow'] == 'revenue' and fund == 'basic' else 'appropriation'
|
|
if fund == 'special' and ln['flow'] == 'revenue':
|
|
nature = 'forecast'
|
|
add_check(n, ln['row'], ln['amounts'][YEAR], fund=fund, flow=ln['flow'], year=YEAR, purpose=sel_purpose,
|
|
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, nature=nature, blk=bid)
|
|
store = leafblock and not ln.get('section')
|
|
if fund == 'basic' and ln['flow'] == 'revenue':
|
|
store = False # basic-budget revenue is stored from annex 2
|
|
if store:
|
|
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex=f'p{n:02d}', flow=ln['flow'], fund=fund, nature=nature, year=YEAR,
|
|
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['pid'], block=blk, function=p.get('func'),
|
|
scheme=ln['scheme'], code=ln['code'], amount=ln['amounts'][YEAR],
|
|
src=f'{BID}.src.p{n:02d}', srcRow=ln['row'])
|
|
|
|
# ------------------------------------------------------------------ annex 2 (revenue forecasts, 3 years)
|
|
def parse_revenue():
|
|
CUR_ANNEX[0] = 'a2'
|
|
raw = read_rows(annex_file(2), label_col=0)
|
|
# the label column differs between years (2025 .xls has an extra empty first column): find it from the header row
|
|
lc = next((i for r in raw for i, c in enumerate(r['cells']) if str(c).startswith('Ieņēmumu avots')), 0)
|
|
rows = read_rows(annex_file(2), label_col=lc) if lc else raw
|
|
part, fund, mode = None, 'basic', 'tree'
|
|
lines, fees, info = [], [], []
|
|
cur_resort = None
|
|
for r in rows:
|
|
c = r['cells']
|
|
lab = str(c[lc]) if lc < len(c) else ''
|
|
if re.match(r'^I\.\s', lab):
|
|
fund, mode = 'basic', 'tree'; continue
|
|
if re.match(r'^II\.\s', lab):
|
|
fund, mode = 'special', 'tree'; continue
|
|
if lab.startswith('Ministrija (cita'):
|
|
mode = 'fees'; continue
|
|
if lab.startswith('Informatīvi'):
|
|
mode = 'info'; continue
|
|
amts = {y: num(c[lc + 1 + i]) for i, y in enumerate(Y3) if lc + 1 + i < len(c) and num(c[lc + 1 + i]) is not None}
|
|
if not amts or not lab:
|
|
continue
|
|
if mode == 'tree':
|
|
lines.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'amounts': amts, 'fund': fund})
|
|
elif mode == 'fees':
|
|
fees.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'bold': r['bold'], 'amounts': amts})
|
|
else:
|
|
info.append({'row': r['row'], 'label': lab, 'amounts': amts})
|
|
stored = {}
|
|
for fund in ('basic', 'special'):
|
|
blk = {'lines': [dict(l) for l in lines if l['fund'] == fund]}
|
|
resolve_lines(blk)
|
|
register_block(f'a2:{fund}', blk['lines'])
|
|
for ln in blk['lines']:
|
|
if ln.get('skip'):
|
|
continue
|
|
for y, v in ln['amounts'].items():
|
|
add_check(2, ln['row'], v, fund=fund, flow='revenue', year=y, nature='forecast', label=ln['label'],
|
|
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=ln.get('codes'), blk=f'a2:{fund}')
|
|
if ln.get('leaf') and not ln.get('section'):
|
|
if fund == 'special' and y == YEAR:
|
|
continue # 2026 special revenue is stored from annex 5 (by subprogramme)
|
|
aid = add_alloc(srcAnnex='p02', flow='revenue', fund=fund, nature='forecast', year=y,
|
|
holder=None, purpose=None, scheme=ln['scheme'], code=ln['code'], amount=v,
|
|
src=f'{BID}.src.p02', srcRow=ln['row'])
|
|
stored[(fund, ln['code'], y)] = aid
|
|
# fees by administering resort: a breakdown of stored revenue leaves -> partOf
|
|
leaf_codes = sorted({k[1] for k in stored if k[0] == 'basic'}, key=len, reverse=True)
|
|
for f in fees:
|
|
n_ = norm(f['label'])
|
|
if n_.startswith('ieņēmumi valsts pamatbudžetā kopā'):
|
|
for y, v in f['amounts'].items():
|
|
add_check(2, f['row'], v, kind='fees_total', year=y)
|
|
continue
|
|
if f['bold']:
|
|
name = f['label']
|
|
m = [a for a, d in actors.items() if d['kind'] == 'resort' and norm(d['name']) == n_]
|
|
cur_resort = m[0] if m else None
|
|
if not cur_resort:
|
|
report['fee_resort_unmatched'].append(f['label'])
|
|
for y, v in f['amounts'].items():
|
|
add_check(2, f['row'], v, kind='fees_resort', holder=cur_resort, year=y)
|
|
continue
|
|
sch, code = code_for(f['label'], 'revenue')
|
|
register_class(sch, code, f['label'], None)
|
|
parent = next((stored[('basic', lc, y0)] for lc in leaf_codes for y0 in [YEAR] if code.startswith(lc.rstrip('0') or lc) and ('basic', lc, YEAR) in stored), None)
|
|
if parent is None:
|
|
report['fee_parent_unmatched'].append(f'{code} {f["label"][:60]}')
|
|
for y, v in f['amounts'].items():
|
|
par = None
|
|
if parent:
|
|
pc = next(k for k, a in stored.items() if a == parent)[1]
|
|
par = stored.get(('basic', pc, y))
|
|
add_alloc(srcAnnex='p02', flow='revenue', fund='basic', nature='forecast', year=y, holder=cur_resort,
|
|
purpose=None, scheme=sch, code=code, amount=v, partOf=par, src=f'{BID}.src.p02', srcRow=f['row'])
|
|
for i in info:
|
|
for y, v in i['amounts'].items():
|
|
params.append({'id': f'{BID}.par.p02-{i["row"]}-{y}', 'name': i['label'], 'year': y, 'value': v, 'unit': 'EUR',
|
|
'src': f'{BID}.src.p02'})
|
|
|
|
# ------------------------------------------------------------------ annex 3 (resort summaries, 3 years; 2027-28 stored as ceilings)
|
|
def parse_resort_summary():
|
|
CUR_ANNEX[0] = 'a3'
|
|
rows = read_rows(annex_file(3), label_col=0)
|
|
fund_names = {'valsts pamatbudžets': 'basic', 'valsts speciālais budžets': 'special'}
|
|
def header(r, path):
|
|
lab = str(r['cells'][0])
|
|
n_ = norm(lab)
|
|
if num(r['cells'][1] if len(r['cells']) > 1 else '') is not None:
|
|
return None
|
|
if n_ in fund_names:
|
|
p = {k: v for k, v in path.items() if k in ('resort', 'rpid')}
|
|
p['fund'] = fund_names[n_]
|
|
return p
|
|
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
|
if m:
|
|
aid, pid = resort_actor(m.group(1), m.group(2))
|
|
return {'resort': m.group(1), 'rpid': pid, 'fund': 'basic'}
|
|
if re.match(r'^I\.\s+Valsts pamatfunkciju', lab):
|
|
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'core'; return p
|
|
if re.match(r'^II\.\s+ES politiku', lab):
|
|
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'eu'; return p
|
|
return None
|
|
blocks = tree_blocks(rows, 0, {Y3[0]: 1, Y3[1]: 2, Y3[2]: 3}, header, lambda r, lab: r['indent'])
|
|
keys = [(b['path'].get('resort'), b['path'].get('fund')) for b in blocks]
|
|
for bi, b in enumerate(blocks):
|
|
lines = resolve_lines(b)
|
|
bid = f'a3:{bi}'
|
|
register_block(bid, lines)
|
|
p = b['path']
|
|
fund = p.get('fund', 'basic')
|
|
# a resort's special-budget part may appear without its own fund header: detect by section content later
|
|
leafblock = 'resort' in p and 'block' in p
|
|
sel = dict(fund=fund, purpose=p.get('rpid'), block=p.get('block'), blk=bid, holderView=True)
|
|
for ln in lines:
|
|
if ln.get('skip'):
|
|
continue
|
|
for y, v in ln['amounts'].items():
|
|
if ln['flow'] == 'balance':
|
|
add_check(3, ln['row'], v, kind='balance', year=y, nature='appropriation' if y == YEAR else 'ceiling', **sel)
|
|
continue
|
|
nature = 'forecast' if ln['flow'] == 'revenue' else ('appropriation' if y == YEAR else 'ceiling')
|
|
add_check(3, ln['row'], v, flow=ln['flow'], year=y, nature=nature,
|
|
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
|
|
if leafblock and not ln.get('section') and y != YEAR and ln['flow'] != 'revenue':
|
|
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p03', flow=ln['flow'], fund=fund, nature='ceiling', year=y,
|
|
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['rpid'], block=p['block'],
|
|
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p03', srcRow=ln['row'])
|
|
|
|
# ------------------------------------------------------------------ annex 11 (long-term commitments)
|
|
def parse_commitments():
|
|
CUR_ANNEX[0] = 'a11'
|
|
rows = read_rows(annex_file(11), label_col=1)
|
|
hdr = next(r for r in rows if 'Projekta kods' in [str(x) for x in r['cells']])
|
|
years = {}
|
|
for i, c in enumerate(hdr['cells']):
|
|
m = re.match(r'^(\d{4})\.', str(c))
|
|
if m:
|
|
years[int(m.group(1))] = i
|
|
elif str(c).startswith('Tālākā'):
|
|
years['later'] = i
|
|
last = max(y for y in years if y != 'later')
|
|
kinds = {}
|
|
def level_fn(r, lab):
|
|
cnt = None
|
|
for fl in ('expenditure', 'resource', 'financing', 'revenue'):
|
|
c = label_levels.get((fl, norm(lab)))
|
|
if c:
|
|
cnt = c.most_common(1)[0][0] + 1
|
|
break
|
|
if cnt is None and norm(lab) not in SECTION:
|
|
report['a11_level_unknown'].append(lab)
|
|
return 99
|
|
return cnt or 0
|
|
progs_seen = {}
|
|
state = {'prev_header': None}
|
|
def header(r, path):
|
|
h = _header(r, path)
|
|
state['prev_header'] = h is not None and h.get('_kindable', False)
|
|
if h is not None:
|
|
h.pop('_kindable', None)
|
|
return h
|
|
def _header(r, path):
|
|
c = r['cells']
|
|
lab = str(c[1]) if len(c) > 1 else ''
|
|
if any(num(c[i]) is not None for i in years.values() if i < len(c)):
|
|
return None
|
|
if lab.upper().startswith('VALSTS PAMATBUDŽETS'):
|
|
return {'fund': 'basic'}
|
|
if lab.upper().startswith('VALSTS SPECIĀLAIS BUDŽETS'):
|
|
return {'fund': 'special'}
|
|
m = re.match(r'^(\d{10})\s+(.+)$', lab)
|
|
if m:
|
|
kinds[m.group(1)] = m.group(2)
|
|
register_class('commitment', m.group(1), m.group(2), None)
|
|
if state['prev_header']: # directly under a resort or programme header
|
|
p = dict(path); p['kind'] = m.group(1)
|
|
if 'pcode' not in p:
|
|
p['rkind'] = m.group(1)
|
|
return p
|
|
return {'fund': path.get('fund', 'basic'), 'kind': m.group(1), 'topkind': m.group(1)} # new top-level commitment-kind section
|
|
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
|
|
if m:
|
|
aid, pid = resort_actor(m.group(1), m.group(2))
|
|
return {'fund': path.get('fund', 'basic'), 'resort': m.group(1), 'rpid': pid, '_kindable': True,
|
|
'topkind': path.get('topkind'), 'kind': path.get('topkind')}
|
|
code = str(c[0])
|
|
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
|
|
rc = path['resort']
|
|
pid = f'{BID}.pr.{rc}.{code}'
|
|
if pid not in purposes:
|
|
parent = purposes.get(f'{BID}.pr.{rc}.{code[:2]}.00.00') and f'{BID}.pr.{rc}.{code[:2]}.00.00' or path['rpid']
|
|
prog_purpose(rc, code, lab, None, parent)
|
|
report['a11_new_programme'].append(f'{rc} {code} {lab[:50]}')
|
|
p = {k: v for k, v in path.items() if k in ('fund', 'resort', 'rpid', 'topkind', 'rkind')}
|
|
p.update(pcode=code, pid=pid, _kindable=True, kind=path.get('rkind') or path.get('topkind'))
|
|
return p
|
|
proj = str(c[2]) if len(c) > 2 else ''
|
|
if proj and 'resort' in path:
|
|
parent = path.get('pid') or path['rpid']
|
|
pj = project_purpose(path['resort'], path.get('pcode'), proj, lab, parent)
|
|
p = dict(path); p['project'] = proj; p['pjid'] = pj
|
|
return p
|
|
return None
|
|
amount_cols = {y: i for y, i in years.items()}
|
|
blocks = tree_blocks(rows, 1, amount_cols, header, level_fn)
|
|
|
|
def more_specific(k2, k): # commitment kind k2 is a sub-type of k (or equal); codes nest in 2-digit groups
|
|
from calc import kind_stem
|
|
if not k:
|
|
return True
|
|
return bool(k2 and k2.startswith(kind_stem(k)))
|
|
def refined(bi):
|
|
"""A detailed block is refined (not a leaf) if, within the same resort+programme, a later or earlier block adds
|
|
a project under a compatible kind, or shows the same project under a more specific kind."""
|
|
p = blocks[bi]['path']
|
|
for j, b2 in enumerate(blocks):
|
|
if j == bi:
|
|
continue
|
|
q = b2['path']
|
|
if q.get('resort') != p.get('resort') or q.get('fund') != p.get('fund') or not q.get('pcode'):
|
|
continue
|
|
if q['pcode'] != p.get('pcode'):
|
|
# a programme block is refined by its subprogramme blocks (same resort, same programme number)
|
|
if (p.get('pcode') or '').endswith('.00.00') and q['pcode'][:2] == p['pcode'][:2] and not q['pcode'].endswith('.00.00') \
|
|
and not p.get('project') and more_specific(q.get('kind'), p.get('kind')):
|
|
return True
|
|
continue
|
|
if not p.get('project') and q.get('project') and more_specific(q.get('kind'), p.get('kind')):
|
|
return True
|
|
if p.get('project') and q.get('project') == p['project'] and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
|
|
return True
|
|
if not p.get('project') and not q.get('project') and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
|
|
return True
|
|
return False
|
|
# leaf blocks: those whose path is not extended by a later block
|
|
def key(p):
|
|
return (p.get('fund'), p.get('kind'), p.get('resort'), p.get('pcode'), p.get('project'))
|
|
keys = [key(b['path']) for b in blocks]
|
|
def extends(a, b): # b is strictly more specific than a
|
|
return a != b and all(x is None or x == y or (i == 1 and x and y and y.startswith(x.rstrip('0'))) for i, (x, y) in enumerate(zip(a, b)))
|
|
for bi, b in enumerate(blocks):
|
|
lines = resolve_lines(b)
|
|
bid = f'a11:{bi}'
|
|
register_block(bid, lines)
|
|
p = b['path']
|
|
leafblock = bool(p.get('pcode')) and not refined(bi)
|
|
sel = dict(fund=p.get('fund', 'basic'), ckind=p.get('kind'), purpose=p.get('pjid') or p.get('pid') or p.get('rpid'), blk=bid)
|
|
for ln in lines:
|
|
if ln.get('skip'):
|
|
continue
|
|
for y, v in ln['amounts'].items():
|
|
yy, until = (last + 1, True) if y == 'later' else (y, False)
|
|
if ln['flow'] == 'balance':
|
|
add_check(11, ln['row'], v, kind='balance', year=yy, untilEnd=until, nature='commitment', **sel)
|
|
continue
|
|
add_check(11, ln['row'], v, flow=ln['flow'], year=yy, untilEnd=until, nature='commitment',
|
|
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
|
|
if leafblock and not ln.get('section') and v != 0:
|
|
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p11', flow=ln['flow'], fund=sel['fund'], nature='commitment', year=yy, untilEnd=until,
|
|
holder=f'{BID}.ac.r{p["resort"]}' if p.get('resort') else None,
|
|
purpose=sel['purpose'], commitmentKind=p.get('kind'),
|
|
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p11', srcRow=ln['row'])
|
|
|
|
# ------------------------------------------------------------------ annexes 6-10 (earmarked grants to municipalities)
|
|
def parse_grants(n):
|
|
path_ = annex_file(n)
|
|
t = docx_tables(path_)[-1]
|
|
title = next((r[0] for r in t if r and len(r[0]) > 40), f'{n}. pielikums')
|
|
gid = f'{BID}.pr.g{n:02d}'
|
|
purposes[gid] = {'kind': 'grant', 'code': f'P{n:02d}', 'name': title}
|
|
period = (f'{YEAR}-01-01', f'{YEAR}-12-31')
|
|
cols = None
|
|
for ri, r in enumerate(t):
|
|
first = r[0] if r else ''
|
|
m = re.match(r'^(I{1,2})\.\s', first)
|
|
if m:
|
|
period = (f'{YEAR}-01-01', f'{YEAR}-08-31') if m.group(1) == 'I' else (f'{YEAR}-09-01', f'{YEAR}-12-31')
|
|
continue
|
|
if first.startswith('Pašvaldības'):
|
|
cols = r; continue
|
|
vals = [num(x) for x in r[1:]]
|
|
if not first or not any(v is not None for v in vals):
|
|
continue
|
|
n_ = norm(first)
|
|
if n_ in ('kopā', 'pavisam kopā'):
|
|
heads_ = [norm(h) for h in (cols or [])[1:]]
|
|
for ci, v in enumerate(vals):
|
|
if v is not None:
|
|
h = heads_[ci] if ci < len(heads_) else ''
|
|
add_check(n, ri + 1, v, kind='grant_total', purpose=gid, code=f'law:G{n:02d}-{slug(h or "summa", 30) or "summa"}',
|
|
periodFrom=None if n_ == 'pavisam kopā' else period[0], periodTo=None if n_ == 'pavisam kopā' else period[1], year=YEAR)
|
|
continue
|
|
recip = f'{BID}.ac.unallocated' if n_.startswith('nesadalītie') else municipality(first)
|
|
if recip.endswith('unallocated'):
|
|
actors[recip] = {'kind': 'unallocated', 'name': 'Nesadalītie līdzekļi'}
|
|
heads = [norm(h) for h in (cols or [])[1:]]
|
|
# column roles: 'pavisam kopā' = total; 'tai skaitā …' = part of previous; otherwise the main amount
|
|
ids = {}
|
|
total_idx = next((i for i, h in enumerate(heads) if h.startswith('pavisam kopā')), None)
|
|
main_idx = 0
|
|
order = ([total_idx] if total_idx is not None else []) + [i for i in range(len(vals)) if i != total_idx]
|
|
for ci in order:
|
|
v = vals[ci] if ci < len(vals) else None
|
|
if v is None:
|
|
continue
|
|
h = heads[ci] if ci < len(heads) else ''
|
|
part = None
|
|
if h.startswith('tai skaitā'):
|
|
part = ids.get(ci - 1)
|
|
elif total_idx is not None and ci != total_idx:
|
|
part = ids.get(total_idx)
|
|
aid = add_alloc(srcAnnex=f'p{n:02d}', flow='expenditure', fund='basic', nature='earmarkedGrant', year=YEAR,
|
|
periodFrom=period[0], periodTo=period[1], recipient=recip, purpose=gid,
|
|
scheme='law', code=f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', amount=v, partOf=part,
|
|
src=f'{BID}.src.p{n:02d}', srcRow=ri + 1)
|
|
register_class('law', f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', (cols[ci + 1] if cols and ci + 1 < len(cols) else 'Summa') or 'Summa', None)
|
|
ids[ci] = aid
|
|
|
|
# ------------------------------------------------------------------ annex 1 (consolidated; checks and GDP parameter)
|
|
params = []
|
|
def parse_consolidated():
|
|
rows = read_rows(annex_file(1), label_col=0)
|
|
hdr = next(r for r in rows if any(re.match(r'^\d{4}\.', str(c)) for c in r['cells']))
|
|
ycols = {int(re.match(r'^(\d{4})', str(c)).group(1)): i for i, c in enumerate(hdr['cells']) if re.match(r'^\d{4}\.', str(c))}
|
|
lc = next(i for i, c in enumerate(hdr['cells']) if str(c) == 'Nosaukums')
|
|
pct = False
|
|
for r in rows:
|
|
c = r['cells']
|
|
lab = str(c[lc]) if lc < len(c) else ''
|
|
if lab.startswith('Procentos no IKP'):
|
|
pct = True; continue
|
|
for y, i in ycols.items():
|
|
v = num(c[i]) if i < len(c) else None
|
|
if v is None or not lab:
|
|
continue
|
|
if lab.startswith('IKP milj'):
|
|
params.append({'id': f'{BID}.par.ikp-{y}', 'name': 'IKP prognoze', 'code': 'IKP', 'year': y, 'value': v,
|
|
'unit': 'milj. EUR', 'src': f'{BID}.src.p01'})
|
|
continue
|
|
add_check(1, r['row'], v, kind='a1', label=lab, indent=r['indent'], pct=pct, year=y)
|
|
|
|
# ------------------------------------------------------------------ law text
|
|
provisions = []
|
|
def parse_law():
|
|
s = open(LAW_HTML, encoding='utf-8', errors='replace').read()
|
|
chapter = None
|
|
for m in re.finditer(r"<div class='(TV212|TV213)'([^>]*)>(.*?)(?=<div class='TV21[23]'|<div class='TV9|$)", s, re.S):
|
|
kind, tag, body = m.groups()
|
|
mm = re.search(r'data-num="(\d+)"', tag)
|
|
nr = mm.group(1) if mm else None
|
|
text = body.replace('</p>', '\n').replace('<br />', ' ')
|
|
text = html.unescape(re.sub(r'<[^>]+>', '', text))
|
|
text = re.sub(r'[ \t]+', ' ', re.sub(r'\n\s*\n+', '\n', text)).strip()
|
|
if kind == 'TV212':
|
|
chapter = re.sub(r'\s+', ' ', text)
|
|
continue
|
|
if not nr:
|
|
continue
|
|
text = re.sub(r'\n?\d+\s*$', '', text).strip()
|
|
provisions.append({'id': f'{BID}.p{int(nr):03d}', 'chapter': chapter, 'article': int(nr), 'text': text,
|
|
'src': f'{BID}.src.law'})
|
|
|
|
# ------------------------------------------------------------------ run
|
|
parse_law()
|
|
parse_programmes(4, 'basic')
|
|
parse_programmes(5, 'special')
|
|
parse_revenue()
|
|
parse_resort_summary()
|
|
parse_commitments()
|
|
for g in (6, 7, 8, 9, 10):
|
|
parse_grants(g)
|
|
parse_consolidated()
|
|
finalize_classes()
|
|
finalize_lines()
|
|
|
|
# special budget revenue 2026 belongs to the resort that holds the special budget; annex 2 2027+ special revenue too
|
|
special_holders = sorted({a['holder'] for a in allocs if a['fund'] == 'special' and a.get('holder')})
|
|
for a in allocs:
|
|
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('holder') and len(special_holders) == 1:
|
|
a['holder'] = special_holders[0]
|
|
a['purpose'] = a.get('purpose') or f"{BID}.pr.{special_holders[0].rsplit('.r', 1)[1]}"
|
|
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('block'):
|
|
a['block'] = 'core' # the special budget (social insurance) is entirely basic functions
|
|
|
|
# ------------------------------------------------------------------ residuals: amounts the law shows only at an aggregate level
|
|
purpose_parent = {k: v.get('parent') for k, v in purposes.items()}
|
|
cls_parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
|
|
intra_codes = {f'{k[0]}:{k[1]}' for k, v in classitems.items() if is_intra(v['name'])}
|
|
purpose_holder = {k: v.get('holder') for k, v in purposes.items()}
|
|
calc = Calc(allocs, purpose_parent, cls_parent, intra_codes, blockcodes, purpose_holder)
|
|
def depth(p):
|
|
d = 0
|
|
while p:
|
|
d += 1
|
|
p = purpose_parent.get(p)
|
|
return d
|
|
cands = []
|
|
for c in checks:
|
|
s_ = c['sel']
|
|
if c['annex'] not in (3, 4, 5, 11) or s_.get('kind') or not s_.get('code') or not s_.get('codes') or len(s_['codes']) != 1 or s_.get('flow') == 'revenue':
|
|
continue
|
|
if c['annex'] == 3 and s_['year'] == YEAR:
|
|
continue
|
|
cands.append(c)
|
|
cands.sort(key=lambda c: (-depth(c['sel'].get('purpose')), c['sel'].get('block') is None, c['annex'] in (3,), c['annex']))
|
|
nres = collections.Counter()
|
|
for c in cands:
|
|
s_ = c['sel']
|
|
got = calc.total(s_.get('fund'), s_['flow'], s_['year'], s_['nature'], purpose=s_.get('purpose'), block=s_.get('block'),
|
|
codes=s_['codes'], ckind=s_.get('ckind'), untilEnd=s_.get('untilEnd', False), blk=s_.get('blk'))
|
|
diff = c['value'] - got
|
|
if abs(diff) < 0.5:
|
|
continue
|
|
if not (s_['flow'] == 'financing' or abs(got) < 0.5):
|
|
report['unexplained'].append(f"annex {c['annex']} row {c['row']} {s_['code']} {s_.get('purpose')} {s_['year']}: printed {c['value']} children {got}")
|
|
continue
|
|
sch, code = s_['code'].split(':', 1)
|
|
if is_intra(classitems.get((sch, code), {}).get('name', '')) and s_.get('purpose') is None:
|
|
continue
|
|
pur = s_.get('purpose')
|
|
holder = (purposes.get(pur) or {}).get('holder') if pur else None
|
|
a = dict(srcAnnex=f"p{c['annex']:02d}", flow=s_['flow'], fund=s_.get('fund') or 'basic', nature=s_['nature'], year=s_['year'],
|
|
untilEnd=s_.get('untilEnd', False), holder=holder, purpose=pur, block=s_.get('block'), commitmentKind=s_.get('ckind'),
|
|
scheme=sch, code=code, amount=diff, src=f"{BID}.src.p{c['annex']:02d}", srcRow=c['row'])
|
|
add_alloc(**a)
|
|
calc.add(allocs[-1])
|
|
nres[c['annex']] += 1
|
|
splits = [c for c in checks if c['annex'] == 3 and not c['sel'].get('kind') and c['sel']['year'] == YEAR
|
|
and c['sel'].get('flow') == 'financing' and c['sel'].get('block') and c['sel'].get('code')
|
|
and len(c['sel'].get('codes') or []) == 1]
|
|
splits.sort(key=lambda c: -depth(c['sel'].get('purpose')))
|
|
for c in splits:
|
|
s_ = c['sel']
|
|
got = calc.total(s_.get('fund'), 'financing', YEAR, s_['nature'], purpose=s_.get('purpose'), block=s_['block'], codes=s_['codes'],
|
|
blk=s_.get('blk'))
|
|
diff = c['value'] - got
|
|
if abs(diff) < 0.5:
|
|
continue
|
|
sch, code = s_['code'].split(':', 1)
|
|
pur = s_.get('purpose')
|
|
holder = (purposes.get(pur) or {}).get('holder') if pur else None
|
|
# annex 3 attributes an amount that annex 4 shows without a block to a block: reclassify within the same purpose (totals unchanged)
|
|
for pur_, hol_, blk_, amt in ((pur, holder, s_['block'], diff), (pur, holder, None, -diff)):
|
|
add_alloc(srcAnnex='p03', flow='financing', fund=s_.get('fund') or 'basic', nature=s_['nature'], year=YEAR, holder=hol_,
|
|
purpose=pur_, block=blk_, scheme=sch, code=code, amount=amt, src=f'{BID}.src.p03', srcRow=c['row'])
|
|
calc.add(allocs[-1])
|
|
nres['3-block-split'] += 1
|
|
report['residual_allocations'] = [f'annex {k}: {v}' for k, v in sorted(nres.items(), key=str)]
|
|
|
|
# ------------------------------------------------------------------ sources
|
|
law_title = re.search(r"<div class='TV207'[^>]*>(.*?)</div>", open(LAW_HTML, encoding='utf-8').read(), re.S)
|
|
law_title = html.unescape(re.sub(r'<[^>]+>', '', law_title.group(1))).strip() if law_title else f'Par valsts budžetu {YEAR}. gadam'
|
|
links = json.load(open(os.path.join(ADIR, 'links.json'), encoding='utf-8'))
|
|
sources = [{'id': f'{BID}.src.law', 'kind': 'lawText', 'title': law_title, 'url': links['law'], 'sha256': sha(LAW_HTML)}]
|
|
for n in range(1, 13):
|
|
f = annex_file(n)
|
|
sources.append({'id': f'{BID}.src.p{n:02d}', 'kind': 'form' if n == 12 else 'annex', 'annex': n,
|
|
'title': f'{n}. pielikums', 'url': links['annex'][str(n)], 'sha256': sha(f)})
|
|
|
|
# ------------------------------------------------------------------ write XML
|
|
def attrs(d, order):
|
|
out = []
|
|
for k in order:
|
|
v = d.get(k)
|
|
if v is None or v == '' or v is False:
|
|
continue
|
|
if isinstance(v, bool):
|
|
v = 'true'
|
|
if isinstance(v, float):
|
|
v = (f'{v:.2f}'.rstrip('0').rstrip('.')) if v != int(v) else str(int(v))
|
|
out.append(f'{k}={quoteattr(str(v))}')
|
|
return ' '.join(out)
|
|
|
|
os.makedirs(OUT, exist_ok=True)
|
|
xp = os.path.join(OUT, f'{BID}.xml')
|
|
with open(xp, 'w', encoding='utf-8') as fo:
|
|
fo.write('<?xml version="1.0" encoding="UTF-8"?>\n')
|
|
fo.write(f'<Budget xmlns="urn:pppa:vpk:budzets:0.1" {attrs({"id": BID, "title": law_title, "year": YEAR, "horizonTo": YEAR + 2, "status": "adopted", "act": links["law"], "published": links.get("published"), "version": links.get("version")}, ["id", "title", "year", "horizonTo", "status", "act", "published", "version"])}>\n')
|
|
for s_ in sources:
|
|
fo.write(f' <Source {attrs(s_, ["id", "kind", "annex", "title", "url", "sha256"])}/>\n')
|
|
for p in provisions:
|
|
fo.write(f' <Provision {attrs(p, ["id", "chapter", "article", "src"])}>{escape(p["text"])}</Provision>\n')
|
|
for p in params:
|
|
fo.write(f' <Parameter {attrs(p, ["id", "name", "code", "year", "value", "unit", "src"])}/>\n')
|
|
for aid, a in sorted(actors.items()):
|
|
fo.write(f' <Actor {attrs(dict(a, id=aid), ["id", "kind", "code", "vpk", "name"])}/>\n')
|
|
for pid, p in purposes.items():
|
|
fo.write(f' <Purpose {attrs(dict(p, id=pid), ["id", "kind", "code", "name", "parent", "holder", "block", "function"])}/>\n')
|
|
for (sch, code), c in sorted(classitems.items()):
|
|
ps, par = (c['parent'].split(':', 1) if c['parent'] else (None, None))
|
|
fo.write(f' <ClassItem {attrs({"scheme": sch, "code": code, "name": c["name"], "parent": par, "parentScheme": ps if ps != sch else None, "intraFund": is_intra(c["name"]) or None}, ["scheme", "code", "name", "parent", "parentScheme", "intraFund"])}/>\n')
|
|
for a in allocs:
|
|
a = {k: v for k, v in a.items() if not k.startswith('_')}
|
|
fo.write(f' <Allocation {attrs(a, ["id", "flow", "fund", "nature", "year", "untilEnd", "periodFrom", "periodTo", "holder", "recipient", "purpose", "block", "commitmentKind", "function", "scheme", "code", "amount", "partOf", "src", "srcRow"])}/>\n')
|
|
fo.write('</Budget>\n')
|
|
|
|
json.dump({'checks': checks, 'blockcodes': {k: {f: sorted(v) for f, v in d.items()} for k, d in blockcodes.items()}},
|
|
open(os.path.join(OUT, f'{BID}.checks.json'), 'w', encoding='utf-8'), ensure_ascii=False)
|
|
summary = {'provisions': len(provisions), 'parameters': len(params), 'actors': len(actors), 'purposes': len(purposes),
|
|
'classitems': len(classitems), 'allocations': len(allocs), 'checks': len(checks),
|
|
'allocations_by_annex': collections.Counter(a['src'].rsplit('.', 1)[1] for a in allocs)}
|
|
print(json.dumps(summary, ensure_ascii=False, default=str))
|
|
for k, v in report.items():
|
|
print(f'REPORT {k}: {len(v)}')
|
|
for x in v[:8]:
|
|
print(' ', x)
|
|
json.dump(report, open(os.path.join(OUT, f'{BID}.report.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
|