Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation), nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi. Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML. Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
175 lines
9.6 KiB
Python
175 lines
9.6 KiB
Python
"""Round-trip proof: recompute every printed number of the budget law annexes from the XML alone.
|
|
|
|
Usage: python verify_budget.py OUT_DIR YEAR
|
|
Reads lv-vb-YEAR.xml (validated against the XSD) and lv-vb-YEAR.checks.json (each printed number with its meaning).
|
|
A check passes when the value computed from the XML equals the printed value (difference < 0.5 EUR; < 0.005 for % of GDP).
|
|
"""
|
|
import sys, os, re, json, collections
|
|
TOOLS = os.path.dirname(os.path.abspath(__file__))
|
|
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
|
|
sys.path.insert(0, TOOLS)
|
|
from lxml import etree
|
|
from calc import Calc
|
|
|
|
OUT, YEAR = sys.argv[1], int(sys.argv[2])
|
|
BID = f'lv-vb-{YEAR}'
|
|
NS = '{urn:pppa:vpk:budzets:0.1}'
|
|
schema = etree.XMLSchema(etree.parse(os.path.join(ROOT, 'schemas', 'valsts-budzets-0.1.xsd')))
|
|
doc = etree.parse(os.path.join(OUT, f'{BID}.xml'))
|
|
valid = schema.validate(doc)
|
|
print('XSD valid:', valid)
|
|
for e in list(schema.error_log)[:10]:
|
|
print(' ', e.line, e.message[:200])
|
|
|
|
def norm(s):
|
|
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
|
|
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
|
|
|
|
root = doc.getroot()
|
|
purpose_parent = {p.get('id'): p.get('parent') for p in root.iter(NS + 'Purpose')}
|
|
cls_parent, cls_by_name, intra = {}, collections.defaultdict(set), set()
|
|
for c in root.iter(NS + 'ClassItem'):
|
|
k = f"{c.get('scheme')}:{c.get('code')}"
|
|
if c.get('parent'):
|
|
cls_parent[k] = f"{c.get('parentScheme') or c.get('scheme')}:{c.get('parent')}"
|
|
if c.get('intraFund') == 'true':
|
|
intra.add(k)
|
|
cls_by_name[norm(c.get('name'))].add(k)
|
|
gdp = {int(p.get('year')): float(p.get('value')) for p in root.iter(NS + 'Parameter') if p.get('code') == 'IKP'}
|
|
A = []
|
|
for a in root.iter(NS + 'Allocation'):
|
|
d = dict(a.attrib)
|
|
d['year'] = int(d['year'])
|
|
d['untilEnd'] = d.get('untilEnd') == 'true'
|
|
A.append(d)
|
|
data = json.load(open(os.path.join(OUT, f'{BID}.checks.json'), encoding='utf-8'))
|
|
purpose_holder = {p.get('id'): p.get('holder') for p in root.iter(NS + 'Purpose')}
|
|
calc = Calc(A, purpose_parent, cls_parent, intra, data['blockcodes'], purpose_holder)
|
|
print('allocations:', len(A))
|
|
|
|
res = collections.defaultdict(lambda: {'n': 0, 'ok': 0, 'bad': [], 'skip': 0})
|
|
def judge(c, got, tol=0.5):
|
|
r = res[c['annex']]
|
|
r['n'] += 1
|
|
if got is None:
|
|
r['skip'] += 1
|
|
r['bad'].append(('not computed', c['row'], c['sel'].get('label', ''), c['value'], None))
|
|
elif abs(got - c['value']) < tol:
|
|
r['ok'] += 1
|
|
else:
|
|
r['bad'].append(('mismatch', c['row'], json.dumps({k: v for k, v in c['sel'].items() if k != 'codes'}, ensure_ascii=False)[:230], c['value'], round(got, 2)))
|
|
|
|
# ------------------------------------------------------------------ annex 1: consolidated budget by formula
|
|
def a1_value(label, year, ctx, pct):
|
|
n = norm(label)
|
|
base = 'appropriation' if year == YEAR else 'ceiling'
|
|
T = calc.total
|
|
def rev(f): return T(f, 'revenue', year, 'forecast')
|
|
def exp(f, pre=None): return T(f, 'expenditure', year, base, ekk_prefix=pre)
|
|
def fin(f, codes=None): return T(f, 'financing', year, base, codes=codes)
|
|
def cap(f): return exp(f, '5') + exp(f, '9')
|
|
b2s_m, b2s_c, s2b_m, s2b_c = exp('basic', '712'), exp('basic', '912'), exp('special', '711'), exp('special', '911')
|
|
PA, SA = rev('basic') - (s2b_m + s2b_c), rev('special') - (b2s_m + b2s_c)
|
|
PB, SB = exp('basic') - (b2s_m + b2s_c), exp('special') - (s2b_m + s2b_c)
|
|
PB2, SB2 = cap('basic') - b2s_c, cap('special') - s2b_c
|
|
PB1, SB1 = PB - PB2, SB - SB2
|
|
if pct:
|
|
g = gdp.get(year)
|
|
if not g:
|
|
return None
|
|
val = {'valsts budžeta ieņēmumi': PA + SA, 'valsts budžeta izdevumi': PB + SB, 'valsts budžeta finansiālā bilance': PA + SA - PB - SB}
|
|
for k, v in val.items():
|
|
if n.startswith(k):
|
|
return round(v / (g * 1e6) * 100, 2)
|
|
return None
|
|
first = n.split(' ')[0] if n else ''
|
|
exact = {'ka': PA + SA, 'pa': PA, 'sa': SA, 'kb': PB + SB, 'kb1': PB1 + SB1, 'kb2': PB2 + SB2,
|
|
'pb': PB, 'pb1': PB1, 'pb2': PB2, 'sb': SB, 'sb1': SB1, 'sb2': SB2}
|
|
if first in exact:
|
|
return exact[first]
|
|
prefix = [
|
|
('valsts pamatbudžeta ieņēmumi', rev('basic')), ('valsts speciālā budžeta ieņēmumi', rev('special')),
|
|
('valsts pamatbudžeta izdevumi', exp('basic')), ('valsts speciālā budžeta izdevumi', exp('special')),
|
|
('valsts pamatbudžeta uzturēšanas izdevumi', exp('basic') - cap('basic')), ('valsts pamatbudžeta kapitālie izdevumi', cap('basic')),
|
|
('valsts speciālā budžeta uzturēšanas izdevumi', exp('special') - cap('special')), ('valsts speciālā budžeta kapitālie izdevumi', cap('special')),
|
|
('valsts budžeta finansiālā bilance', PA + SA - PB - SB),
|
|
('valsts pamatbudžeta finansiālā bilance', rev('basic') - exp('basic')),
|
|
('valsts speciālā budžeta finansiālā bilance', rev('special') - exp('special')),
|
|
]
|
|
for k, v in prefix:
|
|
if n.startswith(k):
|
|
return v
|
|
if n.startswith('mīnus transferts no valsts speciālā'): return s2b_m + s2b_c
|
|
if n.startswith('mīnus transferts no valsts pamatbudžeta'): return b2s_m + b2s_c
|
|
if n.startswith('mīnus transferts valsts speciāl'):
|
|
return {'gross': b2s_m + b2s_c, 'maint': b2s_m, 'cap': b2s_c}[ctx['exp_part']]
|
|
if n.startswith('mīnus transferts valsts pamatbudžet'):
|
|
return {'gross': s2b_m + s2b_c, 'maint': s2b_m, 'cap': s2b_c}[ctx['exp_part']]
|
|
f = {'fin_all': None, 'fin_basic': 'basic', 'fin_special': 'special'}.get(ctx['part'])
|
|
if n == 'finansēšana' and ctx['part'].startswith('fin'):
|
|
return fin(f)
|
|
if not ctx['part'].startswith('fin'):
|
|
a2 = a2_codes.get((ctx['fund'], n))
|
|
if a2:
|
|
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=a2, blk=f"a2:{ctx['fund']}")
|
|
codes = cls_by_name.get(n)
|
|
if not codes:
|
|
return None
|
|
if ctx['part'].startswith('fin'):
|
|
return fin(f, codes)
|
|
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=codes)
|
|
|
|
a2_codes = {}
|
|
for c in data['checks']:
|
|
if c['annex'] == 2 and c['sel'].get('label') and c['sel'].get('codes'):
|
|
a2_codes.setdefault((c['sel']['fund'], norm(c['sel']['label'])), c['sel']['codes'])
|
|
ctx = {'fund': 'basic', 'part': 'rev', 'exp_part': 'gross'}
|
|
for c in sorted([c for c in data['checks'] if c['annex'] == 1], key=lambda c: (c['row'], c['sel']['year'])):
|
|
n = norm(c['sel']['label'])
|
|
if n.startswith('valsts pamatbudžeta ieņēmumi'): ctx.update(fund='basic', part='rev')
|
|
if n.startswith('valsts speciālā budžeta ieņēmumi'): ctx.update(fund='special', part='rev')
|
|
if n.startswith('valsts budžeta finansiālā bilance'): ctx.update(part='fin_all')
|
|
if n.startswith('valsts pamatbudžeta finansiālā bilance'): ctx.update(part='fin_basic')
|
|
if n.startswith('valsts speciālā budžeta finansiālā bilance'): ctx.update(part='fin_special')
|
|
if n.startswith(('valsts pamatbudžeta', 'valsts speciālā budžeta')) and 'izdevumi' in n:
|
|
ctx.update(exp_part='maint' if 'uzturēšanas' in n else ('cap' if 'kapitālie' in n else 'gross'))
|
|
judge(c, a1_value(c['sel']['label'], c['sel']['year'], ctx, c['sel']['pct']), tol=0.005 if c['sel']['pct'] else 0.5)
|
|
|
|
# ------------------------------------------------------------------ all other annexes
|
|
for c in data['checks']:
|
|
if c['annex'] == 1:
|
|
continue
|
|
s = c['sel']
|
|
kind = s.get('kind')
|
|
if kind == 'grant_total':
|
|
got = sum(float(a['amount']) for a in A if a.get('purpose') == s['purpose'] and f"{a['scheme']}:{a['code']}" == s['code']
|
|
and (not s.get('periodFrom') or a.get('periodFrom') == s['periodFrom']))
|
|
judge(c, got); continue
|
|
if kind in ('fees_resort', 'fees_total'):
|
|
got = sum(float(a['amount']) for a in A if a.get('partOf') and a['src'].endswith('.p02') and a['year'] == s['year']
|
|
and (kind == 'fees_total' or a.get('holder') == s['holder']))
|
|
judge(c, got); continue
|
|
common = dict(purpose=s.get('purpose'), block=s.get('block'), ckind=s.get('ckind'), untilEnd=s.get('untilEnd', False), blk=s.get('blk'),
|
|
holder_view=bool(s.get('holderView')))
|
|
if kind == 'balance':
|
|
printed = data['blockcodes'].get(s.get('blk'), {})
|
|
if c['annex'] != 11 and ((s.get('purpose') is None and not s.get('block')) or s.get('fund') == 'special'):
|
|
inflow = calc.total(s.get('fund'), 'revenue', s['year'], 'forecast', **common)
|
|
else:
|
|
inflow = calc.total(s.get('fund'), 'resource', s['year'], s['nature'], **common)
|
|
judge(c, inflow - calc.total(s.get('fund'), 'expenditure', s['year'], s['nature'], **common)); continue
|
|
judge(c, calc.total(s.get('fund'), s['flow'], s['year'], s['nature'], codes=s.get('codes'), **common))
|
|
|
|
# ------------------------------------------------------------------ report
|
|
tot_n = tot_ok = 0
|
|
summary = {}
|
|
for an in sorted(res):
|
|
r = res[an]
|
|
tot_n += r['n']; tot_ok += r['ok']
|
|
summary[an] = {'checks': r['n'], 'ok': r['ok'], 'mismatch': r['n'] - r['ok'] - r['skip'], 'not_computed': r['skip']}
|
|
print(f"annex {an:2d}: {r['n']:6d} printed numbers, {r['ok']:6d} reproduced ({100 * r['ok'] / max(r['n'], 1):.2f}%), {r['skip']} not computed")
|
|
print(f'TOTAL: {tot_ok}/{tot_n} printed numbers reproduced from XML ({100 * tot_ok / max(tot_n, 1):.3f}%)')
|
|
json.dump({'xsd_valid': valid, 'allocations': len(A), 'annexes': summary, 'total': tot_n, 'reproduced': tot_ok,
|
|
'issues': {an: r['bad'][:300] for an, r in res.items()}},
|
|
open(os.path.join(OUT, f'{BID}.verify.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)
|