1
0
Files
Rihards Gailums 71250e7b0e Valsts budžets kā kods v0.1: shēma, 2025. un 2026. gada budžeta likumi kā dati
Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation),
nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi.
Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML.
Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
2026-10-10 22:34:28 +00:00

175 lines
9.6 KiB
Python

"""Round-trip proof: recompute every printed number of the budget law annexes from the XML alone.
Usage: python verify_budget.py OUT_DIR YEAR
Reads lv-vb-YEAR.xml (validated against the XSD) and lv-vb-YEAR.checks.json (each printed number with its meaning).
A check passes when the value computed from the XML equals the printed value (difference < 0.5 EUR; < 0.005 for % of GDP).
"""
import sys, os, re, json, collections
TOOLS = os.path.dirname(os.path.abspath(__file__))
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
sys.path.insert(0, TOOLS)
from lxml import etree
from calc import Calc
OUT, YEAR = sys.argv[1], int(sys.argv[2])
BID = f'lv-vb-{YEAR}'
NS = '{urn:pppa:vpk:budzets:0.1}'
schema = etree.XMLSchema(etree.parse(os.path.join(ROOT, 'schemas', 'valsts-budzets-0.1.xsd')))
doc = etree.parse(os.path.join(OUT, f'{BID}.xml'))
valid = schema.validate(doc)
print('XSD valid:', valid)
for e in list(schema.error_log)[:10]:
print(' ', e.line, e.message[:200])
def norm(s):
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
root = doc.getroot()
purpose_parent = {p.get('id'): p.get('parent') for p in root.iter(NS + 'Purpose')}
cls_parent, cls_by_name, intra = {}, collections.defaultdict(set), set()
for c in root.iter(NS + 'ClassItem'):
k = f"{c.get('scheme')}:{c.get('code')}"
if c.get('parent'):
cls_parent[k] = f"{c.get('parentScheme') or c.get('scheme')}:{c.get('parent')}"
if c.get('intraFund') == 'true':
intra.add(k)
cls_by_name[norm(c.get('name'))].add(k)
gdp = {int(p.get('year')): float(p.get('value')) for p in root.iter(NS + 'Parameter') if p.get('code') == 'IKP'}
A = []
for a in root.iter(NS + 'Allocation'):
d = dict(a.attrib)
d['year'] = int(d['year'])
d['untilEnd'] = d.get('untilEnd') == 'true'
A.append(d)
data = json.load(open(os.path.join(OUT, f'{BID}.checks.json'), encoding='utf-8'))
purpose_holder = {p.get('id'): p.get('holder') for p in root.iter(NS + 'Purpose')}
calc = Calc(A, purpose_parent, cls_parent, intra, data['blockcodes'], purpose_holder)
print('allocations:', len(A))
res = collections.defaultdict(lambda: {'n': 0, 'ok': 0, 'bad': [], 'skip': 0})
def judge(c, got, tol=0.5):
r = res[c['annex']]
r['n'] += 1
if got is None:
r['skip'] += 1
r['bad'].append(('not computed', c['row'], c['sel'].get('label', ''), c['value'], None))
elif abs(got - c['value']) < tol:
r['ok'] += 1
else:
r['bad'].append(('mismatch', c['row'], json.dumps({k: v for k, v in c['sel'].items() if k != 'codes'}, ensure_ascii=False)[:230], c['value'], round(got, 2)))
# ------------------------------------------------------------------ annex 1: consolidated budget by formula
def a1_value(label, year, ctx, pct):
n = norm(label)
base = 'appropriation' if year == YEAR else 'ceiling'
T = calc.total
def rev(f): return T(f, 'revenue', year, 'forecast')
def exp(f, pre=None): return T(f, 'expenditure', year, base, ekk_prefix=pre)
def fin(f, codes=None): return T(f, 'financing', year, base, codes=codes)
def cap(f): return exp(f, '5') + exp(f, '9')
b2s_m, b2s_c, s2b_m, s2b_c = exp('basic', '712'), exp('basic', '912'), exp('special', '711'), exp('special', '911')
PA, SA = rev('basic') - (s2b_m + s2b_c), rev('special') - (b2s_m + b2s_c)
PB, SB = exp('basic') - (b2s_m + b2s_c), exp('special') - (s2b_m + s2b_c)
PB2, SB2 = cap('basic') - b2s_c, cap('special') - s2b_c
PB1, SB1 = PB - PB2, SB - SB2
if pct:
g = gdp.get(year)
if not g:
return None
val = {'valsts budžeta ieņēmumi': PA + SA, 'valsts budžeta izdevumi': PB + SB, 'valsts budžeta finansiālā bilance': PA + SA - PB - SB}
for k, v in val.items():
if n.startswith(k):
return round(v / (g * 1e6) * 100, 2)
return None
first = n.split(' ')[0] if n else ''
exact = {'ka': PA + SA, 'pa': PA, 'sa': SA, 'kb': PB + SB, 'kb1': PB1 + SB1, 'kb2': PB2 + SB2,
'pb': PB, 'pb1': PB1, 'pb2': PB2, 'sb': SB, 'sb1': SB1, 'sb2': SB2}
if first in exact:
return exact[first]
prefix = [
('valsts pamatbudžeta ieņēmumi', rev('basic')), ('valsts speciālā budžeta ieņēmumi', rev('special')),
('valsts pamatbudžeta izdevumi', exp('basic')), ('valsts speciālā budžeta izdevumi', exp('special')),
('valsts pamatbudžeta uzturēšanas izdevumi', exp('basic') - cap('basic')), ('valsts pamatbudžeta kapitālie izdevumi', cap('basic')),
('valsts speciālā budžeta uzturēšanas izdevumi', exp('special') - cap('special')), ('valsts speciālā budžeta kapitālie izdevumi', cap('special')),
('valsts budžeta finansiālā bilance', PA + SA - PB - SB),
('valsts pamatbudžeta finansiālā bilance', rev('basic') - exp('basic')),
('valsts speciālā budžeta finansiālā bilance', rev('special') - exp('special')),
]
for k, v in prefix:
if n.startswith(k):
return v
if n.startswith('mīnus transferts no valsts speciālā'): return s2b_m + s2b_c
if n.startswith('mīnus transferts no valsts pamatbudžeta'): return b2s_m + b2s_c
if n.startswith('mīnus transferts valsts speciāl'):
return {'gross': b2s_m + b2s_c, 'maint': b2s_m, 'cap': b2s_c}[ctx['exp_part']]
if n.startswith('mīnus transferts valsts pamatbudžet'):
return {'gross': s2b_m + s2b_c, 'maint': s2b_m, 'cap': s2b_c}[ctx['exp_part']]
f = {'fin_all': None, 'fin_basic': 'basic', 'fin_special': 'special'}.get(ctx['part'])
if n == 'finansēšana' and ctx['part'].startswith('fin'):
return fin(f)
if not ctx['part'].startswith('fin'):
a2 = a2_codes.get((ctx['fund'], n))
if a2:
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=a2, blk=f"a2:{ctx['fund']}")
codes = cls_by_name.get(n)
if not codes:
return None
if ctx['part'].startswith('fin'):
return fin(f, codes)
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=codes)
a2_codes = {}
for c in data['checks']:
if c['annex'] == 2 and c['sel'].get('label') and c['sel'].get('codes'):
a2_codes.setdefault((c['sel']['fund'], norm(c['sel']['label'])), c['sel']['codes'])
ctx = {'fund': 'basic', 'part': 'rev', 'exp_part': 'gross'}
for c in sorted([c for c in data['checks'] if c['annex'] == 1], key=lambda c: (c['row'], c['sel']['year'])):
n = norm(c['sel']['label'])
if n.startswith('valsts pamatbudžeta ieņēmumi'): ctx.update(fund='basic', part='rev')
if n.startswith('valsts speciālā budžeta ieņēmumi'): ctx.update(fund='special', part='rev')
if n.startswith('valsts budžeta finansiālā bilance'): ctx.update(part='fin_all')
if n.startswith('valsts pamatbudžeta finansiālā bilance'): ctx.update(part='fin_basic')
if n.startswith('valsts speciālā budžeta finansiālā bilance'): ctx.update(part='fin_special')
if n.startswith(('valsts pamatbudžeta', 'valsts speciālā budžeta')) and 'izdevumi' in n:
ctx.update(exp_part='maint' if 'uzturēšanas' in n else ('cap' if 'kapitālie' in n else 'gross'))
judge(c, a1_value(c['sel']['label'], c['sel']['year'], ctx, c['sel']['pct']), tol=0.005 if c['sel']['pct'] else 0.5)
# ------------------------------------------------------------------ all other annexes
for c in data['checks']:
if c['annex'] == 1:
continue
s = c['sel']
kind = s.get('kind')
if kind == 'grant_total':
got = sum(float(a['amount']) for a in A if a.get('purpose') == s['purpose'] and f"{a['scheme']}:{a['code']}" == s['code']
and (not s.get('periodFrom') or a.get('periodFrom') == s['periodFrom']))
judge(c, got); continue
if kind in ('fees_resort', 'fees_total'):
got = sum(float(a['amount']) for a in A if a.get('partOf') and a['src'].endswith('.p02') and a['year'] == s['year']
and (kind == 'fees_total' or a.get('holder') == s['holder']))
judge(c, got); continue
common = dict(purpose=s.get('purpose'), block=s.get('block'), ckind=s.get('ckind'), untilEnd=s.get('untilEnd', False), blk=s.get('blk'),
holder_view=bool(s.get('holderView')))
if kind == 'balance':
printed = data['blockcodes'].get(s.get('blk'), {})
if c['annex'] != 11 and ((s.get('purpose') is None and not s.get('block')) or s.get('fund') == 'special'):
inflow = calc.total(s.get('fund'), 'revenue', s['year'], 'forecast', **common)
else:
inflow = calc.total(s.get('fund'), 'resource', s['year'], s['nature'], **common)
judge(c, inflow - calc.total(s.get('fund'), 'expenditure', s['year'], s['nature'], **common)); continue
judge(c, calc.total(s.get('fund'), s['flow'], s['year'], s['nature'], codes=s.get('codes'), **common))
# ------------------------------------------------------------------ report
tot_n = tot_ok = 0
summary = {}
for an in sorted(res):
r = res[an]
tot_n += r['n']; tot_ok += r['ok']
summary[an] = {'checks': r['n'], 'ok': r['ok'], 'mismatch': r['n'] - r['ok'] - r['skip'], 'not_computed': r['skip']}
print(f"annex {an:2d}: {r['n']:6d} printed numbers, {r['ok']:6d} reproduced ({100 * r['ok'] / max(r['n'], 1):.2f}%), {r['skip']} not computed")
print(f'TOTAL: {tot_ok}/{tot_n} printed numbers reproduced from XML ({100 * tot_ok / max(tot_n, 1):.3f}%)')
json.dump({'xsd_valid': valid, 'allocations': len(A), 'annexes': summary, 'total': tot_n, 'reproduced': tot_ok,
'issues': {an: r['bad'][:300] for an, r in res.items()}},
open(os.path.join(OUT, f'{BID}.verify.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)