1
0

Valsts budžets kā kods v0.1: shēma, 2025. un 2026. gada budžeta likumi kā dati

Loģisks budžeta modelis (Source, Provision, Parameter, Actor, Purpose, Indicator, Rule, ClassItem, Allocation),
nevis likuma pielikumu izkārtojuma kopija. Abi likumi: teksts un visi 12 pielikumi.
Atpakaļsaderība: 2026 — 58 457 no 58 457, 2025 — 55 677 no 55 677 pielikumos drukāto skaitļu atjaunoti tikai no XML.
Avoti (likumi.lv, klasifikāciju MK noteikumi), rīki, datu līgums B15, dokumentācija.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
2026-10-10 22:34:28 +00:00
commit 71250e7b0e
56 changed files with 26144 additions and 0 deletions

90
tools/calc.py Normal file
View File

@@ -0,0 +1,90 @@
"""Shared summation rules for the budget model (used by the parser for residuals and by the verifier).
A printed number in the law is reproduced by summing Allocation records that match its selector:
fund, flow, year, nature, untilEnd, purpose subtree, block, commitment kind, and classification codes.
Consolidation: a node's total includes an intra-fund transfer only if that node itself prints the transfer's code;
otherwise intra-fund transfers are eliminated (they cancel out inside the node).
"""
import collections
def kind_stem(k):
"""Commitment-kind codes are hierarchical in 2-digit groups: 0101100000 -> 010110."""
s = k or ''
while len(s) > 2 and s.endswith('00'):
s = s[:-2]
return s
class Calc:
def __init__(self, allocs, purpose_parent, cls_parent, intra_codes, blockcodes, purpose_holder=None):
self.purpose_parent = purpose_parent
self.purpose_holder = purpose_holder or {}
self.cls_parent = cls_parent
self.intra = intra_codes
self.blockcodes = {k: {f: set(v) for f, v in d.items()} for k, d in blockcodes.items()}
self.idx = collections.defaultdict(list)
for a in allocs:
self.add(a)
def chain(self, p):
out = set()
while p and p not in out:
out.add(p)
p = self.purpose_parent.get(p)
return out
def code_chain(self, k):
out = set()
while k and k not in out:
out.add(k)
k = self.cls_parent.get(k)
return out
def add(self, a):
a['_pchain'] = self.chain(a.get('purpose'))
a['_ck'] = f"{a['scheme']}:{a['code']}"
a['_cchain'] = self.code_chain(a['_ck'])
a['_intra'] = bool(a['_cchain'] & self.intra)
self.idx[(a['fund'], a['flow'], int(a['year']), a['nature'])].append(a)
def total(self, fund, flow, year, nature, purpose=None, block=None, codes=None, ckind=None, untilEnd=False,
blk=None, ekk_prefix=None, consolidated=None, holder_view=False):
"""holder_view: also count budget-level amounts (no purpose) whose holder administers the selected purpose."""
printed = self.blockcodes.get(blk, {}).get(flow) if blk else None
if consolidated is None:
consolidated = purpose is None
codes = set(codes) if codes else None
s = 0.0
for f in ([fund] if fund else ['basic', 'special']):
for a in self.idx.get((f, flow, int(year), nature), ()):
if a.get('partOf'):
continue
if bool(a.get('untilEnd')) != bool(untilEnd):
continue
if purpose and purpose not in a['_pchain']:
if not (holder_view and not a.get('purpose') and a.get('holder') and a.get('holder') == self.purpose_holder.get(purpose)):
continue
if block and a.get('block') != block:
continue
if ckind and not (a.get('commitmentKind') or '').startswith(kind_stem(ckind)):
continue
if ekk_prefix and not (a['scheme'] == 'ekk' and a['code'].startswith(ekk_prefix)):
continue
ck, cc = a['_ck'], a['_cchain']
if printed is not None:
shown = ck in printed
if not shown and a['_intra']:
continue
if codes is not None:
if not (ck in codes or (not shown and cc & codes)):
continue
elif codes is not None:
if not (cc & codes):
continue
if consolidated and a['_intra'] and ck not in codes:
continue
elif consolidated and a['_intra']:
continue
s += float(a['amount'])
return s

26
tools/klasifikacijas.py Normal file
View File

@@ -0,0 +1,26 @@
"""Classification code lists from the regulation tables on likumi.lv:
MK 27.12.2005. noteikumi Nr. 1031 (expenditure, EKK), MK noteikumi par budžetu ieņēmumu klasifikāciju (revenue), MK 22.11.2005. noteikumi Nr. 875 (financing).
Input: source/klasifikacijas/<likumi.lv id>.html Output: source/kodi/klasdict.json (name -> [[scheme, code], ...])
"""
import re, html, json, os
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
out = {}
CODE = re.compile(r'(?:[A-Z]{1,2})?\d{1,2}(?:\.\d{1,2}){1,5}\.?|\d{4,5}|F\d{8}|[A-Z]\d{1,2}(?:\.\d+)*\.?')
for act, kind in (('124833', 'ekk'), ('124831', 'revenue'), ('122159', 'financing')):
s = open(os.path.join(ROOT, 'source', 'klasifikacijas', f'{act}.html'), encoding='utf-8', errors='replace').read()
n = 0
for tr in re.findall(r'(?is)<tr[^>]*>(.*?)</tr>', s):
cells = [re.sub(r'\s+', ' ', html.unescape(re.sub(r'<[^>]+>', ' ', c))).strip() for c in re.findall(r'(?is)<td[^>]*>(.*?)</td>', tr)]
for i, c in enumerate(cells[:-1]):
if CODE.fullmatch(c) and cells[i + 1] and not CODE.fullmatch(cells[i + 1]) and not cells[i + 1].startswith('Kodā'):
code = c.rstrip('.')
if kind == 'revenue' and '.' in code:
p = code.split('.'); code = (p[0].zfill(2) + ''.join(p[1:])).ljust(5, '0')
if kind == 'financing' and '.' in code:
code = 'F' + code.replace('.', '')
out.setdefault(cells[i + 1].strip(' .;:'), set()).add((kind, code)); n += 1
break
print(act, kind, 'rows', n)
for k in [k for k in out if re.search(r'Dotācija no vispār|savstarpējie|Kapitālo izdevumu transferti$|7200|uz valsts pamatbudžetu', k)][:12]:
print(' ', sorted(out[k])[:3], k[:100])
json.dump({k: sorted(v) for k, v in out.items()}, open(os.path.join(ROOT, 'source', 'kodi', 'klasdict.json'), 'w'), ensure_ascii=False)

23
tools/kodi_no_sap.py Normal file
View File

@@ -0,0 +1,23 @@
"""Code-name pairs from the hidden SAP BW (BEx) sheets inside the budget law annex files.
The printed annex tables show line names only; the hidden ZQZBC_* sheets carry the code next to the name.
Input: source/<year>/P01.XLSX, P02.XLSX, P03.XLSX, P05.XLSX, P11.XLSX Output: source/kodi/codepairs.json (name -> [codes])
Usage: python tools/kodi_no_sap.py [year] (default 2026)
"""
import sys, os, re, json, collections
TOOLS = os.path.dirname(os.path.abspath(__file__))
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
sys.path.insert(0, TOOLS)
from rd import sheets
year = sys.argv[1] if len(sys.argv) > 1 else '2026'
pairs = collections.OrderedDict()
for f in ('P01.XLSX', 'P02.XLSX', 'P05.XLSX', 'P03.XLSX', 'P11.XLSX'):
for name, rows in sheets(os.path.join(ROOT, 'source', year, f)):
if not name.startswith('ZQZ'):
continue
for r in rows:
for i in range(len(r) - 1):
c, nm = str(r[i]), r[i + 1]
if re.fullmatch(r'[A-Z]{0,3}\d[\dA-Z]*|[A-Z]{1,5}\d?T?|\d{4,5}', c) and isinstance(nm, str) and len(nm) > 3 and not nm.startswith('Nosaukums'):
pairs.setdefault(nm.strip(), set()).add(c)
json.dump({k: sorted(v) for k, v in pairs.items()}, open(os.path.join(ROOT, 'source', 'kodi', 'codepairs.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=0)
print('pairs', len(pairs))

32
tools/odcs-config.yaml Normal file
View File

@@ -0,0 +1,32 @@
# B15 datu līguma konfigurācija lakehouse rīkam tools/xsd_to_odcs.py (tas pats rīks, ar ko ģenerēti B01–B14).
# python3 tools/xsd_to_odcs.py <šī datne> <repozitorija sakne> contracts Budget-as-Code
base_url: https://processgit.org
contracts:
- id: B15
file: B15-valsts-budzets.odcs.yaml
name: Valsts budžets (budžeta likums kā dati)
version: 0.1.0
org: Valsts-Pirmkods
repo: Budget-as-Code
pattern: data/lv-vb-*.xml
filename_is_id: true
schema: schemas/valsts-budzets-0.1.xsd
root: Budget
namespace: urn:pppa:vpk:budzets:0.1
owner: PPP Asociācija (PPPA)
data_product: Valsts budžeta datukopa
purpose: Pieņemtais valsts budžeta likums kā dati — panti, ieņēmumi, apropriācijas pa programmām un apakšprogrammām, nākamo gadu maksimālie apjomi, ilgtermiņa saistības un mērķdotācijas pašvaldībām; katrs likuma pielikumos drukātais skaitlis atjaunojams no datnes.
limitations: Koncepcijas demonstrācija, nav oficiāls izdevums. Budžeta paskaidrojumi (mērķi, rezultatīvie rādītāji) un gada laikā veiktās apropriācijas izmaiņas vēl nav iekļautas.
identifiers: "lv-vb-GGGG; pants .pNNN; resors .ac.rSS; programma .pr.SS.PP.AA.00; projekts .pj.…; naudas fakts .a.pNN.r<rinda>.y<gads>"
objects:
- {element: Budget, keys: [id]}
- {element: Source, keys: [id]}
- {element: Provision, keys: [id]}
- {element: Parameter, keys: [id]}
- {element: Actor, keys: [id]}
- {element: Purpose, keys: [id]}
- {element: ClassItem, keys: [scheme, code]}
- {element: Allocation, keys: [id]}
rules:
- {rule: printed_numbers_reproduced, description: "Katrs likuma pielikumos drukātais skaitlis atjaunojams no datnes (tools/verify_budget.py)", dimension: accuracy}
- {rule: resort_has_vpk_id, description: "Resoram, kas ir institūcija, norādīts VPK ID"}

862
tools/parse_budget.py Normal file
View File

@@ -0,0 +1,862 @@
"""Convert an adopted Latvian state budget law (text + 12 annexes) into one XML file
following valsts-budzets-0.1.xsd, plus a JSON list of every printed number as a check.
Usage: python parse_budget.py YEAR LAW_HTML ANNEX_DIR OUT_DIR
Annex files are found by number (P04.XLSX, 4_PIELIKUMS.XLS, ...).
Storage rule: each fact is stored once, from its most detailed source; every other printed number becomes a check.
"""
import sys, os, re, json, html, hashlib, collections, datetime
TOOLS = os.path.dirname(os.path.abspath(__file__))
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
KODI = os.path.join(ROOT, 'source', 'kodi')
sys.path.insert(0, TOOLS)
from rows import read_rows
from rd import docx_tables, num
from xml.sax.saxutils import quoteattr, escape
from calc import Calc
YEAR, LAW_HTML, ADIR, OUT = int(sys.argv[1]), sys.argv[2], sys.argv[3], sys.argv[4]
Y3 = [YEAR, YEAR + 1, YEAR + 2]
BID = f'lv-vb-{YEAR}'
report = collections.defaultdict(list)
# ------------------------------------------------------------------ helpers
def norm(s):
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
s = re.sub(r'^i estādes', 'iestādes', s)
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
def slug(s, n=40):
t = norm(s).translate(str.maketrans('āčēģīķļņšūž', 'acegiklnsuz'))
return re.sub(r'\s+', '-', t)[:n].strip('-')
def annex_file(n):
for f in os.listdir(ADIR):
m = re.match(r'^P?0*(\d+)[_.]', f, re.I)
if m and int(m.group(1)) == n:
return os.path.join(ADIR, f)
raise FileNotFoundError(f'annex {n}')
def sha(path):
return hashlib.sha256(open(path, 'rb').read()).hexdigest()
# ------------------------------------------------------------------ code book
SCHEME_OF_FLOW = {'expenditure': 'ekk', 'resource': 'revenue', 'revenue': 'revenue', 'financing': 'financing'}
book = collections.defaultdict(lambda: collections.defaultdict(set)) # scheme -> normname -> codes
for nm, kc in json.load(open(os.path.join(KODI, 'klasdict.json'), encoding='utf-8')).items():
for kind, c in kc:
book[kind][norm(nm)].add(c)
KL = {'Izdevumi': 'ekk', 'Ieņēmumi': 'revenue', 'Finansēšana': 'financing'}
for line in open(os.path.join(KODI, 'vk_codes.tsv'), encoding='utf-8'):
kl, v = line.rstrip('\n').split('\t')
m = re.match(r'^([A-Z]{0,2}\d[\d.]*)\s+(.+)$', v)
if m and kl in KL:
book[KL[kl]][norm(m.group(2))].add(m.group(1))
for nm, codes in json.load(open(os.path.join(KODI, 'codepairs.json'), encoding='utf-8')).items():
for c in codes:
sch = 'financing' if c.startswith('F') or re.fullmatch(r'[A-Z]{1,2}F\d+', c) else ('ekk' if re.fullmatch(r'\d{4}', c) else ('revenue' if re.fullmatch(r'\d{5}', c) else 'law'))
book[sch][norm(nm)].add(c)
classitems = {} # (scheme, code) -> {'name', 'parent'}
SECTION = {'ieņēmumi kopā': 'revenue', 'resursi izdevumu segšanai': 'resource', 'izdevumi kopā': 'expenditure',
'finansēšana': 'financing', 'finansiālā bilance': 'balance'}
def canon(sch, c):
if sch == 'financing' and re.fullmatch(r'[PKS]F\d{8}', c):
return c[1:]
return c
def code_for(label, flow):
n = norm(label)
pref = SCHEME_OF_FLOW.get(flow, 'law')
for sch in (pref, 'law', 'revenue', 'ekk', 'financing'):
cs = book[sch].get(n)
if cs:
cs = {canon(sch, x) for x in cs}
c = sorted(cs, key=lambda x: (len(x), x))[0]
if len(cs) > 1:
report['ambiguous_code'].append(f'{label[:60]} -> {sorted(cs)} (took {c})')
return sch, c
c = 'L-' + slug(label)
report['synthesised_code'].append(label)
return 'law', c
INTRA = ('savstarpējie transferti', 'no valsts pamatbudžeta uz valsts pamatbudžetu', 'no valsts speciālā budžeta uz valsts speciālo budžetu',
'atmaksām valsts pamatbudžet', 'atmaksa valsts budžetā par veiktajiem', 'valsts pamatbudžeta iestāžu saņemtie transferti no valsts pamatbudžeta',
'pārējie valsts pamatbudžetā saņemtie transferti no valsts pamatbudžeta', 'valsts speciālā budžeta iestāžu saņemtie transferti no valsts speciālā budžeta')
def is_intra(name):
n = norm(name)
return any(norm(k) in n for k in INTRA)
VOTE_WEIGHT = {'a2': 1, 'a3': 1, 'a4': 1, 'a5': 1, 'a11': 0}
CUR_ANNEX = ['a4']
def register_class(sch, code, name, parent):
k = (sch, code)
d = classitems.setdefault(k, {'name': name, 'parent': None, 'votes': collections.Counter()})
w = VOTE_WEIGHT.get(CUR_ANNEX[0], 1)
if w:
d['votes'][parent] += w
def structural_ok(child, parent):
"""Numeric classification codes carry their own hierarchy: 21200 cannot sit under 21100, 7131 can sit under 7130."""
cs, cc = child.split(':', 1)
ps, pc = parent.split(':', 1)
if not (cc.isdigit() and pc.isdigit() and cs == ps and len(cc) == len(pc)):
return True
stem = pc.rstrip('0')
return cc != pc and cc.startswith(stem)
def finalize_classes():
"""Canonical parent = majority vote over indented annex blocks, restricted to structurally possible parents."""
for k, d in classitems.items():
votes = d['votes']
ck = f'{k[0]}:{k[1]}'
real = [(p, n) for p, n in votes.most_common() if p and structural_ok(ck, p)]
rejected = [p for p in votes if p and not structural_ok(ck, p)]
if rejected:
report['parent_rejected_by_structure'].append(f'{ck} {d["name"][:40]}: not under {rejected}')
d['parent'] = real[0][0] if real and (not votes.get(None) or real[0][1] >= votes[None] or rejected) else None
if len(real) > 1:
report['class_parent_votes'].append(f'{k[0]}:{k[1]} {d["name"][:50]} votes {dict(votes)} -> {d["parent"]}')
# ------------------------------------------------------------------ actors and purposes
actors, purposes = {}, {}
VPK_RESORTS = {l.split('\t')[0] for l in open(os.path.join(KODI, 'vpk_resorts.tsv'), encoding='utf-8') if l.strip()}
def resort_actor(code, name):
aid = f'{BID}.ac.r{code}'
vpk = f'{code}-0000' if f'{code}-0000' in VPK_RESORTS else None
actors.setdefault(aid, {'kind': 'resort', 'code': code, 'name': name, 'vpk': vpk})
pid = f'{BID}.pr.{code}'
purposes.setdefault(pid, {'kind': 'resort', 'code': code, 'name': name, 'holder': aid})
return aid, pid
def prog_purpose(rcode, pcode, name, function, parent_pid):
pid = f'{BID}.pr.{rcode}.{pcode}'
kind = 'programme' if pcode.endswith('.00.00') else 'subprogramme'
d = purposes.setdefault(pid, {'kind': kind, 'code': pcode, 'name': name, 'parent': parent_pid,
'holder': f'{BID}.ac.r{rcode}', 'function': function})
if function and not d.get('function'):
d['function'] = function
return pid
def project_purpose(rcode, pcode, projcode, name, parent_pid):
pid = f'{BID}.pj.{rcode}.{pcode or "x"}.{slug(projcode, 60)}'
purposes.setdefault(pid, {'kind': 'project', 'code': projcode, 'name': name, 'parent': parent_pid,
'holder': f'{BID}.ac.r{rcode}'})
return pid
def municipality(name):
aid = f'{BID}.ac.m-{slug(name)}'
actors.setdefault(aid, {'kind': 'municipality', 'name': name})
return aid
# block (core/eu) per subprogramme, from Valsts kase execution data; fallback by code range
blockmap = {}
for line in open(os.path.join(KODI, 'vk_blocks.tsv'), encoding='utf-8'):
g, m, p, sp, blk, bt, n = line.rstrip('\n').split('\t')
if int(g) in (YEAR, YEAR - 1):
blockmap.setdefault((m, sp), 'eu' if blk.startswith('Ārvalstu') else 'core')
def block_of(rcode, pcode):
b = blockmap.get((rcode, pcode))
if b:
return b
report['block_fallback'].append(f'{rcode} {pcode}')
return 'eu' if pcode[:2] in ('60', '61', '62', '63', '64', '65', '66', '67', '68', '69', '70', '71', '72', '73', '74', '75', '76', '77', '78', '79', '80', '81', '82', '83', '84', '85') else 'core'
# ------------------------------------------------------------------ outputs
allocs, checks = [], []
blockcodes = collections.defaultdict(lambda: collections.defaultdict(set)) # block id -> flow -> printed codes
def register_block(bid, lines):
for ln in lines:
if not ln.get('skip') and not ln.get('section') and ln.get('flow') not in (None, 'balance'):
blockcodes[bid][ln['flow']].add(f"{ln['scheme']}:{ln['code']}")
_alloc_ids = collections.Counter()
def add_alloc(**a):
"""Allocation id = source annex + source row + year (+ 'l' for 'later years' column, + -N if the same cell yields several facts).
Derived from the source, so re-parsing the same law gives the same ids."""
base = f'{BID}.a.{a.pop("srcAnnex")}.r{a.get("srcRow", 0)}.y{a["year"]}{"l" if a.get("untilEnd") else ""}'
_alloc_ids[base] += 1
a['id'] = base if _alloc_ids[base] == 1 else f'{base}-{_alloc_ids[base]}'
allocs.append(a)
return a['id']
def add_check(annex, row, value, **sel):
checks.append({'annex': annex, 'row': row, 'value': value, 'sel': sel})
pending = [] # (bid, flow, ck, alloc kwargs) for lines of detailed blocks; leafness decided on the canonical tree
def queue_alloc(_bid, _flow, _ck, **kw):
pending.append((_bid, _flow, _ck, kw))
def canon_chain(ck, parent):
out, k = [], ck
while k and k not in out:
out.append(k)
k = parent.get(k)
return out
def finalize_lines():
parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
below = {} # (bid, flow, ck) -> printed codes in that block whose canonical chain contains ck
for bid, fl in blockcodes.items():
for flow, codes in fl.items():
for c in codes:
for anc in canon_chain(c, parent):
if anc in codes:
below.setdefault((bid, flow, anc), []).append(c)
for c in checks:
r = c['sel'].get('codes')
if isinstance(r, dict) and 'resolve' in r:
bid, flow, ck = r['resolve']
c['sel']['codes'] = sorted(set(below.get((bid, flow, ck), [ck])) | {ck})
for bid, flow, ck, kw in pending:
if len(set(below.get((bid, flow, ck), [ck])) - {ck}) == 0:
add_alloc(**kw)
# ------------------------------------------------------------------ generic tree-annex reader
def tree_blocks(rows, label_col, amount_cols, header_fn, level_fn):
"""Split rows into blocks keyed by header path; each block has lines with level, flow, amounts."""
blocks, path, cur = [], {}, None
for r in rows:
cells = r['cells']
h = header_fn(r, path)
if h:
path = h
cur = {'path': dict(path), 'lines': []}
blocks.append(cur)
continue
label = str(cells[label_col]) if len(cells) > label_col else ''
amts = {}
for y, ci in amount_cols.items():
v = num(cells[ci]) if ci < len(cells) else None
if v is not None:
amts[y] = v
if not label or not amts:
continue
if cur is None:
cur = {'path': dict(path), 'lines': []}
blocks.append(cur)
cur['lines'].append({'row': r['row'], 'label': label, 'level': level_fn(r, label), 'amounts': amts})
return blocks
def resolve_lines(block):
"""Assign flow, code and parent code to every line; mark leaves (no deeper line follows in the same section)."""
lines, flow, stack = block['lines'], None, []
for i, ln in enumerate(lines):
n = norm(ln['label'])
if n in SECTION:
flow = SECTION[n]
ln.update(flow=flow, section=True, scheme='law', code='S-' + flow, leaf=False)
stack = [(ln['level'], ln)]
continue
if flow is None:
ln.update(flow=None, skip=True)
continue
while stack and stack[-1][0] >= ln['level']:
stack.pop()
parent = stack[-1][1] if stack else None
sch, code = code_for(ln['label'], flow)
ln.update(flow=flow, scheme=sch, code=code, section=False)
if parent is not None and not parent.get('section'):
register_class(sch, code, ln['label'], f'{parent["scheme"]}:{parent["code"]}')
else:
register_class(sch, code, ln['label'], None)
stack.append((ln['level'], ln))
for i, ln in enumerate(lines):
if ln.get('skip'):
continue
desc = [] if ln.get('section') else [f"{ln['scheme']}:{ln['code']}"]
for x in lines[i + 1:]:
if x.get('skip'):
continue
if x.get('section') or x['level'] <= ln['level'] or x.get('flow') != ln.get('flow'):
break
desc.append(f"{x['scheme']}:{x['code']}")
ln['codes'] = desc
for i, ln in enumerate(lines):
if ln.get('section') or ln.get('skip'):
continue
nxt = next((x for x in lines[i + 1:] if not x.get('skip')), None)
ln['leaf'] = nxt is None or nxt.get('section') or nxt['level'] <= ln['level'] or nxt.get('flow') != ln['flow']
return lines
# ------------------------------------------------------------------ annex 4 and 5 (2026 appropriations by subprogramme)
label_levels = collections.defaultdict(collections.Counter) # (flow, normlabel) -> relative level counts, learned for annex 11
def parse_programmes(n, fund):
CUR_ANNEX[0] = f'a{n}'
path_ = annex_file(n)
rows = read_rows(path_, label_col=2)
progs_seen = {}
def header(r, path):
c = r['cells']
lab = str(c[2]) if len(c) > 2 else ''
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
if m and num(c[3] if len(c) > 3 else '') is None:
aid, pid = resort_actor(m.group(1), m.group(2))
return {'resort': m.group(1), 'rpid': pid}
code = str(c[0])
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
rc = path['resort']
parent = path['rpid']
pcode = code
if not code.endswith('.00.00'):
pp = code[:2] + '.00.00'
if (rc, pp) in progs_seen:
parent = progs_seen[(rc, pp)]
func = str(c[1]) if re.fullmatch(r'\d{2}\.\d{3}', str(c[1])) else None
pid = prog_purpose(rc, pcode, lab, func, parent)
progs_seen[(rc, pcode)] = pid
return {'resort': rc, 'rpid': path['rpid'], 'pcode': pcode, 'pid': pid, 'parent': parent, 'func': func}
return None
blocks = tree_blocks(rows, 2, {YEAR: 3}, header, lambda r, lab: r['indent'])
# leaf blocks: subprogrammes, or programmes without subprogrammes
has_child = set(b['path'].get('parent') for b in blocks if b['path'].get('pcode'))
for bi, b in enumerate(blocks):
lines = resolve_lines(b)
bid = f'a{n}:{bi}'
register_block(bid, lines)
p = b['path']
if 'pid' in p:
sel_purpose = p['pid']
elif 'rpid' in p:
sel_purpose = p['rpid']
else:
sel_purpose = None
leafblock = 'pid' in p and p['pid'] not in has_child
blk = block_of(p['resort'], p['pcode']) if 'pcode' in p else None
# learn relative levels for annex 11
base = None
for ln in lines:
if ln.get('section'):
base = ln['level']
elif not ln.get('skip') and base is not None:
label_levels[(ln['flow'], norm(ln['label']))][ln['level'] - base] += 1
for ln in lines:
if ln.get('skip') or ln['flow'] == 'balance':
if ln.get('flow') == 'balance' or ln.get('section') and ln['flow'] == 'balance':
add_check(n, ln['row'], ln['amounts'][YEAR], kind='balance', fund=fund, year=YEAR, purpose=sel_purpose, nature='appropriation', blk=bid)
continue
nature = 'forecast' if ln['flow'] == 'revenue' and fund == 'basic' else 'appropriation'
if fund == 'special' and ln['flow'] == 'revenue':
nature = 'forecast'
add_check(n, ln['row'], ln['amounts'][YEAR], fund=fund, flow=ln['flow'], year=YEAR, purpose=sel_purpose,
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, nature=nature, blk=bid)
store = leafblock and not ln.get('section')
if fund == 'basic' and ln['flow'] == 'revenue':
store = False # basic-budget revenue is stored from annex 2
if store:
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex=f'p{n:02d}', flow=ln['flow'], fund=fund, nature=nature, year=YEAR,
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['pid'], block=blk, function=p.get('func'),
scheme=ln['scheme'], code=ln['code'], amount=ln['amounts'][YEAR],
src=f'{BID}.src.p{n:02d}', srcRow=ln['row'])
# ------------------------------------------------------------------ annex 2 (revenue forecasts, 3 years)
def parse_revenue():
CUR_ANNEX[0] = 'a2'
raw = read_rows(annex_file(2), label_col=0)
# the label column differs between years (2025 .xls has an extra empty first column): find it from the header row
lc = next((i for r in raw for i, c in enumerate(r['cells']) if str(c).startswith('Ieņēmumu avots')), 0)
rows = read_rows(annex_file(2), label_col=lc) if lc else raw
part, fund, mode = None, 'basic', 'tree'
lines, fees, info = [], [], []
cur_resort = None
for r in rows:
c = r['cells']
lab = str(c[lc]) if lc < len(c) else ''
if re.match(r'^I\.\s', lab):
fund, mode = 'basic', 'tree'; continue
if re.match(r'^II\.\s', lab):
fund, mode = 'special', 'tree'; continue
if lab.startswith('Ministrija (cita'):
mode = 'fees'; continue
if lab.startswith('Informatīvi'):
mode = 'info'; continue
amts = {y: num(c[lc + 1 + i]) for i, y in enumerate(Y3) if lc + 1 + i < len(c) and num(c[lc + 1 + i]) is not None}
if not amts or not lab:
continue
if mode == 'tree':
lines.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'amounts': amts, 'fund': fund})
elif mode == 'fees':
fees.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'bold': r['bold'], 'amounts': amts})
else:
info.append({'row': r['row'], 'label': lab, 'amounts': amts})
stored = {}
for fund in ('basic', 'special'):
blk = {'lines': [dict(l) for l in lines if l['fund'] == fund]}
resolve_lines(blk)
register_block(f'a2:{fund}', blk['lines'])
for ln in blk['lines']:
if ln.get('skip'):
continue
for y, v in ln['amounts'].items():
add_check(2, ln['row'], v, fund=fund, flow='revenue', year=y, nature='forecast', label=ln['label'],
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=ln.get('codes'), blk=f'a2:{fund}')
if ln.get('leaf') and not ln.get('section'):
if fund == 'special' and y == YEAR:
continue # 2026 special revenue is stored from annex 5 (by subprogramme)
aid = add_alloc(srcAnnex='p02', flow='revenue', fund=fund, nature='forecast', year=y,
holder=None, purpose=None, scheme=ln['scheme'], code=ln['code'], amount=v,
src=f'{BID}.src.p02', srcRow=ln['row'])
stored[(fund, ln['code'], y)] = aid
# fees by administering resort: a breakdown of stored revenue leaves -> partOf
leaf_codes = sorted({k[1] for k in stored if k[0] == 'basic'}, key=len, reverse=True)
for f in fees:
n_ = norm(f['label'])
if n_.startswith('ieņēmumi valsts pamatbudžetā kopā'):
for y, v in f['amounts'].items():
add_check(2, f['row'], v, kind='fees_total', year=y)
continue
if f['bold']:
name = f['label']
m = [a for a, d in actors.items() if d['kind'] == 'resort' and norm(d['name']) == n_]
cur_resort = m[0] if m else None
if not cur_resort:
report['fee_resort_unmatched'].append(f['label'])
for y, v in f['amounts'].items():
add_check(2, f['row'], v, kind='fees_resort', holder=cur_resort, year=y)
continue
sch, code = code_for(f['label'], 'revenue')
register_class(sch, code, f['label'], None)
parent = next((stored[('basic', lc, y0)] for lc in leaf_codes for y0 in [YEAR] if code.startswith(lc.rstrip('0') or lc) and ('basic', lc, YEAR) in stored), None)
if parent is None:
report['fee_parent_unmatched'].append(f'{code} {f["label"][:60]}')
for y, v in f['amounts'].items():
par = None
if parent:
pc = next(k for k, a in stored.items() if a == parent)[1]
par = stored.get(('basic', pc, y))
add_alloc(srcAnnex='p02', flow='revenue', fund='basic', nature='forecast', year=y, holder=cur_resort,
purpose=None, scheme=sch, code=code, amount=v, partOf=par, src=f'{BID}.src.p02', srcRow=f['row'])
for i in info:
for y, v in i['amounts'].items():
params.append({'id': f'{BID}.par.p02-{i["row"]}-{y}', 'name': i['label'], 'year': y, 'value': v, 'unit': 'EUR',
'src': f'{BID}.src.p02'})
# ------------------------------------------------------------------ annex 3 (resort summaries, 3 years; 2027-28 stored as ceilings)
def parse_resort_summary():
CUR_ANNEX[0] = 'a3'
rows = read_rows(annex_file(3), label_col=0)
fund_names = {'valsts pamatbudžets': 'basic', 'valsts speciālais budžets': 'special'}
def header(r, path):
lab = str(r['cells'][0])
n_ = norm(lab)
if num(r['cells'][1] if len(r['cells']) > 1 else '') is not None:
return None
if n_ in fund_names:
p = {k: v for k, v in path.items() if k in ('resort', 'rpid')}
p['fund'] = fund_names[n_]
return p
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
if m:
aid, pid = resort_actor(m.group(1), m.group(2))
return {'resort': m.group(1), 'rpid': pid, 'fund': 'basic'}
if re.match(r'^I\.\s+Valsts pamatfunkciju', lab):
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'core'; return p
if re.match(r'^II\.\s+ES politiku', lab):
p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'eu'; return p
return None
blocks = tree_blocks(rows, 0, {Y3[0]: 1, Y3[1]: 2, Y3[2]: 3}, header, lambda r, lab: r['indent'])
keys = [(b['path'].get('resort'), b['path'].get('fund')) for b in blocks]
for bi, b in enumerate(blocks):
lines = resolve_lines(b)
bid = f'a3:{bi}'
register_block(bid, lines)
p = b['path']
fund = p.get('fund', 'basic')
# a resort's special-budget part may appear without its own fund header: detect by section content later
leafblock = 'resort' in p and 'block' in p
sel = dict(fund=fund, purpose=p.get('rpid'), block=p.get('block'), blk=bid, holderView=True)
for ln in lines:
if ln.get('skip'):
continue
for y, v in ln['amounts'].items():
if ln['flow'] == 'balance':
add_check(3, ln['row'], v, kind='balance', year=y, nature='appropriation' if y == YEAR else 'ceiling', **sel)
continue
nature = 'forecast' if ln['flow'] == 'revenue' else ('appropriation' if y == YEAR else 'ceiling')
add_check(3, ln['row'], v, flow=ln['flow'], year=y, nature=nature,
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
if leafblock and not ln.get('section') and y != YEAR and ln['flow'] != 'revenue':
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p03', flow=ln['flow'], fund=fund, nature='ceiling', year=y,
holder=f'{BID}.ac.r{p["resort"]}', purpose=p['rpid'], block=p['block'],
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p03', srcRow=ln['row'])
# ------------------------------------------------------------------ annex 11 (long-term commitments)
def parse_commitments():
CUR_ANNEX[0] = 'a11'
rows = read_rows(annex_file(11), label_col=1)
hdr = next(r for r in rows if 'Projekta kods' in [str(x) for x in r['cells']])
years = {}
for i, c in enumerate(hdr['cells']):
m = re.match(r'^(\d{4})\.', str(c))
if m:
years[int(m.group(1))] = i
elif str(c).startswith('Tālākā'):
years['later'] = i
last = max(y for y in years if y != 'later')
kinds = {}
def level_fn(r, lab):
cnt = None
for fl in ('expenditure', 'resource', 'financing', 'revenue'):
c = label_levels.get((fl, norm(lab)))
if c:
cnt = c.most_common(1)[0][0] + 1
break
if cnt is None and norm(lab) not in SECTION:
report['a11_level_unknown'].append(lab)
return 99
return cnt or 0
progs_seen = {}
state = {'prev_header': None}
def header(r, path):
h = _header(r, path)
state['prev_header'] = h is not None and h.get('_kindable', False)
if h is not None:
h.pop('_kindable', None)
return h
def _header(r, path):
c = r['cells']
lab = str(c[1]) if len(c) > 1 else ''
if any(num(c[i]) is not None for i in years.values() if i < len(c)):
return None
if lab.upper().startswith('VALSTS PAMATBUDŽETS'):
return {'fund': 'basic'}
if lab.upper().startswith('VALSTS SPECIĀLAIS BUDŽETS'):
return {'fund': 'special'}
m = re.match(r'^(\d{10})\s+(.+)$', lab)
if m:
kinds[m.group(1)] = m.group(2)
register_class('commitment', m.group(1), m.group(2), None)
if state['prev_header']: # directly under a resort or programme header
p = dict(path); p['kind'] = m.group(1)
if 'pcode' not in p:
p['rkind'] = m.group(1)
return p
return {'fund': path.get('fund', 'basic'), 'kind': m.group(1), 'topkind': m.group(1)} # new top-level commitment-kind section
m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab)
if m:
aid, pid = resort_actor(m.group(1), m.group(2))
return {'fund': path.get('fund', 'basic'), 'resort': m.group(1), 'rpid': pid, '_kindable': True,
'topkind': path.get('topkind'), 'kind': path.get('topkind')}
code = str(c[0])
if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path:
rc = path['resort']
pid = f'{BID}.pr.{rc}.{code}'
if pid not in purposes:
parent = purposes.get(f'{BID}.pr.{rc}.{code[:2]}.00.00') and f'{BID}.pr.{rc}.{code[:2]}.00.00' or path['rpid']
prog_purpose(rc, code, lab, None, parent)
report['a11_new_programme'].append(f'{rc} {code} {lab[:50]}')
p = {k: v for k, v in path.items() if k in ('fund', 'resort', 'rpid', 'topkind', 'rkind')}
p.update(pcode=code, pid=pid, _kindable=True, kind=path.get('rkind') or path.get('topkind'))
return p
proj = str(c[2]) if len(c) > 2 else ''
if proj and 'resort' in path:
parent = path.get('pid') or path['rpid']
pj = project_purpose(path['resort'], path.get('pcode'), proj, lab, parent)
p = dict(path); p['project'] = proj; p['pjid'] = pj
return p
return None
amount_cols = {y: i for y, i in years.items()}
blocks = tree_blocks(rows, 1, amount_cols, header, level_fn)
def more_specific(k2, k): # commitment kind k2 is a sub-type of k (or equal); codes nest in 2-digit groups
from calc import kind_stem
if not k:
return True
return bool(k2 and k2.startswith(kind_stem(k)))
def refined(bi):
"""A detailed block is refined (not a leaf) if, within the same resort+programme, a later or earlier block adds
a project under a compatible kind, or shows the same project under a more specific kind."""
p = blocks[bi]['path']
for j, b2 in enumerate(blocks):
if j == bi:
continue
q = b2['path']
if q.get('resort') != p.get('resort') or q.get('fund') != p.get('fund') or not q.get('pcode'):
continue
if q['pcode'] != p.get('pcode'):
# a programme block is refined by its subprogramme blocks (same resort, same programme number)
if (p.get('pcode') or '').endswith('.00.00') and q['pcode'][:2] == p['pcode'][:2] and not q['pcode'].endswith('.00.00') \
and not p.get('project') and more_specific(q.get('kind'), p.get('kind')):
return True
continue
if not p.get('project') and q.get('project') and more_specific(q.get('kind'), p.get('kind')):
return True
if p.get('project') and q.get('project') == p['project'] and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
return True
if not p.get('project') and not q.get('project') and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')):
return True
return False
# leaf blocks: those whose path is not extended by a later block
def key(p):
return (p.get('fund'), p.get('kind'), p.get('resort'), p.get('pcode'), p.get('project'))
keys = [key(b['path']) for b in blocks]
def extends(a, b): # b is strictly more specific than a
return a != b and all(x is None or x == y or (i == 1 and x and y and y.startswith(x.rstrip('0'))) for i, (x, y) in enumerate(zip(a, b)))
for bi, b in enumerate(blocks):
lines = resolve_lines(b)
bid = f'a11:{bi}'
register_block(bid, lines)
p = b['path']
leafblock = bool(p.get('pcode')) and not refined(bi)
sel = dict(fund=p.get('fund', 'basic'), ckind=p.get('kind'), purpose=p.get('pjid') or p.get('pid') or p.get('rpid'), blk=bid)
for ln in lines:
if ln.get('skip'):
continue
for y, v in ln['amounts'].items():
yy, until = (last + 1, True) if y == 'later' else (y, False)
if ln['flow'] == 'balance':
add_check(11, ln['row'], v, kind='balance', year=yy, untilEnd=until, nature='commitment', **sel)
continue
add_check(11, ln['row'], v, flow=ln['flow'], year=yy, untilEnd=until, nature='commitment',
code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel)
if leafblock and not ln.get('section') and v != 0:
queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p11', flow=ln['flow'], fund=sel['fund'], nature='commitment', year=yy, untilEnd=until,
holder=f'{BID}.ac.r{p["resort"]}' if p.get('resort') else None,
purpose=sel['purpose'], commitmentKind=p.get('kind'),
scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p11', srcRow=ln['row'])
# ------------------------------------------------------------------ annexes 6-10 (earmarked grants to municipalities)
def parse_grants(n):
path_ = annex_file(n)
t = docx_tables(path_)[-1]
title = next((r[0] for r in t if r and len(r[0]) > 40), f'{n}. pielikums')
gid = f'{BID}.pr.g{n:02d}'
purposes[gid] = {'kind': 'grant', 'code': f'P{n:02d}', 'name': title}
period = (f'{YEAR}-01-01', f'{YEAR}-12-31')
cols = None
for ri, r in enumerate(t):
first = r[0] if r else ''
m = re.match(r'^(I{1,2})\.\s', first)
if m:
period = (f'{YEAR}-01-01', f'{YEAR}-08-31') if m.group(1) == 'I' else (f'{YEAR}-09-01', f'{YEAR}-12-31')
continue
if first.startswith('Pašvaldības'):
cols = r; continue
vals = [num(x) for x in r[1:]]
if not first or not any(v is not None for v in vals):
continue
n_ = norm(first)
if n_ in ('kopā', 'pavisam kopā'):
heads_ = [norm(h) for h in (cols or [])[1:]]
for ci, v in enumerate(vals):
if v is not None:
h = heads_[ci] if ci < len(heads_) else ''
add_check(n, ri + 1, v, kind='grant_total', purpose=gid, code=f'law:G{n:02d}-{slug(h or "summa", 30) or "summa"}',
periodFrom=None if n_ == 'pavisam kopā' else period[0], periodTo=None if n_ == 'pavisam kopā' else period[1], year=YEAR)
continue
recip = f'{BID}.ac.unallocated' if n_.startswith('nesadalītie') else municipality(first)
if recip.endswith('unallocated'):
actors[recip] = {'kind': 'unallocated', 'name': 'Nesadalītie līdzekļi'}
heads = [norm(h) for h in (cols or [])[1:]]
# column roles: 'pavisam kopā' = total; 'tai skaitā …' = part of previous; otherwise the main amount
ids = {}
total_idx = next((i for i, h in enumerate(heads) if h.startswith('pavisam kopā')), None)
main_idx = 0
order = ([total_idx] if total_idx is not None else []) + [i for i in range(len(vals)) if i != total_idx]
for ci in order:
v = vals[ci] if ci < len(vals) else None
if v is None:
continue
h = heads[ci] if ci < len(heads) else ''
part = None
if h.startswith('tai skaitā'):
part = ids.get(ci - 1)
elif total_idx is not None and ci != total_idx:
part = ids.get(total_idx)
aid = add_alloc(srcAnnex=f'p{n:02d}', flow='expenditure', fund='basic', nature='earmarkedGrant', year=YEAR,
periodFrom=period[0], periodTo=period[1], recipient=recip, purpose=gid,
scheme='law', code=f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', amount=v, partOf=part,
src=f'{BID}.src.p{n:02d}', srcRow=ri + 1)
register_class('law', f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', (cols[ci + 1] if cols and ci + 1 < len(cols) else 'Summa') or 'Summa', None)
ids[ci] = aid
# ------------------------------------------------------------------ annex 1 (consolidated; checks and GDP parameter)
params = []
def parse_consolidated():
rows = read_rows(annex_file(1), label_col=0)
hdr = next(r for r in rows if any(re.match(r'^\d{4}\.', str(c)) for c in r['cells']))
ycols = {int(re.match(r'^(\d{4})', str(c)).group(1)): i for i, c in enumerate(hdr['cells']) if re.match(r'^\d{4}\.', str(c))}
lc = next(i for i, c in enumerate(hdr['cells']) if str(c) == 'Nosaukums')
pct = False
for r in rows:
c = r['cells']
lab = str(c[lc]) if lc < len(c) else ''
if lab.startswith('Procentos no IKP'):
pct = True; continue
for y, i in ycols.items():
v = num(c[i]) if i < len(c) else None
if v is None or not lab:
continue
if lab.startswith('IKP milj'):
params.append({'id': f'{BID}.par.ikp-{y}', 'name': 'IKP prognoze', 'code': 'IKP', 'year': y, 'value': v,
'unit': 'milj. EUR', 'src': f'{BID}.src.p01'})
continue
add_check(1, r['row'], v, kind='a1', label=lab, indent=r['indent'], pct=pct, year=y)
# ------------------------------------------------------------------ law text
provisions = []
def parse_law():
s = open(LAW_HTML, encoding='utf-8', errors='replace').read()
chapter = None
for m in re.finditer(r"<div class='(TV212|TV213)'([^>]*)>(.*?)(?=<div class='TV21[23]'|<div class='TV9|$)", s, re.S):
kind, tag, body = m.groups()
mm = re.search(r'data-num="(\d+)"', tag)
nr = mm.group(1) if mm else None
text = body.replace('</p>', '\n').replace('<br />', ' ')
text = html.unescape(re.sub(r'<[^>]+>', '', text))
text = re.sub(r'[ \t]+', ' ', re.sub(r'\n\s*\n+', '\n', text)).strip()
if kind == 'TV212':
chapter = re.sub(r'\s+', ' ', text)
continue
if not nr:
continue
text = re.sub(r'\n?\d+\s*$', '', text).strip()
provisions.append({'id': f'{BID}.p{int(nr):03d}', 'chapter': chapter, 'article': int(nr), 'text': text,
'src': f'{BID}.src.law'})
# ------------------------------------------------------------------ run
parse_law()
parse_programmes(4, 'basic')
parse_programmes(5, 'special')
parse_revenue()
parse_resort_summary()
parse_commitments()
for g in (6, 7, 8, 9, 10):
parse_grants(g)
parse_consolidated()
finalize_classes()
finalize_lines()
# special budget revenue 2026 belongs to the resort that holds the special budget; annex 2 2027+ special revenue too
special_holders = sorted({a['holder'] for a in allocs if a['fund'] == 'special' and a.get('holder')})
for a in allocs:
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('holder') and len(special_holders) == 1:
a['holder'] = special_holders[0]
a['purpose'] = a.get('purpose') or f"{BID}.pr.{special_holders[0].rsplit('.r', 1)[1]}"
if a['fund'] == 'special' and a['flow'] == 'revenue' and not a.get('block'):
a['block'] = 'core' # the special budget (social insurance) is entirely basic functions
# ------------------------------------------------------------------ residuals: amounts the law shows only at an aggregate level
purpose_parent = {k: v.get('parent') for k, v in purposes.items()}
cls_parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']}
intra_codes = {f'{k[0]}:{k[1]}' for k, v in classitems.items() if is_intra(v['name'])}
purpose_holder = {k: v.get('holder') for k, v in purposes.items()}
calc = Calc(allocs, purpose_parent, cls_parent, intra_codes, blockcodes, purpose_holder)
def depth(p):
d = 0
while p:
d += 1
p = purpose_parent.get(p)
return d
cands = []
for c in checks:
s_ = c['sel']
if c['annex'] not in (3, 4, 5, 11) or s_.get('kind') or not s_.get('code') or not s_.get('codes') or len(s_['codes']) != 1 or s_.get('flow') == 'revenue':
continue
if c['annex'] == 3 and s_['year'] == YEAR:
continue
cands.append(c)
cands.sort(key=lambda c: (-depth(c['sel'].get('purpose')), c['sel'].get('block') is None, c['annex'] in (3,), c['annex']))
nres = collections.Counter()
for c in cands:
s_ = c['sel']
got = calc.total(s_.get('fund'), s_['flow'], s_['year'], s_['nature'], purpose=s_.get('purpose'), block=s_.get('block'),
codes=s_['codes'], ckind=s_.get('ckind'), untilEnd=s_.get('untilEnd', False), blk=s_.get('blk'))
diff = c['value'] - got
if abs(diff) < 0.5:
continue
if not (s_['flow'] == 'financing' or abs(got) < 0.5):
report['unexplained'].append(f"annex {c['annex']} row {c['row']} {s_['code']} {s_.get('purpose')} {s_['year']}: printed {c['value']} children {got}")
continue
sch, code = s_['code'].split(':', 1)
if is_intra(classitems.get((sch, code), {}).get('name', '')) and s_.get('purpose') is None:
continue
pur = s_.get('purpose')
holder = (purposes.get(pur) or {}).get('holder') if pur else None
a = dict(srcAnnex=f"p{c['annex']:02d}", flow=s_['flow'], fund=s_.get('fund') or 'basic', nature=s_['nature'], year=s_['year'],
untilEnd=s_.get('untilEnd', False), holder=holder, purpose=pur, block=s_.get('block'), commitmentKind=s_.get('ckind'),
scheme=sch, code=code, amount=diff, src=f"{BID}.src.p{c['annex']:02d}", srcRow=c['row'])
add_alloc(**a)
calc.add(allocs[-1])
nres[c['annex']] += 1
splits = [c for c in checks if c['annex'] == 3 and not c['sel'].get('kind') and c['sel']['year'] == YEAR
and c['sel'].get('flow') == 'financing' and c['sel'].get('block') and c['sel'].get('code')
and len(c['sel'].get('codes') or []) == 1]
splits.sort(key=lambda c: -depth(c['sel'].get('purpose')))
for c in splits:
s_ = c['sel']
got = calc.total(s_.get('fund'), 'financing', YEAR, s_['nature'], purpose=s_.get('purpose'), block=s_['block'], codes=s_['codes'],
blk=s_.get('blk'))
diff = c['value'] - got
if abs(diff) < 0.5:
continue
sch, code = s_['code'].split(':', 1)
pur = s_.get('purpose')
holder = (purposes.get(pur) or {}).get('holder') if pur else None
# annex 3 attributes an amount that annex 4 shows without a block to a block: reclassify within the same purpose (totals unchanged)
for pur_, hol_, blk_, amt in ((pur, holder, s_['block'], diff), (pur, holder, None, -diff)):
add_alloc(srcAnnex='p03', flow='financing', fund=s_.get('fund') or 'basic', nature=s_['nature'], year=YEAR, holder=hol_,
purpose=pur_, block=blk_, scheme=sch, code=code, amount=amt, src=f'{BID}.src.p03', srcRow=c['row'])
calc.add(allocs[-1])
nres['3-block-split'] += 1
report['residual_allocations'] = [f'annex {k}: {v}' for k, v in sorted(nres.items(), key=str)]
# ------------------------------------------------------------------ sources
law_title = re.search(r"<div class='TV207'[^>]*>(.*?)</div>", open(LAW_HTML, encoding='utf-8').read(), re.S)
law_title = html.unescape(re.sub(r'<[^>]+>', '', law_title.group(1))).strip() if law_title else f'Par valsts budžetu {YEAR}. gadam'
links = json.load(open(os.path.join(ADIR, 'links.json'), encoding='utf-8'))
sources = [{'id': f'{BID}.src.law', 'kind': 'lawText', 'title': law_title, 'url': links['law'], 'sha256': sha(LAW_HTML)}]
for n in range(1, 13):
f = annex_file(n)
sources.append({'id': f'{BID}.src.p{n:02d}', 'kind': 'form' if n == 12 else 'annex', 'annex': n,
'title': f'{n}. pielikums', 'url': links['annex'][str(n)], 'sha256': sha(f)})
# ------------------------------------------------------------------ write XML
def attrs(d, order):
out = []
for k in order:
v = d.get(k)
if v is None or v == '' or v is False:
continue
if isinstance(v, bool):
v = 'true'
if isinstance(v, float):
v = (f'{v:.2f}'.rstrip('0').rstrip('.')) if v != int(v) else str(int(v))
out.append(f'{k}={quoteattr(str(v))}')
return ' '.join(out)
os.makedirs(OUT, exist_ok=True)
xp = os.path.join(OUT, f'{BID}.xml')
with open(xp, 'w', encoding='utf-8') as fo:
fo.write('<?xml version="1.0" encoding="UTF-8"?>\n')
fo.write(f'<Budget xmlns="urn:pppa:vpk:budzets:0.1" {attrs({"id": BID, "title": law_title, "year": YEAR, "horizonTo": YEAR + 2, "status": "adopted", "act": links["law"], "published": links.get("published"), "version": links.get("version")}, ["id", "title", "year", "horizonTo", "status", "act", "published", "version"])}>\n')
for s_ in sources:
fo.write(f' <Source {attrs(s_, ["id", "kind", "annex", "title", "url", "sha256"])}/>\n')
for p in provisions:
fo.write(f' <Provision {attrs(p, ["id", "chapter", "article", "src"])}>{escape(p["text"])}</Provision>\n')
for p in params:
fo.write(f' <Parameter {attrs(p, ["id", "name", "code", "year", "value", "unit", "src"])}/>\n')
for aid, a in sorted(actors.items()):
fo.write(f' <Actor {attrs(dict(a, id=aid), ["id", "kind", "code", "vpk", "name"])}/>\n')
for pid, p in purposes.items():
fo.write(f' <Purpose {attrs(dict(p, id=pid), ["id", "kind", "code", "name", "parent", "holder", "block", "function"])}/>\n')
for (sch, code), c in sorted(classitems.items()):
ps, par = (c['parent'].split(':', 1) if c['parent'] else (None, None))
fo.write(f' <ClassItem {attrs({"scheme": sch, "code": code, "name": c["name"], "parent": par, "parentScheme": ps if ps != sch else None, "intraFund": is_intra(c["name"]) or None}, ["scheme", "code", "name", "parent", "parentScheme", "intraFund"])}/>\n')
for a in allocs:
a = {k: v for k, v in a.items() if not k.startswith('_')}
fo.write(f' <Allocation {attrs(a, ["id", "flow", "fund", "nature", "year", "untilEnd", "periodFrom", "periodTo", "holder", "recipient", "purpose", "block", "commitmentKind", "function", "scheme", "code", "amount", "partOf", "src", "srcRow"])}/>\n')
fo.write('</Budget>\n')
json.dump({'checks': checks, 'blockcodes': {k: {f: sorted(v) for f, v in d.items()} for k, d in blockcodes.items()}},
open(os.path.join(OUT, f'{BID}.checks.json'), 'w', encoding='utf-8'), ensure_ascii=False)
summary = {'provisions': len(provisions), 'parameters': len(params), 'actors': len(actors), 'purposes': len(purposes),
'classitems': len(classitems), 'allocations': len(allocs), 'checks': len(checks),
'allocations_by_annex': collections.Counter(a['src'].rsplit('.', 1)[1] for a in allocs)}
print(json.dumps(summary, ensure_ascii=False, default=str))
for k, v in report.items():
print(f'REPORT {k}: {len(v)}')
for x in v[:8]:
print(' ', x)
json.dump(report, open(os.path.join(OUT, f'{BID}.report.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)

55
tools/rd.py Normal file
View File

@@ -0,0 +1,55 @@
"""Shared readers for budget-law annex files (xlsx via openpyxl, xls via xlrd, docx via stdlib)."""
import zipfile, re, xml.etree.ElementTree as ET
def sheets(path):
"""Yield (sheet_name, rows) with rows as lists of stripped strings/numbers; hidden BEx sheets included."""
if path.lower().endswith('.xlsx'):
import openpyxl
wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
for ws in wb.worksheets:
rows = []
for r in ws.iter_rows(values_only=True):
rows.append([("" if v is None else (v.strip() if isinstance(v, str) else v)) for v in r])
yield ws.title, rows
elif path.lower().endswith('.xls'):
import xlrd
wb = xlrd.open_workbook(path)
for ws in wb.sheets():
rows = []
for i in range(ws.nrows):
rows.append([(c.strip() if isinstance(c, str) else c) for c in ws.row_values(i)])
yield ws.name, rows
def main_sheet(path):
"""The printed annex sheet: the one whose name ends with 'piel' or is the first non-BEx sheet with most rows."""
best = None
for name, rows in sheets(path):
if name.lower().endswith('piel') or re.match(r'^\d+\.?\s*piel', name.lower()):
return name, rows
if name.startswith(('BEx', 'HEADER', 'FOOTER', 'ZQZ', 'var', 'list', 'parms')):
continue
if best is None or len(rows) > len(best[1]):
best = (name, rows)
return best
W = '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}'
def docx_tables(path):
"""List of tables, each a list of rows of cell texts."""
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
out = []
for t in x.iter(W + 'tbl'):
rows = []
for tr in t.findall(W + 'tr'):
rows.append([''.join(n.text or '' for n in tc.iter(W + 't')).strip() for tc in tr.findall(W + 'tc')])
out.append(rows)
return out
def docx_paras(path):
x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml'))
return [''.join(n.text or '' for n in p.iter(W + 't')).strip() for p in x.iter(W + 'p')]
def num(v):
"""Parse an amount cell: numbers, '1 234 567', '-1 234', '–'. Returns float or None."""
if isinstance(v, (int, float)): return float(v)
s = str(v).replace(' ', ' ').replace(' ', '').replace('–', '-').replace('−', '-').replace(',', '.')
return float(s) if re.fullmatch(r'-?\d+(\.\d+)?', s) else None

33
tools/rows.py Normal file
View File

@@ -0,0 +1,33 @@
"""Uniform row reader with indent level of the label column, for xlsx (openpyxl) and xls (xlrd)."""
import re
def _label_col(cells):
return next((i for i, v in enumerate(cells) if isinstance(v, str) and v and not re.fullmatch(r'[\d.]+', v)), 0)
def read_rows(path, sheet_hint='piel', label_col=None, max_col=12):
"""Return list of dicts: row (1-based), cells, indent (of label cell), bold. Stops after 200 empty rows."""
out, empty = [], 0
if path.lower().endswith('.xlsx'):
import openpyxl
wb = openpyxl.load_workbook(path, read_only=False, data_only=True)
ws = next(w for w in wb.worksheets if sheet_hint in w.title.lower().replace('.', ''))
for row in ws.iter_rows(min_row=1, max_col=max_col):
cells = [("" if c.value is None else (c.value.strip() if isinstance(c.value, str) else c.value)) for c in row]
if not any(str(v) for v in cells):
empty += 1
if empty > 200: break
continue
empty = 0
lc = label_col if label_col is not None else _label_col(cells)
c = row[lc] if lc < len(row) else row[0]
out.append({'row': row[0].row, 'cells': cells, 'indent': int(c.alignment.indent or 0), 'bold': bool(c.font and c.font.b)})
else:
import xlrd
bk = xlrd.open_workbook(path, formatting_info=True)
sh = next(bk.sheet_by_name(n) for n in bk.sheet_names() if sheet_hint in n.lower().replace('.', ''))
for r in range(sh.nrows):
cells = [(c.strip() if isinstance(c, str) else c) for c in sh.row_values(r)][:max_col]
if not any(str(v) for v in cells): continue
lc = label_col if label_col is not None else _label_col(cells)
xf = bk.xf_list[sh.cell_xf_index(r, lc)] if lc < sh.ncols else None
out.append({'row': r + 1, 'cells': cells, 'indent': xf.alignment.indent_level if xf else 0,
'bold': bool(xf and bk.font_list[xf.font_index].bold)})
return out

174
tools/verify_budget.py Normal file
View File

@@ -0,0 +1,174 @@
"""Round-trip proof: recompute every printed number of the budget law annexes from the XML alone.
Usage: python verify_budget.py OUT_DIR YEAR
Reads lv-vb-YEAR.xml (validated against the XSD) and lv-vb-YEAR.checks.json (each printed number with its meaning).
A check passes when the value computed from the XML equals the printed value (difference < 0.5 EUR; < 0.005 for % of GDP).
"""
import sys, os, re, json, collections
TOOLS = os.path.dirname(os.path.abspath(__file__))
ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS))
sys.path.insert(0, TOOLS)
from lxml import etree
from calc import Calc
OUT, YEAR = sys.argv[1], int(sys.argv[2])
BID = f'lv-vb-{YEAR}'
NS = '{urn:pppa:vpk:budzets:0.1}'
schema = etree.XMLSchema(etree.parse(os.path.join(ROOT, 'schemas', 'valsts-budzets-0.1.xsd')))
doc = etree.parse(os.path.join(OUT, f'{BID}.xml'))
valid = schema.validate(doc)
print('XSD valid:', valid)
for e in list(schema.error_log)[:10]:
print(' ', e.line, e.message[:200])
def norm(s):
s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-')
return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip()
root = doc.getroot()
purpose_parent = {p.get('id'): p.get('parent') for p in root.iter(NS + 'Purpose')}
cls_parent, cls_by_name, intra = {}, collections.defaultdict(set), set()
for c in root.iter(NS + 'ClassItem'):
k = f"{c.get('scheme')}:{c.get('code')}"
if c.get('parent'):
cls_parent[k] = f"{c.get('parentScheme') or c.get('scheme')}:{c.get('parent')}"
if c.get('intraFund') == 'true':
intra.add(k)
cls_by_name[norm(c.get('name'))].add(k)
gdp = {int(p.get('year')): float(p.get('value')) for p in root.iter(NS + 'Parameter') if p.get('code') == 'IKP'}
A = []
for a in root.iter(NS + 'Allocation'):
d = dict(a.attrib)
d['year'] = int(d['year'])
d['untilEnd'] = d.get('untilEnd') == 'true'
A.append(d)
data = json.load(open(os.path.join(OUT, f'{BID}.checks.json'), encoding='utf-8'))
purpose_holder = {p.get('id'): p.get('holder') for p in root.iter(NS + 'Purpose')}
calc = Calc(A, purpose_parent, cls_parent, intra, data['blockcodes'], purpose_holder)
print('allocations:', len(A))
res = collections.defaultdict(lambda: {'n': 0, 'ok': 0, 'bad': [], 'skip': 0})
def judge(c, got, tol=0.5):
r = res[c['annex']]
r['n'] += 1
if got is None:
r['skip'] += 1
r['bad'].append(('not computed', c['row'], c['sel'].get('label', ''), c['value'], None))
elif abs(got - c['value']) < tol:
r['ok'] += 1
else:
r['bad'].append(('mismatch', c['row'], json.dumps({k: v for k, v in c['sel'].items() if k != 'codes'}, ensure_ascii=False)[:230], c['value'], round(got, 2)))
# ------------------------------------------------------------------ annex 1: consolidated budget by formula
def a1_value(label, year, ctx, pct):
n = norm(label)
base = 'appropriation' if year == YEAR else 'ceiling'
T = calc.total
def rev(f): return T(f, 'revenue', year, 'forecast')
def exp(f, pre=None): return T(f, 'expenditure', year, base, ekk_prefix=pre)
def fin(f, codes=None): return T(f, 'financing', year, base, codes=codes)
def cap(f): return exp(f, '5') + exp(f, '9')
b2s_m, b2s_c, s2b_m, s2b_c = exp('basic', '712'), exp('basic', '912'), exp('special', '711'), exp('special', '911')
PA, SA = rev('basic') - (s2b_m + s2b_c), rev('special') - (b2s_m + b2s_c)
PB, SB = exp('basic') - (b2s_m + b2s_c), exp('special') - (s2b_m + s2b_c)
PB2, SB2 = cap('basic') - b2s_c, cap('special') - s2b_c
PB1, SB1 = PB - PB2, SB - SB2
if pct:
g = gdp.get(year)
if not g:
return None
val = {'valsts budžeta ieņēmumi': PA + SA, 'valsts budžeta izdevumi': PB + SB, 'valsts budžeta finansiālā bilance': PA + SA - PB - SB}
for k, v in val.items():
if n.startswith(k):
return round(v / (g * 1e6) * 100, 2)
return None
first = n.split(' ')[0] if n else ''
exact = {'ka': PA + SA, 'pa': PA, 'sa': SA, 'kb': PB + SB, 'kb1': PB1 + SB1, 'kb2': PB2 + SB2,
'pb': PB, 'pb1': PB1, 'pb2': PB2, 'sb': SB, 'sb1': SB1, 'sb2': SB2}
if first in exact:
return exact[first]
prefix = [
('valsts pamatbudžeta ieņēmumi', rev('basic')), ('valsts speciālā budžeta ieņēmumi', rev('special')),
('valsts pamatbudžeta izdevumi', exp('basic')), ('valsts speciālā budžeta izdevumi', exp('special')),
('valsts pamatbudžeta uzturēšanas izdevumi', exp('basic') - cap('basic')), ('valsts pamatbudžeta kapitālie izdevumi', cap('basic')),
('valsts speciālā budžeta uzturēšanas izdevumi', exp('special') - cap('special')), ('valsts speciālā budžeta kapitālie izdevumi', cap('special')),
('valsts budžeta finansiālā bilance', PA + SA - PB - SB),
('valsts pamatbudžeta finansiālā bilance', rev('basic') - exp('basic')),
('valsts speciālā budžeta finansiālā bilance', rev('special') - exp('special')),
]
for k, v in prefix:
if n.startswith(k):
return v
if n.startswith('mīnus transferts no valsts speciālā'): return s2b_m + s2b_c
if n.startswith('mīnus transferts no valsts pamatbudžeta'): return b2s_m + b2s_c
if n.startswith('mīnus transferts valsts speciāl'):
return {'gross': b2s_m + b2s_c, 'maint': b2s_m, 'cap': b2s_c}[ctx['exp_part']]
if n.startswith('mīnus transferts valsts pamatbudžet'):
return {'gross': s2b_m + s2b_c, 'maint': s2b_m, 'cap': s2b_c}[ctx['exp_part']]
f = {'fin_all': None, 'fin_basic': 'basic', 'fin_special': 'special'}.get(ctx['part'])
if n == 'finansēšana' and ctx['part'].startswith('fin'):
return fin(f)
if not ctx['part'].startswith('fin'):
a2 = a2_codes.get((ctx['fund'], n))
if a2:
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=a2, blk=f"a2:{ctx['fund']}")
codes = cls_by_name.get(n)
if not codes:
return None
if ctx['part'].startswith('fin'):
return fin(f, codes)
return calc.total(ctx['fund'], 'revenue', year, 'forecast', codes=codes)
a2_codes = {}
for c in data['checks']:
if c['annex'] == 2 and c['sel'].get('label') and c['sel'].get('codes'):
a2_codes.setdefault((c['sel']['fund'], norm(c['sel']['label'])), c['sel']['codes'])
ctx = {'fund': 'basic', 'part': 'rev', 'exp_part': 'gross'}
for c in sorted([c for c in data['checks'] if c['annex'] == 1], key=lambda c: (c['row'], c['sel']['year'])):
n = norm(c['sel']['label'])
if n.startswith('valsts pamatbudžeta ieņēmumi'): ctx.update(fund='basic', part='rev')
if n.startswith('valsts speciālā budžeta ieņēmumi'): ctx.update(fund='special', part='rev')
if n.startswith('valsts budžeta finansiālā bilance'): ctx.update(part='fin_all')
if n.startswith('valsts pamatbudžeta finansiālā bilance'): ctx.update(part='fin_basic')
if n.startswith('valsts speciālā budžeta finansiālā bilance'): ctx.update(part='fin_special')
if n.startswith(('valsts pamatbudžeta', 'valsts speciālā budžeta')) and 'izdevumi' in n:
ctx.update(exp_part='maint' if 'uzturēšanas' in n else ('cap' if 'kapitālie' in n else 'gross'))
judge(c, a1_value(c['sel']['label'], c['sel']['year'], ctx, c['sel']['pct']), tol=0.005 if c['sel']['pct'] else 0.5)
# ------------------------------------------------------------------ all other annexes
for c in data['checks']:
if c['annex'] == 1:
continue
s = c['sel']
kind = s.get('kind')
if kind == 'grant_total':
got = sum(float(a['amount']) for a in A if a.get('purpose') == s['purpose'] and f"{a['scheme']}:{a['code']}" == s['code']
and (not s.get('periodFrom') or a.get('periodFrom') == s['periodFrom']))
judge(c, got); continue
if kind in ('fees_resort', 'fees_total'):
got = sum(float(a['amount']) for a in A if a.get('partOf') and a['src'].endswith('.p02') and a['year'] == s['year']
and (kind == 'fees_total' or a.get('holder') == s['holder']))
judge(c, got); continue
common = dict(purpose=s.get('purpose'), block=s.get('block'), ckind=s.get('ckind'), untilEnd=s.get('untilEnd', False), blk=s.get('blk'),
holder_view=bool(s.get('holderView')))
if kind == 'balance':
printed = data['blockcodes'].get(s.get('blk'), {})
if c['annex'] != 11 and ((s.get('purpose') is None and not s.get('block')) or s.get('fund') == 'special'):
inflow = calc.total(s.get('fund'), 'revenue', s['year'], 'forecast', **common)
else:
inflow = calc.total(s.get('fund'), 'resource', s['year'], s['nature'], **common)
judge(c, inflow - calc.total(s.get('fund'), 'expenditure', s['year'], s['nature'], **common)); continue
judge(c, calc.total(s.get('fund'), s['flow'], s['year'], s['nature'], codes=s.get('codes'), **common))
# ------------------------------------------------------------------ report
tot_n = tot_ok = 0
summary = {}
for an in sorted(res):
r = res[an]
tot_n += r['n']; tot_ok += r['ok']
summary[an] = {'checks': r['n'], 'ok': r['ok'], 'mismatch': r['n'] - r['ok'] - r['skip'], 'not_computed': r['skip']}
print(f"annex {an:2d}: {r['n']:6d} printed numbers, {r['ok']:6d} reproduced ({100 * r['ok'] / max(r['n'], 1):.2f}%), {r['skip']} not computed")
print(f'TOTAL: {tot_ok}/{tot_n} printed numbers reproduced from XML ({100 * tot_ok / max(tot_n, 1):.3f}%)')
json.dump({'xsd_valid': valid, 'allocations': len(A), 'annexes': summary, 'total': tot_n, 'reproduced': tot_ok,
'issues': {an: r['bad'][:300] for an, r in res.items()}},
open(os.path.join(OUT, f'{BID}.verify.json'), 'w', encoding='utf-8'), ensure_ascii=False, indent=1)