"""Convert an adopted Latvian state budget law (text + 12 annexes) into one XML file following valsts-budzets-0.1.xsd, plus a JSON list of every printed number as a check. Usage: python parse_budget.py YEAR LAW_HTML ANNEX_DIR OUT_DIR Annex files are found by number (P04.XLSX, 4_PIELIKUMS.XLS, ...). Storage rule: each fact is stored once, from its most detailed source; every other printed number becomes a check. """ import sys, os, re, json, html, hashlib, collections, datetime TOOLS = os.path.dirname(os.path.abspath(__file__)) ROOT = os.environ.get('BUDGET_ROOT', os.path.dirname(TOOLS)) KODI = os.path.join(ROOT, 'source', 'kodi') sys.path.insert(0, TOOLS) from rows import read_rows from rd import docx_tables, num from xml.sax.saxutils import quoteattr, escape from calc import Calc YEAR, LAW_HTML, ADIR, OUT = int(sys.argv[1]), sys.argv[2], sys.argv[3], sys.argv[4] Y3 = [YEAR, YEAR + 1, YEAR + 2] BID = f'lv-vb-{YEAR}' report = collections.defaultdict(list) # ------------------------------------------------------------------ helpers def norm(s): s = str(s).lower().replace(' ', ' ').replace('–', '-').replace('—', '-') s = re.sub(r'^i estādes', 'iestādes', s) return re.sub(r'[^0-9a-zāčēģīķļņšūž]+', ' ', s).strip() def slug(s, n=40): t = norm(s).translate(str.maketrans('āčēģīķļņšūž', 'acegiklnsuz')) return re.sub(r'\s+', '-', t)[:n].strip('-') def annex_file(n): for f in os.listdir(ADIR): m = re.match(r'^P?0*(\d+)[_.]', f, re.I) if m and int(m.group(1)) == n: return os.path.join(ADIR, f) raise FileNotFoundError(f'annex {n}') def sha(path): return hashlib.sha256(open(path, 'rb').read()).hexdigest() # ------------------------------------------------------------------ code book SCHEME_OF_FLOW = {'expenditure': 'ekk', 'resource': 'revenue', 'revenue': 'revenue', 'financing': 'financing'} book = collections.defaultdict(lambda: collections.defaultdict(set)) # scheme -> normname -> codes for nm, kc in json.load(open(os.path.join(KODI, 'klasdict.json'), encoding='utf-8')).items(): for kind, c in kc: book[kind][norm(nm)].add(c) KL = {'Izdevumi': 'ekk', 'Ieņēmumi': 'revenue', 'Finansēšana': 'financing'} for line in open(os.path.join(KODI, 'vk_codes.tsv'), encoding='utf-8'): kl, v = line.rstrip('\n').split('\t') m = re.match(r'^([A-Z]{0,2}\d[\d.]*)\s+(.+)$', v) if m and kl in KL: book[KL[kl]][norm(m.group(2))].add(m.group(1)) for nm, codes in json.load(open(os.path.join(KODI, 'codepairs.json'), encoding='utf-8')).items(): for c in codes: sch = 'financing' if c.startswith('F') or re.fullmatch(r'[A-Z]{1,2}F\d+', c) else ('ekk' if re.fullmatch(r'\d{4}', c) else ('revenue' if re.fullmatch(r'\d{5}', c) else 'law')) book[sch][norm(nm)].add(c) classitems = {} # (scheme, code) -> {'name', 'parent'} SECTION = {'ieņēmumi kopā': 'revenue', 'resursi izdevumu segšanai': 'resource', 'izdevumi kopā': 'expenditure', 'finansēšana': 'financing', 'finansiālā bilance': 'balance'} def canon(sch, c): if sch == 'financing' and re.fullmatch(r'[PKS]F\d{8}', c): return c[1:] return c def code_for(label, flow): n = norm(label) pref = SCHEME_OF_FLOW.get(flow, 'law') for sch in (pref, 'law', 'revenue', 'ekk', 'financing'): cs = book[sch].get(n) if cs: cs = {canon(sch, x) for x in cs} c = sorted(cs, key=lambda x: (len(x), x))[0] if len(cs) > 1: report['ambiguous_code'].append(f'{label[:60]} -> {sorted(cs)} (took {c})') return sch, c c = 'L-' + slug(label) report['synthesised_code'].append(label) return 'law', c INTRA = ('savstarpējie transferti', 'no valsts pamatbudžeta uz valsts pamatbudžetu', 'no valsts speciālā budžeta uz valsts speciālo budžetu', 'atmaksām valsts pamatbudžet', 'atmaksa valsts budžetā par veiktajiem', 'valsts pamatbudžeta iestāžu saņemtie transferti no valsts pamatbudžeta', 'pārējie valsts pamatbudžetā saņemtie transferti no valsts pamatbudžeta', 'valsts speciālā budžeta iestāžu saņemtie transferti no valsts speciālā budžeta') def is_intra(name): n = norm(name) return any(norm(k) in n for k in INTRA) VOTE_WEIGHT = {'a2': 1, 'a3': 1, 'a4': 1, 'a5': 1, 'a11': 0} CUR_ANNEX = ['a4'] def register_class(sch, code, name, parent): k = (sch, code) d = classitems.setdefault(k, {'name': name, 'parent': None, 'votes': collections.Counter()}) w = VOTE_WEIGHT.get(CUR_ANNEX[0], 1) if w: d['votes'][parent] += w def structural_ok(child, parent): """Numeric classification codes carry their own hierarchy: 21200 cannot sit under 21100, 7131 can sit under 7130.""" cs, cc = child.split(':', 1) ps, pc = parent.split(':', 1) if not (cc.isdigit() and pc.isdigit() and cs == ps and len(cc) == len(pc)): return True stem = pc.rstrip('0') return cc != pc and cc.startswith(stem) def finalize_classes(): """Canonical parent = majority vote over indented annex blocks, restricted to structurally possible parents.""" for k, d in classitems.items(): votes = d['votes'] ck = f'{k[0]}:{k[1]}' real = [(p, n) for p, n in votes.most_common() if p and structural_ok(ck, p)] rejected = [p for p in votes if p and not structural_ok(ck, p)] if rejected: report['parent_rejected_by_structure'].append(f'{ck} {d["name"][:40]}: not under {rejected}') d['parent'] = real[0][0] if real and (not votes.get(None) or real[0][1] >= votes[None] or rejected) else None if len(real) > 1: report['class_parent_votes'].append(f'{k[0]}:{k[1]} {d["name"][:50]} votes {dict(votes)} -> {d["parent"]}') # ------------------------------------------------------------------ actors and purposes actors, purposes = {}, {} VPK_RESORTS = {l.split('\t')[0] for l in open(os.path.join(KODI, 'vpk_resorts.tsv'), encoding='utf-8') if l.strip()} def resort_actor(code, name): aid = f'{BID}.ac.r{code}' vpk = f'{code}-0000' if f'{code}-0000' in VPK_RESORTS else None actors.setdefault(aid, {'kind': 'resort', 'code': code, 'name': name, 'vpk': vpk}) pid = f'{BID}.pr.{code}' purposes.setdefault(pid, {'kind': 'resort', 'code': code, 'name': name, 'holder': aid}) return aid, pid def prog_purpose(rcode, pcode, name, function, parent_pid): pid = f'{BID}.pr.{rcode}.{pcode}' kind = 'programme' if pcode.endswith('.00.00') else 'subprogramme' d = purposes.setdefault(pid, {'kind': kind, 'code': pcode, 'name': name, 'parent': parent_pid, 'holder': f'{BID}.ac.r{rcode}', 'function': function}) if function and not d.get('function'): d['function'] = function return pid def project_purpose(rcode, pcode, projcode, name, parent_pid): pid = f'{BID}.pj.{rcode}.{pcode or "x"}.{slug(projcode, 60)}' purposes.setdefault(pid, {'kind': 'project', 'code': projcode, 'name': name, 'parent': parent_pid, 'holder': f'{BID}.ac.r{rcode}'}) return pid def municipality(name): aid = f'{BID}.ac.m-{slug(name)}' actors.setdefault(aid, {'kind': 'municipality', 'name': name}) return aid # block (core/eu) per subprogramme, from Valsts kase execution data; fallback by code range blockmap = {} for line in open(os.path.join(KODI, 'vk_blocks.tsv'), encoding='utf-8'): g, m, p, sp, blk, bt, n = line.rstrip('\n').split('\t') if int(g) in (YEAR, YEAR - 1): blockmap.setdefault((m, sp), 'eu' if blk.startswith('Ārvalstu') else 'core') def block_of(rcode, pcode): b = blockmap.get((rcode, pcode)) if b: return b report['block_fallback'].append(f'{rcode} {pcode}') return 'eu' if pcode[:2] in ('60', '61', '62', '63', '64', '65', '66', '67', '68', '69', '70', '71', '72', '73', '74', '75', '76', '77', '78', '79', '80', '81', '82', '83', '84', '85') else 'core' # ------------------------------------------------------------------ outputs allocs, checks = [], [] blockcodes = collections.defaultdict(lambda: collections.defaultdict(set)) # block id -> flow -> printed codes def register_block(bid, lines): for ln in lines: if not ln.get('skip') and not ln.get('section') and ln.get('flow') not in (None, 'balance'): blockcodes[bid][ln['flow']].add(f"{ln['scheme']}:{ln['code']}") _alloc_ids = collections.Counter() def add_alloc(**a): """Allocation id = source annex + source row + year (+ 'l' for 'later years' column, + -N if the same cell yields several facts). Derived from the source, so re-parsing the same law gives the same ids.""" base = f'{BID}.a.{a.pop("srcAnnex")}.r{a.get("srcRow", 0)}.y{a["year"]}{"l" if a.get("untilEnd") else ""}' _alloc_ids[base] += 1 a['id'] = base if _alloc_ids[base] == 1 else f'{base}-{_alloc_ids[base]}' allocs.append(a) return a['id'] def add_check(annex, row, value, **sel): checks.append({'annex': annex, 'row': row, 'value': value, 'sel': sel}) pending = [] # (bid, flow, ck, alloc kwargs) for lines of detailed blocks; leafness decided on the canonical tree def queue_alloc(_bid, _flow, _ck, **kw): pending.append((_bid, _flow, _ck, kw)) def canon_chain(ck, parent): out, k = [], ck while k and k not in out: out.append(k) k = parent.get(k) return out def finalize_lines(): parent = {f'{k[0]}:{k[1]}': v['parent'] for k, v in classitems.items() if v['parent']} below = {} # (bid, flow, ck) -> printed codes in that block whose canonical chain contains ck for bid, fl in blockcodes.items(): for flow, codes in fl.items(): for c in codes: for anc in canon_chain(c, parent): if anc in codes: below.setdefault((bid, flow, anc), []).append(c) for c in checks: r = c['sel'].get('codes') if isinstance(r, dict) and 'resolve' in r: bid, flow, ck = r['resolve'] c['sel']['codes'] = sorted(set(below.get((bid, flow, ck), [ck])) | {ck}) for bid, flow, ck, kw in pending: if len(set(below.get((bid, flow, ck), [ck])) - {ck}) == 0: add_alloc(**kw) # ------------------------------------------------------------------ generic tree-annex reader def tree_blocks(rows, label_col, amount_cols, header_fn, level_fn): """Split rows into blocks keyed by header path; each block has lines with level, flow, amounts.""" blocks, path, cur = [], {}, None for r in rows: cells = r['cells'] h = header_fn(r, path) if h: path = h cur = {'path': dict(path), 'lines': []} blocks.append(cur) continue label = str(cells[label_col]) if len(cells) > label_col else '' amts = {} for y, ci in amount_cols.items(): v = num(cells[ci]) if ci < len(cells) else None if v is not None: amts[y] = v if not label or not amts: continue if cur is None: cur = {'path': dict(path), 'lines': []} blocks.append(cur) cur['lines'].append({'row': r['row'], 'label': label, 'level': level_fn(r, label), 'amounts': amts}) return blocks def resolve_lines(block): """Assign flow, code and parent code to every line; mark leaves (no deeper line follows in the same section).""" lines, flow, stack = block['lines'], None, [] for i, ln in enumerate(lines): n = norm(ln['label']) if n in SECTION: flow = SECTION[n] ln.update(flow=flow, section=True, scheme='law', code='S-' + flow, leaf=False) stack = [(ln['level'], ln)] continue if flow is None: ln.update(flow=None, skip=True) continue while stack and stack[-1][0] >= ln['level']: stack.pop() parent = stack[-1][1] if stack else None sch, code = code_for(ln['label'], flow) ln.update(flow=flow, scheme=sch, code=code, section=False) if parent is not None and not parent.get('section'): register_class(sch, code, ln['label'], f'{parent["scheme"]}:{parent["code"]}') else: register_class(sch, code, ln['label'], None) stack.append((ln['level'], ln)) for i, ln in enumerate(lines): if ln.get('skip'): continue desc = [] if ln.get('section') else [f"{ln['scheme']}:{ln['code']}"] for x in lines[i + 1:]: if x.get('skip'): continue if x.get('section') or x['level'] <= ln['level'] or x.get('flow') != ln.get('flow'): break desc.append(f"{x['scheme']}:{x['code']}") ln['codes'] = desc for i, ln in enumerate(lines): if ln.get('section') or ln.get('skip'): continue nxt = next((x for x in lines[i + 1:] if not x.get('skip')), None) ln['leaf'] = nxt is None or nxt.get('section') or nxt['level'] <= ln['level'] or nxt.get('flow') != ln['flow'] return lines # ------------------------------------------------------------------ annex 4 and 5 (2026 appropriations by subprogramme) label_levels = collections.defaultdict(collections.Counter) # (flow, normlabel) -> relative level counts, learned for annex 11 def parse_programmes(n, fund): CUR_ANNEX[0] = f'a{n}' path_ = annex_file(n) rows = read_rows(path_, label_col=2) progs_seen = {} def header(r, path): c = r['cells'] lab = str(c[2]) if len(c) > 2 else '' m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab) if m and num(c[3] if len(c) > 3 else '') is None: aid, pid = resort_actor(m.group(1), m.group(2)) return {'resort': m.group(1), 'rpid': pid} code = str(c[0]) if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path: rc = path['resort'] parent = path['rpid'] pcode = code if not code.endswith('.00.00'): pp = code[:2] + '.00.00' if (rc, pp) in progs_seen: parent = progs_seen[(rc, pp)] func = str(c[1]) if re.fullmatch(r'\d{2}\.\d{3}', str(c[1])) else None pid = prog_purpose(rc, pcode, lab, func, parent) progs_seen[(rc, pcode)] = pid return {'resort': rc, 'rpid': path['rpid'], 'pcode': pcode, 'pid': pid, 'parent': parent, 'func': func} return None blocks = tree_blocks(rows, 2, {YEAR: 3}, header, lambda r, lab: r['indent']) # leaf blocks: subprogrammes, or programmes without subprogrammes has_child = set(b['path'].get('parent') for b in blocks if b['path'].get('pcode')) for bi, b in enumerate(blocks): lines = resolve_lines(b) bid = f'a{n}:{bi}' register_block(bid, lines) p = b['path'] if 'pid' in p: sel_purpose = p['pid'] elif 'rpid' in p: sel_purpose = p['rpid'] else: sel_purpose = None leafblock = 'pid' in p and p['pid'] not in has_child blk = block_of(p['resort'], p['pcode']) if 'pcode' in p else None # learn relative levels for annex 11 base = None for ln in lines: if ln.get('section'): base = ln['level'] elif not ln.get('skip') and base is not None: label_levels[(ln['flow'], norm(ln['label']))][ln['level'] - base] += 1 for ln in lines: if ln.get('skip') or ln['flow'] == 'balance': if ln.get('flow') == 'balance' or ln.get('section') and ln['flow'] == 'balance': add_check(n, ln['row'], ln['amounts'][YEAR], kind='balance', fund=fund, year=YEAR, purpose=sel_purpose, nature='appropriation', blk=bid) continue nature = 'forecast' if ln['flow'] == 'revenue' and fund == 'basic' else 'appropriation' if fund == 'special' and ln['flow'] == 'revenue': nature = 'forecast' add_check(n, ln['row'], ln['amounts'][YEAR], fund=fund, flow=ln['flow'], year=YEAR, purpose=sel_purpose, code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, nature=nature, blk=bid) store = leafblock and not ln.get('section') if fund == 'basic' and ln['flow'] == 'revenue': store = False # basic-budget revenue is stored from annex 2 if store: queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex=f'p{n:02d}', flow=ln['flow'], fund=fund, nature=nature, year=YEAR, holder=f'{BID}.ac.r{p["resort"]}', purpose=p['pid'], block=blk, function=p.get('func'), scheme=ln['scheme'], code=ln['code'], amount=ln['amounts'][YEAR], src=f'{BID}.src.p{n:02d}', srcRow=ln['row']) # ------------------------------------------------------------------ annex 2 (revenue forecasts, 3 years) def parse_revenue(): CUR_ANNEX[0] = 'a2' raw = read_rows(annex_file(2), label_col=0) # the label column differs between years (2025 .xls has an extra empty first column): find it from the header row lc = next((i for r in raw for i, c in enumerate(r['cells']) if str(c).startswith('Ieņēmumu avots')), 0) rows = read_rows(annex_file(2), label_col=lc) if lc else raw part, fund, mode = None, 'basic', 'tree' lines, fees, info = [], [], [] cur_resort = None for r in rows: c = r['cells'] lab = str(c[lc]) if lc < len(c) else '' if re.match(r'^I\.\s', lab): fund, mode = 'basic', 'tree'; continue if re.match(r'^II\.\s', lab): fund, mode = 'special', 'tree'; continue if lab.startswith('Ministrija (cita'): mode = 'fees'; continue if lab.startswith('Informatīvi'): mode = 'info'; continue amts = {y: num(c[lc + 1 + i]) for i, y in enumerate(Y3) if lc + 1 + i < len(c) and num(c[lc + 1 + i]) is not None} if not amts or not lab: continue if mode == 'tree': lines.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'amounts': amts, 'fund': fund}) elif mode == 'fees': fees.append({'row': r['row'], 'label': lab, 'level': r['indent'], 'bold': r['bold'], 'amounts': amts}) else: info.append({'row': r['row'], 'label': lab, 'amounts': amts}) stored = {} for fund in ('basic', 'special'): blk = {'lines': [dict(l) for l in lines if l['fund'] == fund]} resolve_lines(blk) register_block(f'a2:{fund}', blk['lines']) for ln in blk['lines']: if ln.get('skip'): continue for y, v in ln['amounts'].items(): add_check(2, ln['row'], v, fund=fund, flow='revenue', year=y, nature='forecast', label=ln['label'], code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=ln.get('codes'), blk=f'a2:{fund}') if ln.get('leaf') and not ln.get('section'): if fund == 'special' and y == YEAR: continue # 2026 special revenue is stored from annex 5 (by subprogramme) aid = add_alloc(srcAnnex='p02', flow='revenue', fund=fund, nature='forecast', year=y, holder=None, purpose=None, scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p02', srcRow=ln['row']) stored[(fund, ln['code'], y)] = aid # fees by administering resort: a breakdown of stored revenue leaves -> partOf leaf_codes = sorted({k[1] for k in stored if k[0] == 'basic'}, key=len, reverse=True) for f in fees: n_ = norm(f['label']) if n_.startswith('ieņēmumi valsts pamatbudžetā kopā'): for y, v in f['amounts'].items(): add_check(2, f['row'], v, kind='fees_total', year=y) continue if f['bold']: name = f['label'] m = [a for a, d in actors.items() if d['kind'] == 'resort' and norm(d['name']) == n_] cur_resort = m[0] if m else None if not cur_resort: report['fee_resort_unmatched'].append(f['label']) for y, v in f['amounts'].items(): add_check(2, f['row'], v, kind='fees_resort', holder=cur_resort, year=y) continue sch, code = code_for(f['label'], 'revenue') register_class(sch, code, f['label'], None) parent = next((stored[('basic', lc, y0)] for lc in leaf_codes for y0 in [YEAR] if code.startswith(lc.rstrip('0') or lc) and ('basic', lc, YEAR) in stored), None) if parent is None: report['fee_parent_unmatched'].append(f'{code} {f["label"][:60]}') for y, v in f['amounts'].items(): par = None if parent: pc = next(k for k, a in stored.items() if a == parent)[1] par = stored.get(('basic', pc, y)) add_alloc(srcAnnex='p02', flow='revenue', fund='basic', nature='forecast', year=y, holder=cur_resort, purpose=None, scheme=sch, code=code, amount=v, partOf=par, src=f'{BID}.src.p02', srcRow=f['row']) for i in info: for y, v in i['amounts'].items(): params.append({'id': f'{BID}.par.p02-{i["row"]}-{y}', 'name': i['label'], 'year': y, 'value': v, 'unit': 'EUR', 'src': f'{BID}.src.p02'}) # ------------------------------------------------------------------ annex 3 (resort summaries, 3 years; 2027-28 stored as ceilings) def parse_resort_summary(): CUR_ANNEX[0] = 'a3' rows = read_rows(annex_file(3), label_col=0) fund_names = {'valsts pamatbudžets': 'basic', 'valsts speciālais budžets': 'special'} def header(r, path): lab = str(r['cells'][0]) n_ = norm(lab) if num(r['cells'][1] if len(r['cells']) > 1 else '') is not None: return None if n_ in fund_names: p = {k: v for k, v in path.items() if k in ('resort', 'rpid')} p['fund'] = fund_names[n_] return p m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab) if m: aid, pid = resort_actor(m.group(1), m.group(2)) return {'resort': m.group(1), 'rpid': pid, 'fund': 'basic'} if re.match(r'^I\.\s+Valsts pamatfunkciju', lab): p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'core'; return p if re.match(r'^II\.\s+ES politiku', lab): p = {k: v for k, v in path.items() if k != 'block'}; p['block'] = 'eu'; return p return None blocks = tree_blocks(rows, 0, {Y3[0]: 1, Y3[1]: 2, Y3[2]: 3}, header, lambda r, lab: r['indent']) keys = [(b['path'].get('resort'), b['path'].get('fund')) for b in blocks] for bi, b in enumerate(blocks): lines = resolve_lines(b) bid = f'a3:{bi}' register_block(bid, lines) p = b['path'] fund = p.get('fund', 'basic') # a resort's special-budget part may appear without its own fund header: detect by section content later leafblock = 'resort' in p and 'block' in p sel = dict(fund=fund, purpose=p.get('rpid'), block=p.get('block'), blk=bid, holderView=True) for ln in lines: if ln.get('skip'): continue for y, v in ln['amounts'].items(): if ln['flow'] == 'balance': add_check(3, ln['row'], v, kind='balance', year=y, nature='appropriation' if y == YEAR else 'ceiling', **sel) continue nature = 'forecast' if ln['flow'] == 'revenue' else ('appropriation' if y == YEAR else 'ceiling') add_check(3, ln['row'], v, flow=ln['flow'], year=y, nature=nature, code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel) if leafblock and not ln.get('section') and y != YEAR and ln['flow'] != 'revenue': queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p03', flow=ln['flow'], fund=fund, nature='ceiling', year=y, holder=f'{BID}.ac.r{p["resort"]}', purpose=p['rpid'], block=p['block'], scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p03', srcRow=ln['row']) # ------------------------------------------------------------------ annex 11 (long-term commitments) def parse_commitments(): CUR_ANNEX[0] = 'a11' rows = read_rows(annex_file(11), label_col=1) hdr = next(r for r in rows if 'Projekta kods' in [str(x) for x in r['cells']]) years = {} for i, c in enumerate(hdr['cells']): m = re.match(r'^(\d{4})\.', str(c)) if m: years[int(m.group(1))] = i elif str(c).startswith('Tālākā'): years['later'] = i last = max(y for y in years if y != 'later') kinds = {} def level_fn(r, lab): cnt = None for fl in ('expenditure', 'resource', 'financing', 'revenue'): c = label_levels.get((fl, norm(lab))) if c: cnt = c.most_common(1)[0][0] + 1 break if cnt is None and norm(lab) not in SECTION: report['a11_level_unknown'].append(lab) return 99 return cnt or 0 progs_seen = {} state = {'prev_header': None} def header(r, path): h = _header(r, path) state['prev_header'] = h is not None and h.get('_kindable', False) if h is not None: h.pop('_kindable', None) return h def _header(r, path): c = r['cells'] lab = str(c[1]) if len(c) > 1 else '' if any(num(c[i]) is not None for i in years.values() if i < len(c)): return None if lab.upper().startswith('VALSTS PAMATBUDŽETS'): return {'fund': 'basic'} if lab.upper().startswith('VALSTS SPECIĀLAIS BUDŽETS'): return {'fund': 'special'} m = re.match(r'^(\d{10})\s+(.+)$', lab) if m: kinds[m.group(1)] = m.group(2) register_class('commitment', m.group(1), m.group(2), None) if state['prev_header']: # directly under a resort or programme header p = dict(path); p['kind'] = m.group(1) if 'pcode' not in p: p['rkind'] = m.group(1) return p return {'fund': path.get('fund', 'basic'), 'kind': m.group(1), 'topkind': m.group(1)} # new top-level commitment-kind section m = re.match(r'^(\d{2})\.?\s+(\D.+)$', lab) if m: aid, pid = resort_actor(m.group(1), m.group(2)) return {'fund': path.get('fund', 'basic'), 'resort': m.group(1), 'rpid': pid, '_kindable': True, 'topkind': path.get('topkind'), 'kind': path.get('topkind')} code = str(c[0]) if re.fullmatch(r'\d{2}\.\d{2}\.\d{2}', code) and 'resort' in path: rc = path['resort'] pid = f'{BID}.pr.{rc}.{code}' if pid not in purposes: parent = purposes.get(f'{BID}.pr.{rc}.{code[:2]}.00.00') and f'{BID}.pr.{rc}.{code[:2]}.00.00' or path['rpid'] prog_purpose(rc, code, lab, None, parent) report['a11_new_programme'].append(f'{rc} {code} {lab[:50]}') p = {k: v for k, v in path.items() if k in ('fund', 'resort', 'rpid', 'topkind', 'rkind')} p.update(pcode=code, pid=pid, _kindable=True, kind=path.get('rkind') or path.get('topkind')) return p proj = str(c[2]) if len(c) > 2 else '' if proj and 'resort' in path: parent = path.get('pid') or path['rpid'] pj = project_purpose(path['resort'], path.get('pcode'), proj, lab, parent) p = dict(path); p['project'] = proj; p['pjid'] = pj return p return None amount_cols = {y: i for y, i in years.items()} blocks = tree_blocks(rows, 1, amount_cols, header, level_fn) def more_specific(k2, k): # commitment kind k2 is a sub-type of k (or equal); codes nest in 2-digit groups from calc import kind_stem if not k: return True return bool(k2 and k2.startswith(kind_stem(k))) def refined(bi): """A detailed block is refined (not a leaf) if, within the same resort+programme, a later or earlier block adds a project under a compatible kind, or shows the same project under a more specific kind.""" p = blocks[bi]['path'] for j, b2 in enumerate(blocks): if j == bi: continue q = b2['path'] if q.get('resort') != p.get('resort') or q.get('fund') != p.get('fund') or not q.get('pcode'): continue if q['pcode'] != p.get('pcode'): # a programme block is refined by its subprogramme blocks (same resort, same programme number) if (p.get('pcode') or '').endswith('.00.00') and q['pcode'][:2] == p['pcode'][:2] and not q['pcode'].endswith('.00.00') \ and not p.get('project') and more_specific(q.get('kind'), p.get('kind')): return True continue if not p.get('project') and q.get('project') and more_specific(q.get('kind'), p.get('kind')): return True if p.get('project') and q.get('project') == p['project'] and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')): return True if not p.get('project') and not q.get('project') and q.get('kind') != p.get('kind') and more_specific(q.get('kind'), p.get('kind')): return True return False # leaf blocks: those whose path is not extended by a later block def key(p): return (p.get('fund'), p.get('kind'), p.get('resort'), p.get('pcode'), p.get('project')) keys = [key(b['path']) for b in blocks] def extends(a, b): # b is strictly more specific than a return a != b and all(x is None or x == y or (i == 1 and x and y and y.startswith(x.rstrip('0'))) for i, (x, y) in enumerate(zip(a, b))) for bi, b in enumerate(blocks): lines = resolve_lines(b) bid = f'a11:{bi}' register_block(bid, lines) p = b['path'] leafblock = bool(p.get('pcode')) and not refined(bi) sel = dict(fund=p.get('fund', 'basic'), ckind=p.get('kind'), purpose=p.get('pjid') or p.get('pid') or p.get('rpid'), blk=bid) for ln in lines: if ln.get('skip'): continue for y, v in ln['amounts'].items(): yy, until = (last + 1, True) if y == 'later' else (y, False) if ln['flow'] == 'balance': add_check(11, ln['row'], v, kind='balance', year=yy, untilEnd=until, nature='commitment', **sel) continue add_check(11, ln['row'], v, flow=ln['flow'], year=yy, untilEnd=until, nature='commitment', code=None if ln.get('section') else f"{ln['scheme']}:{ln['code']}", codes=None if ln.get('section') else {'resolve': [bid, ln['flow'], f"{ln['scheme']}:{ln['code']}"]}, **sel) if leafblock and not ln.get('section') and v != 0: queue_alloc(bid, ln['flow'], f"{ln['scheme']}:{ln['code']}", srcAnnex='p11', flow=ln['flow'], fund=sel['fund'], nature='commitment', year=yy, untilEnd=until, holder=f'{BID}.ac.r{p["resort"]}' if p.get('resort') else None, purpose=sel['purpose'], commitmentKind=p.get('kind'), scheme=ln['scheme'], code=ln['code'], amount=v, src=f'{BID}.src.p11', srcRow=ln['row']) # ------------------------------------------------------------------ annexes 6-10 (earmarked grants to municipalities) def parse_grants(n): path_ = annex_file(n) t = docx_tables(path_)[-1] title = next((r[0] for r in t if r and len(r[0]) > 40), f'{n}. pielikums') gid = f'{BID}.pr.g{n:02d}' purposes[gid] = {'kind': 'grant', 'code': f'P{n:02d}', 'name': title} period = (f'{YEAR}-01-01', f'{YEAR}-12-31') cols = None for ri, r in enumerate(t): first = r[0] if r else '' m = re.match(r'^(I{1,2})\.\s', first) if m: period = (f'{YEAR}-01-01', f'{YEAR}-08-31') if m.group(1) == 'I' else (f'{YEAR}-09-01', f'{YEAR}-12-31') continue if first.startswith('Pašvaldības'): cols = r; continue vals = [num(x) for x in r[1:]] if not first or not any(v is not None for v in vals): continue n_ = norm(first) if n_ in ('kopā', 'pavisam kopā'): heads_ = [norm(h) for h in (cols or [])[1:]] for ci, v in enumerate(vals): if v is not None: h = heads_[ci] if ci < len(heads_) else '' add_check(n, ri + 1, v, kind='grant_total', purpose=gid, code=f'law:G{n:02d}-{slug(h or "summa", 30) or "summa"}', periodFrom=None if n_ == 'pavisam kopā' else period[0], periodTo=None if n_ == 'pavisam kopā' else period[1], year=YEAR) continue recip = f'{BID}.ac.unallocated' if n_.startswith('nesadalītie') else municipality(first) if recip.endswith('unallocated'): actors[recip] = {'kind': 'unallocated', 'name': 'Nesadalītie līdzekļi'} heads = [norm(h) for h in (cols or [])[1:]] # column roles: 'pavisam kopā' = total; 'tai skaitā …' = part of previous; otherwise the main amount ids = {} total_idx = next((i for i, h in enumerate(heads) if h.startswith('pavisam kopā')), None) main_idx = 0 order = ([total_idx] if total_idx is not None else []) + [i for i in range(len(vals)) if i != total_idx] for ci in order: v = vals[ci] if ci < len(vals) else None if v is None: continue h = heads[ci] if ci < len(heads) else '' part = None if h.startswith('tai skaitā'): part = ids.get(ci - 1) elif total_idx is not None and ci != total_idx: part = ids.get(total_idx) aid = add_alloc(srcAnnex=f'p{n:02d}', flow='expenditure', fund='basic', nature='earmarkedGrant', year=YEAR, periodFrom=period[0], periodTo=period[1], recipient=recip, purpose=gid, scheme='law', code=f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', amount=v, partOf=part, src=f'{BID}.src.p{n:02d}', srcRow=ri + 1) register_class('law', f'G{n:02d}-{slug(h or "summa", 30) or "summa"}', (cols[ci + 1] if cols and ci + 1 < len(cols) else 'Summa') or 'Summa', None) ids[ci] = aid # ------------------------------------------------------------------ annex 1 (consolidated; checks and GDP parameter) params = [] def parse_consolidated(): rows = read_rows(annex_file(1), label_col=0) hdr = next(r for r in rows if any(re.match(r'^\d{4}\.', str(c)) for c in r['cells'])) ycols = {int(re.match(r'^(\d{4})', str(c)).group(1)): i for i, c in enumerate(hdr['cells']) if re.match(r'^\d{4}\.', str(c))} lc = next(i for i, c in enumerate(hdr['cells']) if str(c) == 'Nosaukums') pct = False for r in rows: c = r['cells'] lab = str(c[lc]) if lc < len(c) else '' if lab.startswith('Procentos no IKP'): pct = True; continue for y, i in ycols.items(): v = num(c[i]) if i < len(c) else None if v is None or not lab: continue if lab.startswith('IKP milj'): params.append({'id': f'{BID}.par.ikp-{y}', 'name': 'IKP prognoze', 'code': 'IKP', 'year': y, 'value': v, 'unit': 'milj. EUR', 'src': f'{BID}.src.p01'}) continue add_check(1, r['row'], v, kind='a1', label=lab, indent=r['indent'], pct=pct, year=y) # ------------------------------------------------------------------ law text provisions = [] def parse_law(): s = open(LAW_HTML, encoding='utf-8', errors='replace').read() chapter = None for m in re.finditer(r"