#!/usr/bin/env python3 """NAP2027: starpposma JSON (parse_nap.py) + institūciju sasaiste (sources/nap2027/dalibnieki.yaml) → data/nap2027/lv-nap2027.xml python3 tools/parse_nap.py sources/nap2027/NAP2027.pdf build/nap2027.json python3 tools/build_nap2027.py build/nap2027.json data/nap2027/lv-nap2027.xml """ import datetime as dt import difflib import hashlib import json import os import re import sys from xml.sax.saxutils import escape, quoteattr import yaml ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) DOC = "lv-nap2027" PDF = "sources/nap2027/NAP2027.pdf" SOURCE_URL = "https://www.mk.gov.lv/lv/media/15162/download" RETRIEVED = "2026-10-11" FUNDING_NAMES = { "VB": "Valsts budžets", "ES-FONDI": "Eiropas Savienības fondi", "CITI": "Citi finanšu avoti", "PASV": "Pašvaldību budžeti", "HORIZON": "Horizon Europe", "DIGITAL": "Digital Europe", "URBAN": "Urban Europe", "EEZ-NFI": "EEZ un Norvēģijas finanšu instruments", "CH": "Šveices programma", "ES-JAUNATNE": "ES programmas jaunatnes jomā", "PILSONISKA": "Pilsoniskā iniciatīva", } SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789") def sq(s): return re.sub(r"[^0-9a-zāčēģīķļņšūž%]", "", (s or "").lower().translate(SUBS)) def a(name, value): return f" {name}={quoteattr(str(value))}" if value not in (None, "") else "" def el(tag, text, **attrs): if text in (None, ""): return "" return f"<{tag}{''.join(a(k, v) for k, v in attrs.items())}>{escape(str(text))}" def main(src, out): d = json.load(open(src, encoding="utf-8")) m = yaml.safe_load(open(os.path.join(ROOT, "sources/nap2027/dalibnieki.yaml"), encoding="utf-8")) actors = {x["label"]: x for x in m["actors"]} used_actors, used_funding, unresolved, used_current = {}, {}, [], set() # ---------------------------------------------------------------- actors def actor_list(printed): if not printed: return [] parts = [p.strip() for p in re.split(r",\s*", printed) if p.strip()] out_ = [] for p in parts: if p in m.get("qualifiers", {}): if out_ and out_[-1]["label"] == m["qualifiers"][p]: out_[-1]["qualifier"] = p continue for q in m.get("split", {}).get(p, [p]): if q not in actors: unresolved.append(q) continue used_actors[q] = actors[q] out_.append({"label": q, "org": actors[q].get("org")}) return out_ cur = {(n, c["label"]): c for c in m.get("current", []) for n in c["items"]} def actors_xml(tag, printed, n=None): lst = actor_list(printed) for x in lst: c = cur.get((n, x["label"])) if c: x["currentOrg"], x["currentSince"] = c["org"], c["since"] used_current.add((n, x["label"])) if not lst: return "" inner = "".join(f"" for x in lst) return f"<{tag}{a('printed', printed)}>{inner}" # ---------------------------------------------------------------- funding fmap = m["funding"] def funding_xml(printed): if not printed: return "" srcs = [] for p in [p.strip() for p in re.split(r",\s*", printed) if p.strip()]: labels = [p] if p in fmap else [x.strip() for x in p.split("/")] for lab in labels: if lab not in fmap: unresolved.append("finansējums: " + lab) continue used_funding[fmap[lab]] = True srcs.append(f"") return f"{''.join(srcs)}" if srcs else "" # ---------------------------------------------------------------- task indicator → document indicator inds = [x for x in d["items"] if x["kind"] == "indicator"] rv_of = lambda sec: ".".join(sec.split(".")[:2]) match_stats = {"exact": 0, "shortened": 0, "similar": 0, "split": 0, "task-level": 0} def best_indicator(name, sec): s = sq(name) if len(s) < 4: return None, None same = [i for i in inds if rv_of(i["section"]) == rv_of(sec)] for pool in (same, inds): for i in pool: if sq(i["name"]) == s: return i, "exact" for i in pool: t = sq(i["name"]) head = sq(re.split(r"[(,]", i["name"])[0]) # name without the bracketed or comma explanation if len(s) >= 10 and t.startswith(s) and (len(s) >= 0.6 * len(t) or head == s): return i, "shortened" # printed without the bracketed explanation, or slightly shortened r, i = max(((difflib.SequenceMatcher(None, s, sq(i["name"])).ratio() + (0.02 if i in same else 0), i) for i in inds), key=lambda z: z[0]) return (i, "similar") if r >= 0.85 else (None, None) def split_merged(name, sec): """Several indicator names printed without an empty line between them: split before capitalised words (not all-caps abbreviations); keep matched pieces separate, join unmatched neighbours back together.""" w = name.split(" ") cuts = [k for k in range(1, len(w)) if w[k][:1].isupper() and not w[k].isupper() and not w[k - 1].endswith(("(", "–", "-", "/"))] if not cuts: return None segs = [" ".join(w[i:j]) for i, j in zip([0] + cuts, cuts + [len(w)])] res = [] for sgm in segs: i, how = best_indicator(sgm, sec) if i and how in ("exact", "shortened"): res.append((sgm, i, how)) elif res and res[-1][1] is None: res[-1] = (res[-1][0] + " " + sgm, None, None) else: res.append((sgm, None, None)) return res if any(r[1] for r in res) and len(res) > 1 else None def task_indicators(t): xs = [] for nm in t.get("indicatorNames") or []: i, how = best_indicator(nm, t["section"]) parts = [(nm, i, how)] if i else (split_merged(nm, t["section"]) or [(nm, None, None)]) if len(parts) > 1: match_stats["split"] += 1 for name, i, how in parts: match_stats[how or "task-level"] += 1 ref = f"" if i else "" xs.append(f"{el('Name', name)}{ref}") return "".join(xs) # ---------------------------------------------------------------- items def item_xml(x): iid = f"{DOC}.i{x['number']:03d}" body = "" if x["kind"] == "indicator": body = ("" + el("Name", x.get("name")) + el("Unit", x.get("unit")) + el("BaseYear", x.get("baseYear")) + el("BaseValue", x.get("baseValue")) + el("Target", x.get("target2024"), year="2024") + el("Target", x.get("target2027"), year="2027") + el("DataSource", x.get("source")) + "") elif x["kind"] == "task": body = ("" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"), x["number"]) + actors_xml("CoResponsible", x.get("coResponsible"), x["number"]) + funding_xml(x.get("funding")) + task_indicators(x) + "") else: title = x.get("title") if x["kind"] == "strategicGoal" else None body = el("Title", title) + el("Area", x.get("area")) + el("Text", x["text"]) refs = "".join(f'' for n in x.get("footnoteRefs", [])) return f'{body}{refs}' # ---------------------------------------------------------------- sections (tree) secs = [s for s in d["sections"] if s["kind"] not in ("front", "annex")] by_parent = {} for s in secs: by_parent.setdefault(s.get("parent"), []).append(s) items_by_sec = {} for x in d["items"]: items_by_sec.setdefault(x["section"], []).append(x) def sec_num(s): mm = re.search(r"(\d+)$", s["id"]) return mm.group(1) if s["kind"] in ("priority", "actionLine", "theme") and mm else None def section_xml(s): sid = f"{DOC}.{s['id']}" parts = [el("Title", s["title"])] f = s.get("funding") if f and f.get("millionEur"): parts.append(el("IndicativeFunding", f["text"], millionEur=f["millionEur"].replace(",", "."))) for nt in s.get("notes", []): if isinstance(nt, dict): parts.append(el("Note", nt["text"], area=nt.get("area"))) else: parts.append(el("Note", nt)) # items and sub-sections in document order (by first item number) children = [(x["number"], item_xml(x)) for x in items_by_sec.get(s["id"], [])] for c in by_parent.get(s["id"], []): first = min([x["number"] for x in d["items"] if x["section"] == c["id"] or x["section"].startswith(c["id"] + ".")] or [10 ** 6]) children.append((first - 0.5, section_xml(c))) parts += [c for _, c in sorted(children, key=lambda z: z[0])] return f'
' + "".join(parts) + "
" top = [s for s in by_parent.get(None, [])] body = "".join(section_xml(s) for s in top) # ---------------------------------------------------------------- annex un = iter(range(1, 1000)) ev = "".join( (f'' + el("Problem", (e.get("problem") or "").rstrip(":")) + el("Text", e["text"]) + "".join(el("Url", u) for u in e.get("urls", [])) + "".join(f'' for n in e.get("footnoteRefs", [])) + "" for e in d["evidence"]) annex = f'{el("Title", "NAP2027 prioritāšu pamatojuma avoti")}{ev}' notes = "" + "".join(el("Footnote", n["text"], n=n["number"], page=n["page"]) for n in d["footnotes"]) + "" # ---------------------------------------------------------------- head abbr_org = {**{x["label"]: x.get("org") for x in m["actors"] if x["match"] in ("direct", "renamed", "historical")}, **m.get("abbreviations", {})} abbrs = "" + "".join(el("Abbreviation", x["meaning"], term=x["abbr"], org=abbr_org.get(x["abbr"])) for x in d["abbreviations"]) + "" act = "" + "".join( f"" + el("Note", x.get("note")) + "" for lab, x in sorted(used_actors.items(), key=lambda z: (z[1].get("org") or "99", z[0]))) + "" fund = "" + "".join(el("FundingSource", FUNDING_NAMES[c], code=c) for c in FUNDING_NAMES if c in used_funding) + "" sha = hashlib.sha256(open(os.path.join(ROOT, PDF), "rb").read()).hexdigest() meta = ("" + el("Title", "Latvijas Nacionālais attīstības plāns 2021.–2027. gadam") + el("ShortTitle", "NAP2027") + el("DocumentType", "nacionālais attīstības plāns") + '' + "" + el("Body", "Latvijas Republikas Saeima") + el("Act", "Saeimas lēmums") + el("Number", "418/Lm13") + el("Date", "2020-07-02") + "" + el("Developer", "Pārresoru koordinācijas centrs", org="03-9001") + "" + el("Url", SOURCE_URL) + el("File", PDF) + el("SHA256", sha) + el("Pages", 127) + el("Retrieved", RETRIEVED) + "" + "" + el("By", "PPP Asociācija (PPPA), Valsts PirmKods") + el("Method", "Automātiska nolasīšana no PDF (tools/parse_nap.py: vārdi ar koordinātām, tabulu ailes pēc atstarpēm starp ailēm) un " "institūciju sasaiste pēc sources/nap2027/dalibnieki.yaml (tools/build_nap2027.py); pārbaude — tools/validate.py.") + "" + f'' + el("Change", f"Pirmā versija: {len(d['items'])} numurētie punkti [1]–[475], {len(d['evidence'])} pamatojuma punkti, " f"{len(d['footnotes'])} zemsvītras piezīmes, {len(d['abbreviations'])} saīsinājumi.") + el("Change", "Atbildīgās un līdzatbildīgās institūcijas sasaistītas ar VPK ID; drukātais apzīmējums saglabāts.") + el("Change", "Uzdevumiem, kuru jomu pārņēmusi Klimata un enerģētikas ministrija (klimata politika no 2023-01-01, vides aizsardzības politika no 2024-07-01), pie VARAM norādīta pašreizējā institūcija (currentOrg).") + "") xml = ('\n' f'' + meta + abbrs + act + fund + body + annex + notes + "\n") # pretty-print from lxml import etree tree = etree.fromstring(xml.encode("utf-8")) etree.indent(tree, space=" ") os.makedirs(os.path.dirname(out), exist_ok=True) with open(out, "wb") as f: f.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True)) print("wrote", out, "actors", len(used_actors), "funding", len(used_funding), "indicator links", match_stats) want = {(n, c["label"]) for c in m.get("current", []) for n in c["items"]} if want - used_current: unresolved += [f"current: {k}" for k in sorted(want - used_current)] print("current institution set for", len(used_current), "task actors") if unresolved: print("UNRESOLVED", sorted(set(unresolved))) sys.exit(1) if __name__ == "__main__": main(sys.argv[1], sys.argv[2])