1
0
Files
Strategy-as-Code/tools/build_nap2027.py
Rihards Gailums bc915bc6d2 NAP2027 0.1.1: labotas uzdevumu [315]–[318] ailes; pašreizējā institūcija (KEM)
Uzdevumu teksta pirmie vārdi 68. lpp. bija nonākuši ailē „Nr.”; pārbaudei
pievienota apgrieztā vārdu pārbaude. Atribūts currentOrg/currentSince:
septiņiem VARAM vides un klimata uzdevumiem — KEM 20-0000.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
2026-10-11 13:01:14 +00:00

271 lines
14 KiB
Python

#!/usr/bin/env python3
"""NAP2027: starpposma JSON (parse_nap.py) + institūciju sasaiste (sources/nap2027/dalibnieki.yaml) → data/nap2027/lv-nap2027.xml
python3 tools/parse_nap.py sources/nap2027/NAP2027.pdf build/nap2027.json
python3 tools/build_nap2027.py build/nap2027.json data/nap2027/lv-nap2027.xml
"""
import datetime as dt
import difflib
import hashlib
import json
import os
import re
import sys
from xml.sax.saxutils import escape, quoteattr
import yaml
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
DOC = "lv-nap2027"
PDF = "sources/nap2027/NAP2027.pdf"
SOURCE_URL = "https://www.mk.gov.lv/lv/media/15162/download"
RETRIEVED = "2026-10-11"
FUNDING_NAMES = {
"VB": "Valsts budžets", "ES-FONDI": "Eiropas Savienības fondi", "CITI": "Citi finanšu avoti", "PASV": "Pašvaldību budžeti",
"HORIZON": "Horizon Europe", "DIGITAL": "Digital Europe", "URBAN": "Urban Europe",
"EEZ-NFI": "EEZ un Norvēģijas finanšu instruments", "CH": "Šveices programma",
"ES-JAUNATNE": "ES programmas jaunatnes jomā", "PILSONISKA": "Pilsoniskā iniciatīva",
}
SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
def sq(s):
return re.sub(r"[^0-9a-zāčēģīķļņšūž%]", "", (s or "").lower().translate(SUBS))
def a(name, value):
return f" {name}={quoteattr(str(value))}" if value not in (None, "") else ""
def el(tag, text, **attrs):
if text in (None, ""):
return ""
return f"<{tag}{''.join(a(k, v) for k, v in attrs.items())}>{escape(str(text))}</{tag}>"
def main(src, out):
d = json.load(open(src, encoding="utf-8"))
m = yaml.safe_load(open(os.path.join(ROOT, "sources/nap2027/dalibnieki.yaml"), encoding="utf-8"))
actors = {x["label"]: x for x in m["actors"]}
used_actors, used_funding, unresolved, used_current = {}, {}, [], set()
# ---------------------------------------------------------------- actors
def actor_list(printed):
if not printed:
return []
parts = [p.strip() for p in re.split(r",\s*", printed) if p.strip()]
out_ = []
for p in parts:
if p in m.get("qualifiers", {}):
if out_ and out_[-1]["label"] == m["qualifiers"][p]:
out_[-1]["qualifier"] = p
continue
for q in m.get("split", {}).get(p, [p]):
if q not in actors:
unresolved.append(q)
continue
used_actors[q] = actors[q]
out_.append({"label": q, "org": actors[q].get("org")})
return out_
cur = {(n, c["label"]): c for c in m.get("current", []) for n in c["items"]}
def actors_xml(tag, printed, n=None):
lst = actor_list(printed)
for x in lst:
c = cur.get((n, x["label"]))
if c:
x["currentOrg"], x["currentSince"] = c["org"], c["since"]
used_current.add((n, x["label"]))
if not lst:
return ""
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}"
f"{a('currentOrg', x.get('currentOrg'))}{a('currentSince', x.get('currentSince'))}/>" for x in lst)
return f"<{tag}{a('printed', printed)}>{inner}</{tag}>"
# ---------------------------------------------------------------- funding
fmap = m["funding"]
def funding_xml(printed):
if not printed:
return ""
srcs = []
for p in [p.strip() for p in re.split(r",\s*", printed) if p.strip()]:
labels = [p] if p in fmap else [x.strip() for x in p.split("/")]
for lab in labels:
if lab not in fmap:
unresolved.append("finansējums: " + lab)
continue
used_funding[fmap[lab]] = True
srcs.append(f"<Source{a('code', fmap[lab])}{a('label', lab)}/>")
return f"<Funding{a('printed', printed)}>{''.join(srcs)}</Funding>" if srcs else ""
# ---------------------------------------------------------------- task indicator → document indicator
inds = [x for x in d["items"] if x["kind"] == "indicator"]
rv_of = lambda sec: ".".join(sec.split(".")[:2])
match_stats = {"exact": 0, "shortened": 0, "similar": 0, "split": 0, "task-level": 0}
def best_indicator(name, sec):
s = sq(name)
if len(s) < 4:
return None, None
same = [i for i in inds if rv_of(i["section"]) == rv_of(sec)]
for pool in (same, inds):
for i in pool:
if sq(i["name"]) == s:
return i, "exact"
for i in pool:
t = sq(i["name"])
head = sq(re.split(r"[(,]", i["name"])[0]) # name without the bracketed or comma explanation
if len(s) >= 10 and t.startswith(s) and (len(s) >= 0.6 * len(t) or head == s):
return i, "shortened" # printed without the bracketed explanation, or slightly shortened
r, i = max(((difflib.SequenceMatcher(None, s, sq(i["name"])).ratio() + (0.02 if i in same else 0), i) for i in inds),
key=lambda z: z[0])
return (i, "similar") if r >= 0.85 else (None, None)
def split_merged(name, sec):
"""Several indicator names printed without an empty line between them: split before capitalised words
(not all-caps abbreviations); keep matched pieces separate, join unmatched neighbours back together."""
w = name.split(" ")
cuts = [k for k in range(1, len(w)) if w[k][:1].isupper() and not w[k].isupper() and not w[k - 1].endswith(("(", "–", "-", "/"))]
if not cuts:
return None
segs = [" ".join(w[i:j]) for i, j in zip([0] + cuts, cuts + [len(w)])]
res = []
for sgm in segs:
i, how = best_indicator(sgm, sec)
if i and how in ("exact", "shortened"):
res.append((sgm, i, how))
elif res and res[-1][1] is None:
res[-1] = (res[-1][0] + " " + sgm, None, None)
else:
res.append((sgm, None, None))
return res if any(r[1] for r in res) and len(res) > 1 else None
def task_indicators(t):
xs = []
for nm in t.get("indicatorNames") or []:
i, how = best_indicator(nm, t["section"])
parts = [(nm, i, how)] if i else (split_merged(nm, t["section"]) or [(nm, None, None)])
if len(parts) > 1:
match_stats["split"] += 1
for name, i, how in parts:
match_stats[how or "task-level"] += 1
ref = f"<IndicatorRef{a('ref', DOC + '.i%03d' % i['number'])}{a('match', how)}/>" if i else ""
xs.append(f"<TaskIndicator>{el('Name', name)}{ref}</TaskIndicator>")
return "".join(xs)
# ---------------------------------------------------------------- items
def item_xml(x):
iid = f"{DOC}.i{x['number']:03d}"
body = ""
if x["kind"] == "indicator":
body = ("<Indicator>" + el("Name", x.get("name")) + el("Unit", x.get("unit")) + el("BaseYear", x.get("baseYear"))
+ el("BaseValue", x.get("baseValue")) + el("Target", x.get("target2024"), year="2024")
+ el("Target", x.get("target2027"), year="2027") + el("DataSource", x.get("source")) + "</Indicator>")
elif x["kind"] == "task":
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"), x["number"])
+ actors_xml("CoResponsible", x.get("coResponsible"), x["number"]) + funding_xml(x.get("funding"))
+ task_indicators(x) + "</Task>")
else:
title = x.get("title") if x["kind"] == "strategicGoal" else None
body = el("Title", title) + el("Area", x.get("area")) + el("Text", x["text"])
refs = "".join(f'<FootnoteRef n="{n}"/>' for n in x.get("footnoteRefs", []))
return f'<Item id="{iid}" n="{x["number"]}" kind="{x["kind"]}" page="{x["page"]}">{body}{refs}</Item>'
# ---------------------------------------------------------------- sections (tree)
secs = [s for s in d["sections"] if s["kind"] not in ("front", "annex")]
by_parent = {}
for s in secs:
by_parent.setdefault(s.get("parent"), []).append(s)
items_by_sec = {}
for x in d["items"]:
items_by_sec.setdefault(x["section"], []).append(x)
def sec_num(s):
mm = re.search(r"(\d+)$", s["id"])
return mm.group(1) if s["kind"] in ("priority", "actionLine", "theme") and mm else None
def section_xml(s):
sid = f"{DOC}.{s['id']}"
parts = [el("Title", s["title"])]
f = s.get("funding")
if f and f.get("millionEur"):
parts.append(el("IndicativeFunding", f["text"], millionEur=f["millionEur"].replace(",", ".")))
for nt in s.get("notes", []):
if isinstance(nt, dict):
parts.append(el("Note", nt["text"], area=nt.get("area")))
else:
parts.append(el("Note", nt))
# items and sub-sections in document order (by first item number)
children = [(x["number"], item_xml(x)) for x in items_by_sec.get(s["id"], [])]
for c in by_parent.get(s["id"], []):
first = min([x["number"] for x in d["items"] if x["section"] == c["id"] or x["section"].startswith(c["id"] + ".")] or [10 ** 6])
children.append((first - 0.5, section_xml(c)))
parts += [c for _, c in sorted(children, key=lambda z: z[0])]
return f'<Section id="{sid}" kind="{s["kind"]}"{a("n", sec_num(s))}>' + "".join(parts) + "</Section>"
top = [s for s in by_parent.get(None, [])]
body = "".join(section_xml(s) for s in top)
# ---------------------------------------------------------------- annex
un = iter(range(1, 1000))
ev = "".join(
(f'<Evidence id="{DOC}.a{e["number"]:03d}" n="{e["number"]}"' if e["number"] else f'<Evidence id="{DOC}.annex.u{next(un)}"')
+ f' section="{DOC}.{e["section"]}" page="{e["page"]}">'
+ el("Problem", (e.get("problem") or "").rstrip(":")) + el("Text", e["text"]) + "".join(el("Url", u) for u in e.get("urls", []))
+ "".join(f'<FootnoteRef n="{n}"/>' for n in e.get("footnoteRefs", [])) + "</Evidence>"
for e in d["evidence"])
annex = f'<Annex id="{DOC}.annex">{el("Title", "NAP2027 prioritāšu pamatojuma avoti")}{ev}</Annex>'
notes = "<Footnotes>" + "".join(el("Footnote", n["text"], n=n["number"], page=n["page"]) for n in d["footnotes"]) + "</Footnotes>"
# ---------------------------------------------------------------- head
abbr_org = {**{x["label"]: x.get("org") for x in m["actors"] if x["match"] in ("direct", "renamed", "historical")},
**m.get("abbreviations", {})}
abbrs = "<Abbreviations>" + "".join(el("Abbreviation", x["meaning"], term=x["abbr"], org=abbr_org.get(x["abbr"]))
for x in d["abbreviations"]) + "</Abbreviations>"
act = "<Actors>" + "".join(
f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('match', x['match'])}>" + el("Note", x.get("note")) + "</Actor>"
for lab, x in sorted(used_actors.items(), key=lambda z: (z[1].get("org") or "99", z[0]))) + "</Actors>"
fund = "<FundingSources>" + "".join(el("FundingSource", FUNDING_NAMES[c], code=c) for c in FUNDING_NAMES if c in used_funding) + "</FundingSources>"
sha = hashlib.sha256(open(os.path.join(ROOT, PDF), "rb").read()).hexdigest()
meta = ("<Metadata>"
+ el("Title", "Latvijas Nacionālais attīstības plāns 2021.–2027. gadam") + el("ShortTitle", "NAP2027")
+ el("DocumentType", "nacionālais attīstības plāns") + '<Period from="2021" to="2027"/>'
+ "<Approval>" + el("Body", "Latvijas Republikas Saeima") + el("Act", "Saeimas lēmums") + el("Number", "418/Lm13")
+ el("Date", "2020-07-02") + "</Approval>"
+ el("Developer", "Pārresoru koordinācijas centrs", org="03-9001")
+ "<Source>" + el("Url", SOURCE_URL) + el("File", PDF) + el("SHA256", sha) + el("Pages", 127) + el("Retrieved", RETRIEVED) + "</Source>"
+ "<Conversion>" + el("By", "PPP Asociācija (PPPA), Valsts PirmKods") + el("Method",
"Automātiska nolasīšana no PDF (tools/parse_nap.py: vārdi ar koordinātām, tabulu ailes pēc atstarpēm starp ailēm) un "
"institūciju sasaiste pēc sources/nap2027/dalibnieki.yaml (tools/build_nap2027.py); pārbaude — tools/validate.py.") + "</Conversion>"
+ f'<DataVersion number="1" date="{dt.date.today().isoformat()}">'
+ el("Change", f"Pirmā versija: {len(d['items'])} numurētie punkti [1]–[475], {len(d['evidence'])} pamatojuma punkti, "
f"{len(d['footnotes'])} zemsvītras piezīmes, {len(d['abbreviations'])} saīsinājumi.")
+ el("Change", "Atbildīgās un līdzatbildīgās institūcijas sasaistītas ar VPK ID; drukātais apzīmējums saglabāts.")
+ el("Change", "Uzdevumiem, kuru jomu pārņēmusi Klimata un enerģētikas ministrija (klimata politika no 2023-01-01, vides aizsardzības politika no 2024-07-01), pie VARAM norādīta pašreizējā institūcija (currentOrg).")
+ "</DataVersion></Metadata>")
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
f'<PlanningDocument xmlns="urn:pppa:vpk:strategija:0.1" schemaVersion="0.1" id="{DOC}">'
+ meta + abbrs + act + fund + body + annex + notes + "</PlanningDocument>\n")
# pretty-print
from lxml import etree
tree = etree.fromstring(xml.encode("utf-8"))
etree.indent(tree, space=" ")
os.makedirs(os.path.dirname(out), exist_ok=True)
with open(out, "wb") as f:
f.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
print("wrote", out, "actors", len(used_actors), "funding", len(used_funding), "indicator links", match_stats)
want = {(n, c["label"]) for c in m.get("current", []) for n in c["items"]}
if want - used_current:
unresolved += [f"current: {k}" for k in sorted(want - used_current)]
print("current institution set for", len(used_current), "task actors")
if unresolved:
print("UNRESOLVED", sorted(set(unresolved)))
sys.exit(1)
if __name__ == "__main__":
main(sys.argv[1], sys.argv[2])