475 numurētie punkti (mērķi, 131 indikators, 124 uzdevumi ar VPK ID, telpiskās attīstības virzieni), 154 pamatojuma ieraksti, 34 zemsvītras piezīmes. Pārbaude pret avota PDF ar citu nolasītāju — izturēta. Katalogs planosanas-dokumenti.yaml, rīki parse/build/verify/validate. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
258 lines
13 KiB
Python
258 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""NAP2027: starpposma JSON (parse_nap.py) + institūciju sasaiste (sources/nap2027/dalibnieki.yaml) → data/nap2027/lv-nap2027.xml
|
|
|
|
python3 tools/parse_nap.py sources/nap2027/NAP2027.pdf build/nap2027.json
|
|
python3 tools/build_nap2027.py build/nap2027.json data/nap2027/lv-nap2027.xml
|
|
"""
|
|
import datetime as dt
|
|
import difflib
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
from xml.sax.saxutils import escape, quoteattr
|
|
|
|
import yaml
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
DOC = "lv-nap2027"
|
|
PDF = "sources/nap2027/NAP2027.pdf"
|
|
SOURCE_URL = "https://www.mk.gov.lv/lv/media/15162/download"
|
|
RETRIEVED = "2026-10-11"
|
|
FUNDING_NAMES = {
|
|
"VB": "Valsts budžets", "ES-FONDI": "Eiropas Savienības fondi", "CITI": "Citi finanšu avoti", "PASV": "Pašvaldību budžeti",
|
|
"HORIZON": "Horizon Europe", "DIGITAL": "Digital Europe", "URBAN": "Urban Europe",
|
|
"EEZ-NFI": "EEZ un Norvēģijas finanšu instruments", "CH": "Šveices programma",
|
|
"ES-JAUNATNE": "ES programmas jaunatnes jomā", "PILSONISKA": "Pilsoniskā iniciatīva",
|
|
}
|
|
SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
|
|
|
|
|
|
def sq(s):
|
|
return re.sub(r"[^0-9a-zāčēģīķļņšūž%]", "", (s or "").lower().translate(SUBS))
|
|
|
|
|
|
def a(name, value):
|
|
return f" {name}={quoteattr(str(value))}" if value not in (None, "") else ""
|
|
|
|
|
|
def el(tag, text, **attrs):
|
|
if text in (None, ""):
|
|
return ""
|
|
return f"<{tag}{''.join(a(k, v) for k, v in attrs.items())}>{escape(str(text))}</{tag}>"
|
|
|
|
|
|
def main(src, out):
|
|
d = json.load(open(src, encoding="utf-8"))
|
|
m = yaml.safe_load(open(os.path.join(ROOT, "sources/nap2027/dalibnieki.yaml"), encoding="utf-8"))
|
|
actors = {x["label"]: x for x in m["actors"]}
|
|
used_actors, used_funding, unresolved = {}, {}, []
|
|
|
|
# ---------------------------------------------------------------- actors
|
|
def actor_list(printed):
|
|
if not printed:
|
|
return []
|
|
parts = [p.strip() for p in re.split(r",\s*", printed) if p.strip()]
|
|
out_ = []
|
|
for p in parts:
|
|
if p in m.get("qualifiers", {}):
|
|
if out_ and out_[-1]["label"] == m["qualifiers"][p]:
|
|
out_[-1]["qualifier"] = p
|
|
continue
|
|
for q in m.get("split", {}).get(p, [p]):
|
|
if q not in actors:
|
|
unresolved.append(q)
|
|
continue
|
|
used_actors[q] = actors[q]
|
|
out_.append({"label": q, "org": actors[q].get("org")})
|
|
return out_
|
|
|
|
def actors_xml(tag, printed):
|
|
lst = actor_list(printed)
|
|
if not lst:
|
|
return ""
|
|
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}/>" for x in lst)
|
|
return f"<{tag}{a('printed', printed)}>{inner}</{tag}>"
|
|
|
|
# ---------------------------------------------------------------- funding
|
|
fmap = m["funding"]
|
|
|
|
def funding_xml(printed):
|
|
if not printed:
|
|
return ""
|
|
srcs = []
|
|
for p in [p.strip() for p in re.split(r",\s*", printed) if p.strip()]:
|
|
labels = [p] if p in fmap else [x.strip() for x in p.split("/")]
|
|
for lab in labels:
|
|
if lab not in fmap:
|
|
unresolved.append("finansējums: " + lab)
|
|
continue
|
|
used_funding[fmap[lab]] = True
|
|
srcs.append(f"<Source{a('code', fmap[lab])}{a('label', lab)}/>")
|
|
return f"<Funding{a('printed', printed)}>{''.join(srcs)}</Funding>" if srcs else ""
|
|
|
|
# ---------------------------------------------------------------- task indicator → document indicator
|
|
inds = [x for x in d["items"] if x["kind"] == "indicator"]
|
|
rv_of = lambda sec: ".".join(sec.split(".")[:2])
|
|
match_stats = {"exact": 0, "shortened": 0, "similar": 0, "split": 0, "task-level": 0}
|
|
|
|
def best_indicator(name, sec):
|
|
s = sq(name)
|
|
if len(s) < 4:
|
|
return None, None
|
|
same = [i for i in inds if rv_of(i["section"]) == rv_of(sec)]
|
|
for pool in (same, inds):
|
|
for i in pool:
|
|
if sq(i["name"]) == s:
|
|
return i, "exact"
|
|
for i in pool:
|
|
t = sq(i["name"])
|
|
head = sq(re.split(r"[(,]", i["name"])[0]) # name without the bracketed or comma explanation
|
|
if len(s) >= 10 and t.startswith(s) and (len(s) >= 0.6 * len(t) or head == s):
|
|
return i, "shortened" # printed without the bracketed explanation, or slightly shortened
|
|
r, i = max(((difflib.SequenceMatcher(None, s, sq(i["name"])).ratio() + (0.02 if i in same else 0), i) for i in inds),
|
|
key=lambda z: z[0])
|
|
return (i, "similar") if r >= 0.85 else (None, None)
|
|
|
|
def split_merged(name, sec):
|
|
"""Several indicator names printed without an empty line between them: split before capitalised words
|
|
(not all-caps abbreviations); keep matched pieces separate, join unmatched neighbours back together."""
|
|
w = name.split(" ")
|
|
cuts = [k for k in range(1, len(w)) if w[k][:1].isupper() and not w[k].isupper() and not w[k - 1].endswith(("(", "–", "-", "/"))]
|
|
if not cuts:
|
|
return None
|
|
segs = [" ".join(w[i:j]) for i, j in zip([0] + cuts, cuts + [len(w)])]
|
|
res = []
|
|
for sgm in segs:
|
|
i, how = best_indicator(sgm, sec)
|
|
if i and how in ("exact", "shortened"):
|
|
res.append((sgm, i, how))
|
|
elif res and res[-1][1] is None:
|
|
res[-1] = (res[-1][0] + " " + sgm, None, None)
|
|
else:
|
|
res.append((sgm, None, None))
|
|
return res if any(r[1] for r in res) and len(res) > 1 else None
|
|
|
|
def task_indicators(t):
|
|
xs = []
|
|
for nm in t.get("indicatorNames") or []:
|
|
i, how = best_indicator(nm, t["section"])
|
|
parts = [(nm, i, how)] if i else (split_merged(nm, t["section"]) or [(nm, None, None)])
|
|
if len(parts) > 1:
|
|
match_stats["split"] += 1
|
|
for name, i, how in parts:
|
|
match_stats[how or "task-level"] += 1
|
|
ref = f"<IndicatorRef{a('ref', DOC + '.i%03d' % i['number'])}{a('match', how)}/>" if i else ""
|
|
xs.append(f"<TaskIndicator>{el('Name', name)}{ref}</TaskIndicator>")
|
|
return "".join(xs)
|
|
|
|
# ---------------------------------------------------------------- items
|
|
def item_xml(x):
|
|
iid = f"{DOC}.i{x['number']:03d}"
|
|
body = ""
|
|
if x["kind"] == "indicator":
|
|
body = ("<Indicator>" + el("Name", x.get("name")) + el("Unit", x.get("unit")) + el("BaseYear", x.get("baseYear"))
|
|
+ el("BaseValue", x.get("baseValue")) + el("Target", x.get("target2024"), year="2024")
|
|
+ el("Target", x.get("target2027"), year="2027") + el("DataSource", x.get("source")) + "</Indicator>")
|
|
elif x["kind"] == "task":
|
|
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"))
|
|
+ actors_xml("CoResponsible", x.get("coResponsible")) + funding_xml(x.get("funding"))
|
|
+ task_indicators(x) + "</Task>")
|
|
else:
|
|
title = x.get("title") if x["kind"] == "strategicGoal" else None
|
|
body = el("Title", title) + el("Area", x.get("area")) + el("Text", x["text"])
|
|
refs = "".join(f'<FootnoteRef n="{n}"/>' for n in x.get("footnoteRefs", []))
|
|
return f'<Item id="{iid}" n="{x["number"]}" kind="{x["kind"]}" page="{x["page"]}">{body}{refs}</Item>'
|
|
|
|
# ---------------------------------------------------------------- sections (tree)
|
|
secs = [s for s in d["sections"] if s["kind"] not in ("front", "annex")]
|
|
by_parent = {}
|
|
for s in secs:
|
|
by_parent.setdefault(s.get("parent"), []).append(s)
|
|
items_by_sec = {}
|
|
for x in d["items"]:
|
|
items_by_sec.setdefault(x["section"], []).append(x)
|
|
|
|
def sec_num(s):
|
|
mm = re.search(r"(\d+)$", s["id"])
|
|
return mm.group(1) if s["kind"] in ("priority", "actionLine", "theme") and mm else None
|
|
|
|
def section_xml(s):
|
|
sid = f"{DOC}.{s['id']}"
|
|
parts = [el("Title", s["title"])]
|
|
f = s.get("funding")
|
|
if f and f.get("millionEur"):
|
|
parts.append(el("IndicativeFunding", f["text"], millionEur=f["millionEur"].replace(",", ".")))
|
|
for nt in s.get("notes", []):
|
|
if isinstance(nt, dict):
|
|
parts.append(el("Note", nt["text"], area=nt.get("area")))
|
|
else:
|
|
parts.append(el("Note", nt))
|
|
# items and sub-sections in document order (by first item number)
|
|
children = [(x["number"], item_xml(x)) for x in items_by_sec.get(s["id"], [])]
|
|
for c in by_parent.get(s["id"], []):
|
|
first = min([x["number"] for x in d["items"] if x["section"] == c["id"] or x["section"].startswith(c["id"] + ".")] or [10 ** 6])
|
|
children.append((first - 0.5, section_xml(c)))
|
|
parts += [c for _, c in sorted(children, key=lambda z: z[0])]
|
|
return f'<Section id="{sid}" kind="{s["kind"]}"{a("n", sec_num(s))}>' + "".join(parts) + "</Section>"
|
|
|
|
top = [s for s in by_parent.get(None, [])]
|
|
body = "".join(section_xml(s) for s in top)
|
|
|
|
# ---------------------------------------------------------------- annex
|
|
un = iter(range(1, 1000))
|
|
ev = "".join(
|
|
(f'<Evidence id="{DOC}.a{e["number"]:03d}" n="{e["number"]}"' if e["number"] else f'<Evidence id="{DOC}.annex.u{next(un)}"')
|
|
+ f' section="{DOC}.{e["section"]}" page="{e["page"]}">'
|
|
+ el("Problem", (e.get("problem") or "").rstrip(":")) + el("Text", e["text"]) + "".join(el("Url", u) for u in e.get("urls", []))
|
|
+ "".join(f'<FootnoteRef n="{n}"/>' for n in e.get("footnoteRefs", [])) + "</Evidence>"
|
|
for e in d["evidence"])
|
|
annex = f'<Annex id="{DOC}.annex">{el("Title", "NAP2027 prioritāšu pamatojuma avoti")}{ev}</Annex>'
|
|
notes = "<Footnotes>" + "".join(el("Footnote", n["text"], n=n["number"], page=n["page"]) for n in d["footnotes"]) + "</Footnotes>"
|
|
|
|
# ---------------------------------------------------------------- head
|
|
abbr_org = {**{x["label"]: x.get("org") for x in m["actors"] if x["match"] in ("direct", "renamed", "historical")},
|
|
**m.get("abbreviations", {})}
|
|
abbrs = "<Abbreviations>" + "".join(el("Abbreviation", x["meaning"], term=x["abbr"], org=abbr_org.get(x["abbr"]))
|
|
for x in d["abbreviations"]) + "</Abbreviations>"
|
|
act = "<Actors>" + "".join(
|
|
f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('match', x['match'])}>" + el("Note", x.get("note")) + "</Actor>"
|
|
for lab, x in sorted(used_actors.items(), key=lambda z: (z[1].get("org") or "99", z[0]))) + "</Actors>"
|
|
fund = "<FundingSources>" + "".join(el("FundingSource", FUNDING_NAMES[c], code=c) for c in FUNDING_NAMES if c in used_funding) + "</FundingSources>"
|
|
sha = hashlib.sha256(open(os.path.join(ROOT, PDF), "rb").read()).hexdigest()
|
|
meta = ("<Metadata>"
|
|
+ el("Title", "Latvijas Nacionālais attīstības plāns 2021.–2027. gadam") + el("ShortTitle", "NAP2027")
|
|
+ el("DocumentType", "nacionālais attīstības plāns") + '<Period from="2021" to="2027"/>'
|
|
+ "<Approval>" + el("Body", "Latvijas Republikas Saeima") + el("Act", "Saeimas lēmums") + el("Number", "418/Lm13")
|
|
+ el("Date", "2020-07-02") + "</Approval>"
|
|
+ el("Developer", "Pārresoru koordinācijas centrs", org="03-9001")
|
|
+ "<Source>" + el("Url", SOURCE_URL) + el("File", PDF) + el("SHA256", sha) + el("Pages", 127) + el("Retrieved", RETRIEVED) + "</Source>"
|
|
+ "<Conversion>" + el("By", "PPP Asociācija (PPPA), Valsts PirmKods") + el("Method",
|
|
"Automātiska nolasīšana no PDF (tools/parse_nap.py: vārdi ar koordinātām, tabulu ailes pēc atstarpēm starp ailēm) un "
|
|
"institūciju sasaiste pēc sources/nap2027/dalibnieki.yaml (tools/build_nap2027.py); pārbaude — tools/validate.py.") + "</Conversion>"
|
|
+ f'<DataVersion number="1" date="{dt.date.today().isoformat()}">'
|
|
+ el("Change", f"Pirmā versija: {len(d['items'])} numurētie punkti [1]–[475], {len(d['evidence'])} pamatojuma punkti, "
|
|
f"{len(d['footnotes'])} zemsvītras piezīmes, {len(d['abbreviations'])} saīsinājumi.")
|
|
+ el("Change", "Atbildīgās un līdzatbildīgās institūcijas sasaistītas ar VPK ID; drukātais apzīmējums saglabāts.")
|
|
+ "</DataVersion></Metadata>")
|
|
|
|
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
f'<PlanningDocument xmlns="urn:pppa:vpk:strategija:0.1" schemaVersion="0.1" id="{DOC}">'
|
|
+ meta + abbrs + act + fund + body + annex + notes + "</PlanningDocument>\n")
|
|
# pretty-print
|
|
from lxml import etree
|
|
tree = etree.fromstring(xml.encode("utf-8"))
|
|
etree.indent(tree, space=" ")
|
|
os.makedirs(os.path.dirname(out), exist_ok=True)
|
|
with open(out, "wb") as f:
|
|
f.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
|
|
print("wrote", out, "actors", len(used_actors), "funding", len(used_funding), "indicator links", match_stats)
|
|
if unresolved:
|
|
print("UNRESOLVED", sorted(set(unresolved)))
|
|
sys.exit(1)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main(sys.argv[1], sys.argv[2])
|