1
0

NAP2027 0.1.1: labotas uzdevumu [315]–[318] ailes; pašreizējā institūcija (KEM)

Uzdevumu teksta pirmie vārdi 68. lpp. bija nonākuši ailē „Nr.”; pārbaudei
pievienota apgrieztā vārdu pārbaude. Atribūts currentOrg/currentSince:
septiņiem VARAM vides un klimata uzdevumiem — KEM 20-0000.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
2026-10-11 13:01:14 +00:00
parent bcdd26b8d9
commit bc915bc6d2
10 changed files with 114 additions and 37 deletions

View File

@@ -47,7 +47,7 @@ def main(src, out):
d = json.load(open(src, encoding="utf-8"))
m = yaml.safe_load(open(os.path.join(ROOT, "sources/nap2027/dalibnieki.yaml"), encoding="utf-8"))
actors = {x["label"]: x for x in m["actors"]}
used_actors, used_funding, unresolved = {}, {}, []
used_actors, used_funding, unresolved, used_current = {}, {}, [], set()
# ---------------------------------------------------------------- actors
def actor_list(printed):
@@ -68,11 +68,19 @@ def main(src, out):
out_.append({"label": q, "org": actors[q].get("org")})
return out_
def actors_xml(tag, printed):
cur = {(n, c["label"]): c for c in m.get("current", []) for n in c["items"]}
def actors_xml(tag, printed, n=None):
lst = actor_list(printed)
for x in lst:
c = cur.get((n, x["label"]))
if c:
x["currentOrg"], x["currentSince"] = c["org"], c["since"]
used_current.add((n, x["label"]))
if not lst:
return ""
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}/>" for x in lst)
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}"
f"{a('currentOrg', x.get('currentOrg'))}{a('currentSince', x.get('currentSince'))}/>" for x in lst)
return f"<{tag}{a('printed', printed)}>{inner}</{tag}>"
# ---------------------------------------------------------------- funding
@@ -156,8 +164,8 @@ def main(src, out):
+ el("BaseValue", x.get("baseValue")) + el("Target", x.get("target2024"), year="2024")
+ el("Target", x.get("target2027"), year="2027") + el("DataSource", x.get("source")) + "</Indicator>")
elif x["kind"] == "task":
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"))
+ actors_xml("CoResponsible", x.get("coResponsible")) + funding_xml(x.get("funding"))
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"), x["number"])
+ actors_xml("CoResponsible", x.get("coResponsible"), x["number"]) + funding_xml(x.get("funding"))
+ task_indicators(x) + "</Task>")
else:
title = x.get("title") if x["kind"] == "strategicGoal" else None
@@ -235,6 +243,7 @@ def main(src, out):
+ el("Change", f"Pirmā versija: {len(d['items'])} numurētie punkti [1]–[475], {len(d['evidence'])} pamatojuma punkti, "
f"{len(d['footnotes'])} zemsvītras piezīmes, {len(d['abbreviations'])} saīsinājumi.")
+ el("Change", "Atbildīgās un līdzatbildīgās institūcijas sasaistītas ar VPK ID; drukātais apzīmējums saglabāts.")
+ el("Change", "Uzdevumiem, kuru jomu pārņēmusi Klimata un enerģētikas ministrija (klimata politika no 2023-01-01, vides aizsardzības politika no 2024-07-01), pie VARAM norādīta pašreizējā institūcija (currentOrg).")
+ "</DataVersion></Metadata>")
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
@@ -248,6 +257,10 @@ def main(src, out):
with open(out, "wb") as f:
f.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
print("wrote", out, "actors", len(used_actors), "funding", len(used_funding), "indicator links", match_stats)
want = {(n, c["label"]) for c in m.get("current", []) for n in c["items"]}
if want - used_current:
unresolved += [f"current: {k}" for k in sorted(want - used_current)]
print("current institution set for", len(used_current), "task actors")
if unresolved:
print("UNRESOLVED", sorted(set(unresolved)))
sys.exit(1)

View File

@@ -37,9 +37,9 @@ def load_vocab(pdf_path):
def _better(whole, *parts):
"""Join when the whole word is attested more often than any of its fragments."""
"""Join when the whole word is attested more often than any of its fragments, or once while a fragment never is."""
n = VOCAB.get(whole, 0) + VOCAB.get(whole.lower(), 0)
return n >= 2 and any(VOCAB.get(p, 0) < n for p in parts)
return (n >= 2 and any(VOCAB.get(p, 0) < n for p in parts)) or (n >= 1 and any(VOCAB.get(p, 0) == 0 for p in parts))
def repair(s):
@@ -50,7 +50,7 @@ def repair(s):
if out:
a, core = out[-1], t.rstrip(".,;:)”")
ca = a.lstrip("“(")
if "/" in a or "http" in a or "www." in a:
if "http" in a or "www." in a or ("/" in a and re.search(r"[_.%=?&]|/.*/|-$", a)):
out.append(t)
continue
# three fragments "fundamentāl a s"
@@ -101,7 +101,7 @@ def page_lines(pdf):
# footnotes: small text at the bottom of the page
marks = [w["top"] for w in ws if w["size"] < 6.5 and w["text"].isdigit() and w["x0"] < 90 and w["top"] > 550]
fz = min(marks) - 1.5 if marks else 9999
infoot = lambda w: w["size"] < 9.5 and w["top"] >= fz
infoot = lambda w: w["size"] < 10.5 and w["top"] >= fz
foot = [w for w in ws if infoot(w)]
body = [w for w in ws if not infoot(w) and w["size"] >= 7.5]
smalls = [w for w in ws if not infoot(w) and w["size"] < 7.5]
@@ -242,6 +242,7 @@ def assign(words, centers, names, bounds=None):
for w in words:
cx = (w["x0"] + w["x1"]) / 2
i = sum(1 for b in bounds if cx > b)
i = max(i, 1) # the row marker is not among the words: nothing belongs to the "Nr." column
cells[names[i]].append(w)
return cells

View File

@@ -27,7 +27,8 @@ REG_URL = "https://processgit.org/Valsts-Pirmkods/Valdibas-Deklaracija-as-Code/r
def main():
reg_src = sys.argv[sys.argv.index("--registry") + 1] if "--registry" in sys.argv else REG_URL
reg = etree.parse(reg_src if os.path.exists(reg_src) else urllib.request.urlopen(reg_src))
reg = etree.parse(reg_src if os.path.exists(reg_src) else
urllib.request.urlopen(urllib.request.Request(reg_src, headers={"User-Agent": "strategy-as-code-validate/0.1"})))
ids = {o.get("id") for o in reg.getroot() if etree.QName(o).localname == "Organization"}
xsd = etree.XMLSchema(etree.parse(os.path.join(ROOT, "schemas/strategija-0.1.xsd")))
cat = yaml.safe_load(open(os.path.join(ROOT, "planosanas-dokumenti.yaml"), encoding="utf-8"))
@@ -47,7 +48,7 @@ def main():
sha = hashlib.sha256(open(src, "rb").read()).hexdigest()
if sha != root.findtext("s:Metadata/s:Source/s:SHA256", namespaces=NS):
errors.append(f"{d['id']}: avota sha256 nesakrīt")
used = {e.get("org") for e in root.iter() if e.get("org")}
used = {e.get(k) for e in root.iter() for k in ("org", "currentOrg") if e.get(k)}
missing = sorted(used - ids)
if missing:
errors.append(f"{d['id']}: VPK ID nav reģistrā: {', '.join(missing)}")

View File

@@ -21,6 +21,7 @@ from lxml import etree
NS = {"s": "urn:pppa:vpk:strategija:0.1"}
HEADER = "latvijasnacionālaisattīstībasplānsgadam"
BOILER = set()
SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
@@ -55,6 +56,9 @@ def main(xml_path, pdf_path, out):
lay = subprocess.run(["pdftotext", "-layout", pdf_path, "-"], capture_output=True, text=True, check=True).stdout
flat = letters(raw).replace(HEADER, "")
doc = etree.parse(xml_path)
# words that may stand between table rows in the PDF and belong elsewhere in the file: footnotes, sub-section titles
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Footnote", NS))))
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Section/s:Title", NS))))
items = doc.findall(".//s:Item", NS)
res = {"file": xml_path, "pdf_sha256": hashlib.sha256(open(pdf_path, "rb").read()).hexdigest(), "checks": {}}
@@ -67,14 +71,27 @@ def main(xml_path, pdf_path, out):
# layout blocks between consecutive markers
pos = {int(m.group(1)): m.end() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
start = {int(m.group(1)): m.start() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
order = sorted(pos)
def block(n):
k = order.index(n)
end = pos[order[k + 1]] if k + 1 < len(order) else len(lay)
end = start[order[k + 1]] if k + 1 < len(order) else len(lay)
return lay[pos[n]:end]
text_bad, text_split, row_bad, row_words, text_n, row_n = [], [], [], 0, 0, 0
text_bad, text_split, row_bad, row_extra, row_words, text_n, row_n = [], [], [], [], 0, 0, 0
def block_rows(n):
"""The row's lines only: stop at a blank line followed by body text, drop page header/footer lines."""
out = []
for ln in block(n).split("\n"):
st = ln.strip()
if not st or re.fullmatch(r"\d{1,3}", st) or st.startswith("Latvijas Nacionālais attīstības plāns"):
continue
if re.match(r"^(Rīcības virziena|RĪCĪBAS|PRIORITĀTES|Prioritāte|NAP2027|Nr\.|\*|\d{1,2} [A-ZĀČĒĢĪĶĻŅŠŪŽa-z])", st) or len(ln) - len(ln.lstrip()) == 0 and len(st) > 60:
break
out.append(ln)
return "\n".join(out)
for it in items:
n, kind = int(it.get("n")), it.get("kind")
if kind in ("indicator", "task"):
@@ -97,6 +114,19 @@ def main(xml_path, pdf_path, out):
missing.append(w)
if missing:
row_bad.append({"n": n, "missing_words": missing[:10]})
# reverse: every word printed in the row's part of the PDF is in the file (nothing dropped)
xw = set(words(" ".join(vals)))
xjoin = " ".join(sorted(xw))
extra = []
for w in words(block_rows(n)):
w2 = re.sub(r"(?<=[a-zāčēģīķļņšūž])\d{1,2}$", "", w) # footnote number printed after a word
if w in xw or w2 in xw or w in BOILER or w2 in BOILER or (w.isdigit() or len(w) >= 2) and w in xjoin:
continue
if w.isdigit() and len(w) > 2 and (w[:-1] in xw or w[:-2] in xw): # number followed by a footnote number
continue
extra.append(w)
if extra:
row_extra.append({"n": n, "words_not_in_file": extra[:12]})
else:
text_n += 1
ks = [runs(letters(x.text), flat) for x in it.findall("s:Text", NS) + it.findall("s:Area", NS)]
@@ -107,7 +137,8 @@ def main(xml_path, pdf_path, out):
text_split.append(n)
res["checks"]["text_items"] = {"checked": text_n, "not_found_verbatim": text_bad,
"found_in_2_to_4_pieces": text_split}
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad}
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad,
"rows_with_words_not_in_file": row_extra}
ev = doc.findall(".//s:Evidence", NS)
ev_bad = [e.get("id") for e in ev if (runs(letters((e.findtext("s:Problem", namespaces=NS) or "") + e.findtext("s:Text", namespaces=NS)), flat) or 99) > 4]
@@ -122,7 +153,7 @@ def main(xml_path, pdf_path, out):
"indicators": len(inds), "indicators_without_values": no_val}
c = res["checks"]
res["passed"] = not (c["numbering"]["missing"] or c["numbering"]["extra"] or c["numbering"]["duplicates"] or text_bad
or row_bad or ev_bad or no_resp)
or row_bad or row_extra or ev_bad or no_resp)
with open(out, "w", encoding="utf-8") as f:
json.dump(res, f, ensure_ascii=False, indent=1)
print(json.dumps({k: {kk: (vv if not isinstance(vv, list) else (len(vv) if len(vv) > 12 else vv)) for kk, vv in v.items()}