NAP2027 0.1.1: labotas uzdevumu [315]–[318] ailes; pašreizējā institūcija (KEM)
Uzdevumu teksta pirmie vārdi 68. lpp. bija nonākuši ailē „Nr.”; pārbaudei pievienota apgrieztā vārdu pārbaude. Atribūts currentOrg/currentSince: septiņiem VARAM vides un klimata uzdevumiem — KEM 20-0000. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
@@ -47,7 +47,7 @@ def main(src, out):
|
||||
d = json.load(open(src, encoding="utf-8"))
|
||||
m = yaml.safe_load(open(os.path.join(ROOT, "sources/nap2027/dalibnieki.yaml"), encoding="utf-8"))
|
||||
actors = {x["label"]: x for x in m["actors"]}
|
||||
used_actors, used_funding, unresolved = {}, {}, []
|
||||
used_actors, used_funding, unresolved, used_current = {}, {}, [], set()
|
||||
|
||||
# ---------------------------------------------------------------- actors
|
||||
def actor_list(printed):
|
||||
@@ -68,11 +68,19 @@ def main(src, out):
|
||||
out_.append({"label": q, "org": actors[q].get("org")})
|
||||
return out_
|
||||
|
||||
def actors_xml(tag, printed):
|
||||
cur = {(n, c["label"]): c for c in m.get("current", []) for n in c["items"]}
|
||||
|
||||
def actors_xml(tag, printed, n=None):
|
||||
lst = actor_list(printed)
|
||||
for x in lst:
|
||||
c = cur.get((n, x["label"]))
|
||||
if c:
|
||||
x["currentOrg"], x["currentSince"] = c["org"], c["since"]
|
||||
used_current.add((n, x["label"]))
|
||||
if not lst:
|
||||
return ""
|
||||
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}/>" for x in lst)
|
||||
inner = "".join(f"<Actor{a('label', x['label'])}{a('org', x.get('org'))}{a('qualifier', x.get('qualifier'))}"
|
||||
f"{a('currentOrg', x.get('currentOrg'))}{a('currentSince', x.get('currentSince'))}/>" for x in lst)
|
||||
return f"<{tag}{a('printed', printed)}>{inner}</{tag}>"
|
||||
|
||||
# ---------------------------------------------------------------- funding
|
||||
@@ -156,8 +164,8 @@ def main(src, out):
|
||||
+ el("BaseValue", x.get("baseValue")) + el("Target", x.get("target2024"), year="2024")
|
||||
+ el("Target", x.get("target2027"), year="2027") + el("DataSource", x.get("source")) + "</Indicator>")
|
||||
elif x["kind"] == "task":
|
||||
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"))
|
||||
+ actors_xml("CoResponsible", x.get("coResponsible")) + funding_xml(x.get("funding"))
|
||||
body = ("<Task>" + el("Text", x["text"]) + actors_xml("Responsible", x.get("responsible"), x["number"])
|
||||
+ actors_xml("CoResponsible", x.get("coResponsible"), x["number"]) + funding_xml(x.get("funding"))
|
||||
+ task_indicators(x) + "</Task>")
|
||||
else:
|
||||
title = x.get("title") if x["kind"] == "strategicGoal" else None
|
||||
@@ -235,6 +243,7 @@ def main(src, out):
|
||||
+ el("Change", f"Pirmā versija: {len(d['items'])} numurētie punkti [1]–[475], {len(d['evidence'])} pamatojuma punkti, "
|
||||
f"{len(d['footnotes'])} zemsvītras piezīmes, {len(d['abbreviations'])} saīsinājumi.")
|
||||
+ el("Change", "Atbildīgās un līdzatbildīgās institūcijas sasaistītas ar VPK ID; drukātais apzīmējums saglabāts.")
|
||||
+ el("Change", "Uzdevumiem, kuru jomu pārņēmusi Klimata un enerģētikas ministrija (klimata politika no 2023-01-01, vides aizsardzības politika no 2024-07-01), pie VARAM norādīta pašreizējā institūcija (currentOrg).")
|
||||
+ "</DataVersion></Metadata>")
|
||||
|
||||
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
|
||||
@@ -248,6 +257,10 @@ def main(src, out):
|
||||
with open(out, "wb") as f:
|
||||
f.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
|
||||
print("wrote", out, "actors", len(used_actors), "funding", len(used_funding), "indicator links", match_stats)
|
||||
want = {(n, c["label"]) for c in m.get("current", []) for n in c["items"]}
|
||||
if want - used_current:
|
||||
unresolved += [f"current: {k}" for k in sorted(want - used_current)]
|
||||
print("current institution set for", len(used_current), "task actors")
|
||||
if unresolved:
|
||||
print("UNRESOLVED", sorted(set(unresolved)))
|
||||
sys.exit(1)
|
||||
|
||||
@@ -37,9 +37,9 @@ def load_vocab(pdf_path):
|
||||
|
||||
|
||||
def _better(whole, *parts):
|
||||
"""Join when the whole word is attested more often than any of its fragments."""
|
||||
"""Join when the whole word is attested more often than any of its fragments, or once while a fragment never is."""
|
||||
n = VOCAB.get(whole, 0) + VOCAB.get(whole.lower(), 0)
|
||||
return n >= 2 and any(VOCAB.get(p, 0) < n for p in parts)
|
||||
return (n >= 2 and any(VOCAB.get(p, 0) < n for p in parts)) or (n >= 1 and any(VOCAB.get(p, 0) == 0 for p in parts))
|
||||
|
||||
|
||||
def repair(s):
|
||||
@@ -50,7 +50,7 @@ def repair(s):
|
||||
if out:
|
||||
a, core = out[-1], t.rstrip(".,;:)”")
|
||||
ca = a.lstrip("“(")
|
||||
if "/" in a or "http" in a or "www." in a:
|
||||
if "http" in a or "www." in a or ("/" in a and re.search(r"[_.%=?&]|/.*/|-$", a)):
|
||||
out.append(t)
|
||||
continue
|
||||
# three fragments "fundamentāl a s"
|
||||
@@ -101,7 +101,7 @@ def page_lines(pdf):
|
||||
# footnotes: small text at the bottom of the page
|
||||
marks = [w["top"] for w in ws if w["size"] < 6.5 and w["text"].isdigit() and w["x0"] < 90 and w["top"] > 550]
|
||||
fz = min(marks) - 1.5 if marks else 9999
|
||||
infoot = lambda w: w["size"] < 9.5 and w["top"] >= fz
|
||||
infoot = lambda w: w["size"] < 10.5 and w["top"] >= fz
|
||||
foot = [w for w in ws if infoot(w)]
|
||||
body = [w for w in ws if not infoot(w) and w["size"] >= 7.5]
|
||||
smalls = [w for w in ws if not infoot(w) and w["size"] < 7.5]
|
||||
@@ -242,6 +242,7 @@ def assign(words, centers, names, bounds=None):
|
||||
for w in words:
|
||||
cx = (w["x0"] + w["x1"]) / 2
|
||||
i = sum(1 for b in bounds if cx > b)
|
||||
i = max(i, 1) # the row marker is not among the words: nothing belongs to the "Nr." column
|
||||
cells[names[i]].append(w)
|
||||
return cells
|
||||
|
||||
|
||||
@@ -27,7 +27,8 @@ REG_URL = "https://processgit.org/Valsts-Pirmkods/Valdibas-Deklaracija-as-Code/r
|
||||
|
||||
def main():
|
||||
reg_src = sys.argv[sys.argv.index("--registry") + 1] if "--registry" in sys.argv else REG_URL
|
||||
reg = etree.parse(reg_src if os.path.exists(reg_src) else urllib.request.urlopen(reg_src))
|
||||
reg = etree.parse(reg_src if os.path.exists(reg_src) else
|
||||
urllib.request.urlopen(urllib.request.Request(reg_src, headers={"User-Agent": "strategy-as-code-validate/0.1"})))
|
||||
ids = {o.get("id") for o in reg.getroot() if etree.QName(o).localname == "Organization"}
|
||||
xsd = etree.XMLSchema(etree.parse(os.path.join(ROOT, "schemas/strategija-0.1.xsd")))
|
||||
cat = yaml.safe_load(open(os.path.join(ROOT, "planosanas-dokumenti.yaml"), encoding="utf-8"))
|
||||
@@ -47,7 +48,7 @@ def main():
|
||||
sha = hashlib.sha256(open(src, "rb").read()).hexdigest()
|
||||
if sha != root.findtext("s:Metadata/s:Source/s:SHA256", namespaces=NS):
|
||||
errors.append(f"{d['id']}: avota sha256 nesakrīt")
|
||||
used = {e.get("org") for e in root.iter() if e.get("org")}
|
||||
used = {e.get(k) for e in root.iter() for k in ("org", "currentOrg") if e.get(k)}
|
||||
missing = sorted(used - ids)
|
||||
if missing:
|
||||
errors.append(f"{d['id']}: VPK ID nav reģistrā: {', '.join(missing)}")
|
||||
|
||||
@@ -21,6 +21,7 @@ from lxml import etree
|
||||
|
||||
NS = {"s": "urn:pppa:vpk:strategija:0.1"}
|
||||
HEADER = "latvijasnacionālaisattīstībasplānsgadam"
|
||||
BOILER = set()
|
||||
SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
|
||||
|
||||
|
||||
@@ -55,6 +56,9 @@ def main(xml_path, pdf_path, out):
|
||||
lay = subprocess.run(["pdftotext", "-layout", pdf_path, "-"], capture_output=True, text=True, check=True).stdout
|
||||
flat = letters(raw).replace(HEADER, "")
|
||||
doc = etree.parse(xml_path)
|
||||
# words that may stand between table rows in the PDF and belong elsewhere in the file: footnotes, sub-section titles
|
||||
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Footnote", NS))))
|
||||
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Section/s:Title", NS))))
|
||||
items = doc.findall(".//s:Item", NS)
|
||||
res = {"file": xml_path, "pdf_sha256": hashlib.sha256(open(pdf_path, "rb").read()).hexdigest(), "checks": {}}
|
||||
|
||||
@@ -67,14 +71,27 @@ def main(xml_path, pdf_path, out):
|
||||
|
||||
# layout blocks between consecutive markers
|
||||
pos = {int(m.group(1)): m.end() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
|
||||
start = {int(m.group(1)): m.start() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
|
||||
order = sorted(pos)
|
||||
|
||||
def block(n):
|
||||
k = order.index(n)
|
||||
end = pos[order[k + 1]] if k + 1 < len(order) else len(lay)
|
||||
end = start[order[k + 1]] if k + 1 < len(order) else len(lay)
|
||||
return lay[pos[n]:end]
|
||||
|
||||
text_bad, text_split, row_bad, row_words, text_n, row_n = [], [], [], 0, 0, 0
|
||||
text_bad, text_split, row_bad, row_extra, row_words, text_n, row_n = [], [], [], [], 0, 0, 0
|
||||
|
||||
def block_rows(n):
|
||||
"""The row's lines only: stop at a blank line followed by body text, drop page header/footer lines."""
|
||||
out = []
|
||||
for ln in block(n).split("\n"):
|
||||
st = ln.strip()
|
||||
if not st or re.fullmatch(r"\d{1,3}", st) or st.startswith("Latvijas Nacionālais attīstības plāns"):
|
||||
continue
|
||||
if re.match(r"^(Rīcības virziena|RĪCĪBAS|PRIORITĀTES|Prioritāte|NAP2027|Nr\.|\*|\d{1,2} [A-ZĀČĒĢĪĶĻŅŠŪŽa-z])", st) or len(ln) - len(ln.lstrip()) == 0 and len(st) > 60:
|
||||
break
|
||||
out.append(ln)
|
||||
return "\n".join(out)
|
||||
for it in items:
|
||||
n, kind = int(it.get("n")), it.get("kind")
|
||||
if kind in ("indicator", "task"):
|
||||
@@ -97,6 +114,19 @@ def main(xml_path, pdf_path, out):
|
||||
missing.append(w)
|
||||
if missing:
|
||||
row_bad.append({"n": n, "missing_words": missing[:10]})
|
||||
# reverse: every word printed in the row's part of the PDF is in the file (nothing dropped)
|
||||
xw = set(words(" ".join(vals)))
|
||||
xjoin = " ".join(sorted(xw))
|
||||
extra = []
|
||||
for w in words(block_rows(n)):
|
||||
w2 = re.sub(r"(?<=[a-zāčēģīķļņšūž])\d{1,2}$", "", w) # footnote number printed after a word
|
||||
if w in xw or w2 in xw or w in BOILER or w2 in BOILER or (w.isdigit() or len(w) >= 2) and w in xjoin:
|
||||
continue
|
||||
if w.isdigit() and len(w) > 2 and (w[:-1] in xw or w[:-2] in xw): # number followed by a footnote number
|
||||
continue
|
||||
extra.append(w)
|
||||
if extra:
|
||||
row_extra.append({"n": n, "words_not_in_file": extra[:12]})
|
||||
else:
|
||||
text_n += 1
|
||||
ks = [runs(letters(x.text), flat) for x in it.findall("s:Text", NS) + it.findall("s:Area", NS)]
|
||||
@@ -107,7 +137,8 @@ def main(xml_path, pdf_path, out):
|
||||
text_split.append(n)
|
||||
res["checks"]["text_items"] = {"checked": text_n, "not_found_verbatim": text_bad,
|
||||
"found_in_2_to_4_pieces": text_split}
|
||||
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad}
|
||||
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad,
|
||||
"rows_with_words_not_in_file": row_extra}
|
||||
|
||||
ev = doc.findall(".//s:Evidence", NS)
|
||||
ev_bad = [e.get("id") for e in ev if (runs(letters((e.findtext("s:Problem", namespaces=NS) or "") + e.findtext("s:Text", namespaces=NS)), flat) or 99) > 4]
|
||||
@@ -122,7 +153,7 @@ def main(xml_path, pdf_path, out):
|
||||
"indicators": len(inds), "indicators_without_values": no_val}
|
||||
c = res["checks"]
|
||||
res["passed"] = not (c["numbering"]["missing"] or c["numbering"]["extra"] or c["numbering"]["duplicates"] or text_bad
|
||||
or row_bad or ev_bad or no_resp)
|
||||
or row_bad or row_extra or ev_bad or no_resp)
|
||||
with open(out, "w", encoding="utf-8") as f:
|
||||
json.dump(res, f, ensure_ascii=False, indent=1)
|
||||
print(json.dumps({k: {kk: (vv if not isinstance(vv, list) else (len(vv) if len(vv) > 12 else vv)) for kk, vv in v.items()}
|
||||
|
||||
Reference in New Issue
Block a user