1
0

NAP2027 0.1.1: labotas uzdevumu [315]–[318] ailes; pašreizējā institūcija (KEM)

Uzdevumu teksta pirmie vārdi 68. lpp. bija nonākuši ailē „Nr.”; pārbaudei
pievienota apgrieztā vārdu pārbaude. Atribūts currentOrg/currentSince:
septiņiem VARAM vides un klimata uzdevumiem — KEM 20-0000.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
2026-10-11 13:01:14 +00:00
parent bcdd26b8d9
commit bc915bc6d2
10 changed files with 114 additions and 37 deletions

View File

@@ -21,6 +21,7 @@ from lxml import etree
NS = {"s": "urn:pppa:vpk:strategija:0.1"}
HEADER = "latvijasnacionālaisattīstībasplānsgadam"
BOILER = set()
SUBS = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
@@ -55,6 +56,9 @@ def main(xml_path, pdf_path, out):
lay = subprocess.run(["pdftotext", "-layout", pdf_path, "-"], capture_output=True, text=True, check=True).stdout
flat = letters(raw).replace(HEADER, "")
doc = etree.parse(xml_path)
# words that may stand between table rows in the PDF and belong elsewhere in the file: footnotes, sub-section titles
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Footnote", NS))))
BOILER.update(words(" ".join(x.text for x in doc.findall(".//s:Section/s:Title", NS))))
items = doc.findall(".//s:Item", NS)
res = {"file": xml_path, "pdf_sha256": hashlib.sha256(open(pdf_path, "rb").read()).hexdigest(), "checks": {}}
@@ -67,14 +71,27 @@ def main(xml_path, pdf_path, out):
# layout blocks between consecutive markers
pos = {int(m.group(1)): m.end() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
start = {int(m.group(1)): m.start() for m in re.finditer(r"\[(\d{1,3})\]", lay)}
order = sorted(pos)
def block(n):
k = order.index(n)
end = pos[order[k + 1]] if k + 1 < len(order) else len(lay)
end = start[order[k + 1]] if k + 1 < len(order) else len(lay)
return lay[pos[n]:end]
text_bad, text_split, row_bad, row_words, text_n, row_n = [], [], [], 0, 0, 0
text_bad, text_split, row_bad, row_extra, row_words, text_n, row_n = [], [], [], [], 0, 0, 0
def block_rows(n):
"""The row's lines only: stop at a blank line followed by body text, drop page header/footer lines."""
out = []
for ln in block(n).split("\n"):
st = ln.strip()
if not st or re.fullmatch(r"\d{1,3}", st) or st.startswith("Latvijas Nacionālais attīstības plāns"):
continue
if re.match(r"^(Rīcības virziena|RĪCĪBAS|PRIORITĀTES|Prioritāte|NAP2027|Nr\.|\*|\d{1,2} [A-ZĀČĒĢĪĶĻŅŠŪŽa-z])", st) or len(ln) - len(ln.lstrip()) == 0 and len(st) > 60:
break
out.append(ln)
return "\n".join(out)
for it in items:
n, kind = int(it.get("n")), it.get("kind")
if kind in ("indicator", "task"):
@@ -97,6 +114,19 @@ def main(xml_path, pdf_path, out):
missing.append(w)
if missing:
row_bad.append({"n": n, "missing_words": missing[:10]})
# reverse: every word printed in the row's part of the PDF is in the file (nothing dropped)
xw = set(words(" ".join(vals)))
xjoin = " ".join(sorted(xw))
extra = []
for w in words(block_rows(n)):
w2 = re.sub(r"(?<=[a-zāčēģīķļņšūž])\d{1,2}$", "", w) # footnote number printed after a word
if w in xw or w2 in xw or w in BOILER or w2 in BOILER or (w.isdigit() or len(w) >= 2) and w in xjoin:
continue
if w.isdigit() and len(w) > 2 and (w[:-1] in xw or w[:-2] in xw): # number followed by a footnote number
continue
extra.append(w)
if extra:
row_extra.append({"n": n, "words_not_in_file": extra[:12]})
else:
text_n += 1
ks = [runs(letters(x.text), flat) for x in it.findall("s:Text", NS) + it.findall("s:Area", NS)]
@@ -107,7 +137,8 @@ def main(xml_path, pdf_path, out):
text_split.append(n)
res["checks"]["text_items"] = {"checked": text_n, "not_found_verbatim": text_bad,
"found_in_2_to_4_pieces": text_split}
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad}
res["checks"]["table_rows"] = {"checked": row_n, "words": row_words, "rows_with_missing_words": row_bad,
"rows_with_words_not_in_file": row_extra}
ev = doc.findall(".//s:Evidence", NS)
ev_bad = [e.get("id") for e in ev if (runs(letters((e.findtext("s:Problem", namespaces=NS) or "") + e.findtext("s:Text", namespaces=NS)), flat) or 99) > 4]
@@ -122,7 +153,7 @@ def main(xml_path, pdf_path, out):
"indicators": len(inds), "indicators_without_values": no_val}
c = res["checks"]
res["passed"] = not (c["numbering"]["missing"] or c["numbering"]["extra"] or c["numbering"]["duplicates"] or text_bad
or row_bad or ev_bad or no_resp)
or row_bad or row_extra or ev_bad or no_resp)
with open(out, "w", encoding="utf-8") as f:
json.dump(res, f, ensure_ascii=False, indent=1)
print(json.dumps({k: {kk: (vv if not isinstance(vv, list) else (len(vv) if len(vv) > 12 else vv)) for kk, vv in v.items()}