Nolikumi-as-Code 0.1.0: 93 iestāžu nolikumi kā teksts un XML
Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts — likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/), datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
62
tools/build_all.py
Normal file
62
tools/build_all.py
Normal file
@@ -0,0 +1,62 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Visi nolikumi: katalogs nolikumi.yaml (VPK ID → likumi.lv id) + likumi.lv HTML kešs → sources/*.txt, data/*.xml.
|
||||
|
||||
python3 tools/build_all.py <html kešs> <organizacijas.xml> <ielādes datums>
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import unicodedata
|
||||
|
||||
import yaml
|
||||
from lxml import etree
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
import convert # noqa: E402
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
DROP = {"latvijas", "republikas", "lr", "valsts", "un", "–", "-"}
|
||||
|
||||
|
||||
def ascii_(s):
|
||||
s = unicodedata.normalize("NFKD", s)
|
||||
return "".join(c for c in s if not unicodedata.combining(c))
|
||||
|
||||
|
||||
def short_name(name):
|
||||
w = [x for x in re.split(r"[^\wĀ-ž]+", name.lower()) if x]
|
||||
core = [x for x in w if x not in DROP] or w
|
||||
return "-".join(ascii_(x) for x in core[:4])
|
||||
|
||||
|
||||
def stem_for(org, reg):
|
||||
o = reg.get(org) or {}
|
||||
ab = next((x for x in o.get("abbr", []) if re.fullmatch(r"[A-ZĀ-Ž][A-Za-zĀ-ž]{1,9}", x)), None)
|
||||
return f"{org}-{ascii_(ab)}" if ab else f"{org}-{short_name(o.get('name', org))}"
|
||||
|
||||
|
||||
def main(cache, reg_path, retrieved):
|
||||
ns = {"v": "urn:pppa:cac:valdiba:0.2"}
|
||||
reg = {}
|
||||
for o in etree.parse(reg_path).getroot().findall("v:Organization", ns):
|
||||
reg[o.get("id")] = {"name": o.findtext("v:Name", namespaces=ns),
|
||||
"abbr": [x.text for x in o.findall("v:Abbreviation", ns)]}
|
||||
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
||||
report = []
|
||||
for e in cat["nolikumi"]:
|
||||
stem = stem_for(e["org"], reg)
|
||||
e["file"] = f"data/{stem}.xml"
|
||||
page = open(os.path.join(cache, f"{e['likumi_id']}.html"), encoding="utf-8").read()
|
||||
name = e.get("institution") or reg[e["org"]]["name"]
|
||||
r = convert.build(page, e["org"], name, stem, e["likumi_id"], retrieved)
|
||||
report.append({"org": e["org"], "file": e["file"], "chapters": r["chapters"], "points": r["points"],
|
||||
"status": r["pase"]["Statuss"], "title": r["pase"]["Nosaukums"]})
|
||||
with open(os.path.join(ROOT, "nolikumi.yaml"), "w", encoding="utf-8") as fh:
|
||||
yaml.safe_dump(cat, fh, allow_unicode=True, sort_keys=False, width=200)
|
||||
json.dump(report, open("/tmp/build_report.json", "w"), ensure_ascii=False, indent=1)
|
||||
print("built", len(report))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(*sys.argv[1:4])
|
||||
275
tools/convert.py
Normal file
275
tools/convert.py
Normal file
@@ -0,0 +1,275 @@
|
||||
#!/usr/bin/env python3
|
||||
"""likumi.lv konsolidētā redakcija (HTML) → sources/<fails>.txt un data/<fails>.xml (shēma nolikums-0.1).
|
||||
|
||||
python3 tools/convert.py <html> <VPK ID> <iestādes nosaukums> <datnes nosaukums bez paplašinājuma> <likumi.lv id> <ielādes datums>
|
||||
|
||||
HTML netiek glabāts repozitorijā: avots ir teksta datne, XML atbilst tai (tools/validate.py pārbauda).
|
||||
"""
|
||||
import hashlib
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from xml.sax.saxutils import escape, quoteattr
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def clean(x):
|
||||
x = re.sub(r"<br\s*/?>", " ", x or "")
|
||||
x = re.sub(r"<[^>]+>", "", x)
|
||||
return re.sub(r"\s+", " ", html.unescape(x)).replace(" ", " ").strip()
|
||||
|
||||
|
||||
def a(k, v):
|
||||
return f" {k}={quoteattr(str(v))}" if v not in (None, "") else ""
|
||||
|
||||
|
||||
def el(tag, text, **at):
|
||||
return f"<{tag}{''.join(a(k, v) for k, v in at.items())}>{escape(text)}</{tag}>" if text else ""
|
||||
|
||||
|
||||
def chapter_kind(t):
|
||||
t = (t or "").lower()
|
||||
if "vispārīg" in t:
|
||||
return "general"
|
||||
if "noslēguma" in t or "pārejas" in t:
|
||||
return "final"
|
||||
if "padotībā" in t or "kapitāl" in t:
|
||||
return "subordinates"
|
||||
if "tiesiskum" in t or "pārskat" in t:
|
||||
return "legality"
|
||||
if "funkcij" in t or "uzdevum" in t or "kompetenc" in t or "tiesīb" in t:
|
||||
return "competence"
|
||||
if "pārvald" in t or "struktūr" in t or "amatpersonu" in t or "darba organiz" in t or "vadīb" in t:
|
||||
return "management"
|
||||
return "other"
|
||||
|
||||
|
||||
def lead_kind(lead):
|
||||
"""Kind of the items a lead sentence introduces: "… ir šādas funkcijas:" → function, "… šādi uzdevumi:" → task,
|
||||
"… šādas tiesības:" → right; "Lai nodrošinātu funkciju izpildi, … :" → task."""
|
||||
l = lead.lower()
|
||||
m = re.search(r"šād(?:as|i|u)\s+(funkcij|uzdevum|tiesīb|kompetenc)", l)
|
||||
if m:
|
||||
return {"funkcij": "function", "uzdevum": "task", "tiesīb": "right", "kompetenc": "task"}[m.group(1)]
|
||||
if re.search(r"funkciju (izpild|īstenoš)|nodrošinātu .*funkcij", l):
|
||||
return "task"
|
||||
for k, w in (("right", "tiesīb"), ("task", "uzdevum"), ("function", "funkcij")):
|
||||
if w in l:
|
||||
return k
|
||||
return "provision"
|
||||
|
||||
|
||||
def iso(d):
|
||||
m = re.match(r"(\d{2})\.(\d{2})\.(\d{4})", d or "")
|
||||
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
|
||||
|
||||
|
||||
SUP = str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹")
|
||||
|
||||
|
||||
def text_of(el):
|
||||
"""Visible text of an element: <sup> digits as superscript characters, <br> as space, notes kept out by the caller."""
|
||||
import copy
|
||||
e = copy.deepcopy(el)
|
||||
for s in e.iter("sup"):
|
||||
s.text = (s.text or "").translate(SUP)
|
||||
for b in e.iter("br"):
|
||||
b.tail = " " + (b.tail or "")
|
||||
return re.sub(r"\s+", " ", e.text_content().replace("\u00ad", "").replace("\u00a0", " ")).strip()
|
||||
|
||||
|
||||
def lines_of(el):
|
||||
import copy
|
||||
e = copy.deepcopy(el)
|
||||
for b in e.iter("br"):
|
||||
b.tail = "\n" + (b.tail or "")
|
||||
return [re.sub(r"\s+", " ", x).strip() for x in e.text_content().split("\n") if x.strip()]
|
||||
|
||||
|
||||
def block_lines(el):
|
||||
"""Text lines of an annex block; a table becomes one line per row with cells separated by ' | '."""
|
||||
rows = el.xpath(".//tr")
|
||||
if not rows:
|
||||
return [text_of(el)] if text_of(el) else []
|
||||
out = []
|
||||
for tr in rows:
|
||||
cells = [text_of(td) for td in tr.xpath("./td|./th")]
|
||||
if any(cells):
|
||||
out.append(" | ".join(cells))
|
||||
return out
|
||||
|
||||
|
||||
def parse(page):
|
||||
import lxml.html
|
||||
pase_txt = clean(re.sub(r"<script.*?</script>|<style.*?</style>", " ", page, flags=re.S))
|
||||
k = pase_txt.rfind("Tiesību akta pase")
|
||||
seg = pase_txt[k:k + 900]
|
||||
keys = ["Nosaukums", "Statuss", "Izdevējs", "Veids", "Numurs", "Pieņemts", "Stājas spēkā", "Zaudē spēku", "Publicēts"]
|
||||
|
||||
def f(name):
|
||||
m = re.search(re.escape(name) + r":\s*(.*?)\s*(?:" + "|".join(re.escape(x) + ":" for x in keys if x != name) + r"|Satura rādītājs|$)", seg)
|
||||
return m.group(1).strip() if m else None
|
||||
pase = {x: f(x) for x in keys}
|
||||
pase["Statuss"] = " ".join((pase["Statuss"] or "").split()[:2])
|
||||
if pase["Publicēts"]:
|
||||
pase["Publicēts"] = re.split(r"\s+(?:OP numurs|Dokumenta valoda|Saistītie dokumenti)", pase["Publicēts"])[0].strip()
|
||||
root = lxml.html.fromstring(page)
|
||||
t207 = root.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0]
|
||||
body = t207.getparent()
|
||||
head = title = basis = None
|
||||
chapters, cur, sigs, annexes, tail = [], None, [], [], []
|
||||
after_points = False
|
||||
for el in body:
|
||||
cls = (el.get("class") or "").split()
|
||||
c = cls[0] if cls else el.tag
|
||||
if c == "TV206":
|
||||
head = " ".join(lines_of(el))
|
||||
elif c == "TV207":
|
||||
title = text_of(el)
|
||||
elif c == "TV900" and not chapters:
|
||||
basis = text_of(el)
|
||||
elif sigs and c.startswith("TV") and text_of(el):
|
||||
# after the signatories: annexes (pielikumi) with their own headings, numbered rows and tables
|
||||
for t in block_lines(el):
|
||||
annexes.append(t)
|
||||
tail.append(t)
|
||||
elif c == "TV212":
|
||||
t = text_of(el)
|
||||
m = re.match(r"^(.*?)\s*(\((?:Nodaļas nosaukums|Nodaļa)[^()]*(?:\([^()]*\)[^()]*)*\))$", t)
|
||||
cur = {"title": m.group(1) if m else t, "note": m.group(2) if m else None, "points": []}
|
||||
chapters.append(cur)
|
||||
elif c == "TV213" and el.get("data-num"):
|
||||
if cur is None:
|
||||
cur = {"title": None, "note": None, "points": []}
|
||||
chapters.append(cur)
|
||||
notes = [text_of(p) for p in el.xpath(".//p[contains(@class,'labojumu_pamats')]")]
|
||||
paras = [text_of(p) for p in el.xpath("./p[contains(concat(' ', normalize-space(@class), ' '), ' TV213 ')]")]
|
||||
paras = [p for p in paras if p]
|
||||
if paras:
|
||||
cur["points"].append({"num": el.get("data-num"), "paras": paras, "notes": [n for n in notes if n]})
|
||||
after_points = True
|
||||
elif after_points and c in ("TV216", "TV217", "TV218", "TV219"):
|
||||
sigs += lines_of(el)
|
||||
tail += lines_of(el)
|
||||
elif after_points and c.startswith("TV") and text_of(el):
|
||||
annexes.append(text_of(el))
|
||||
tail.append(text_of(el))
|
||||
return pase, head, title, basis, chapters, sigs, annexes, tail
|
||||
|
||||
|
||||
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
||||
|
||||
|
||||
def split_num(text):
|
||||
"""Printed point number at the start of a paragraph → (normalised number, rest).
|
||||
"3.1. x" → 3.1; inserted points with superscript: "4.4.¹ x" and "4.4¹. x" → 4.4¹; "24.¹1. x" and "24.¹ 1. x" → 24¹.1"""
|
||||
m = re.match(r"^(\d+)", text)
|
||||
if not m:
|
||||
return None, text
|
||||
segs, i = [m.group(1)], m.end()
|
||||
while True:
|
||||
rest = text[i:]
|
||||
sup = re.match(r"^\.?([" + SUPC + r"]+)", rest)
|
||||
if sup and not segs[-1].endswith(tuple(SUPC)):
|
||||
segs[-1] += sup.group(1)
|
||||
i += sup.end()
|
||||
rest = text[i:]
|
||||
nxt = re.match(r"^\.?\s?(\d+)(?=[." + SUPC + r"])", rest) # 24¹.1 written as "24.¹1." or "24.¹ 1."
|
||||
if nxt:
|
||||
segs.append(nxt.group(1)); i += nxt.end()
|
||||
continue
|
||||
nxt = re.match(r"^\.(\d+)(?=[." + SUPC + r"])", rest)
|
||||
if nxt:
|
||||
segs.append(nxt.group(1)); i += nxt.end()
|
||||
continue
|
||||
break
|
||||
end = re.match(r"^\.\s*", text[i:]) or (re.match(r"^\s+", text[i:]) if segs[-1][-1] in SUPC else None)
|
||||
if not end:
|
||||
return None, text
|
||||
return ".".join(segs), text[i + end.end():]
|
||||
|
||||
|
||||
def build(page, org, inst_name, stem, likumi_id, retrieved):
|
||||
pase, head, title, basis, chapters, sigs, annexes, tail = parse(page)
|
||||
doc = f"lv-nol-{org}"
|
||||
# ---------------------------------------------------------------- text source
|
||||
lines = [x for x in (head, title, basis) if x]
|
||||
for ch in chapters:
|
||||
if ch["title"]:
|
||||
lines.append("")
|
||||
lines.append(ch["title"] + (" " + ch["note"] if ch.get("note") else ""))
|
||||
for p in ch["points"]:
|
||||
lines += p["paras"]
|
||||
lines += p["notes"]
|
||||
if tail: # signatories and annexes in document order
|
||||
lines.append("")
|
||||
lines += tail
|
||||
txt = "\n".join(lines).strip() + "\n"
|
||||
txt_rel = f"sources/{stem}.txt"
|
||||
os.makedirs(os.path.join(ROOT, "sources"), exist_ok=True)
|
||||
open(os.path.join(ROOT, txt_rel), "w", encoding="utf-8").write(txt)
|
||||
sha = hashlib.sha256(txt.encode("utf-8")).hexdigest()
|
||||
# ---------------------------------------------------------------- XML
|
||||
out = []
|
||||
for ci, ch in enumerate(chapters, 1):
|
||||
ck = chapter_kind(ch["title"]) if ch["title"] else "none"
|
||||
roman = re.match(r"^([IVXLC]+)\.\s", ch["title"] or "")
|
||||
parts = [f'<Chapter id="{doc}.c{ci}"{a("n", roman.group(1) if roman else None)} kind="{ck}">', el("Title", ch["title"]), el("Note", ch.get("note"))]
|
||||
for p in ch["points"]:
|
||||
n0, t0 = split_num(p["paras"][0]) if p["paras"] else (p["num"], "")
|
||||
n0 = n0 or p["num"]
|
||||
lead = t0.rstrip().endswith(":") and len(p["paras"]) > 1
|
||||
deleted = t0.strip().startswith("(svītrots")
|
||||
k0 = "deleted" if deleted else "lead" if lead else "provision"
|
||||
child_kind = lead_kind(t0) if (lead and ck == "competence") else "provision"
|
||||
# nested sub-points by printed number depth
|
||||
stack = [(n0, [])]
|
||||
node = {"n": n0, "text": t0, "kind": k0, "notes": p["notes"], "kids": []}
|
||||
path = [node]
|
||||
for para in p["paras"][1:]:
|
||||
n, t = split_num(para)
|
||||
if n is None: # continuation paragraph without its own number
|
||||
path[-1]["text"] += " " + para
|
||||
continue
|
||||
kd = "deleted" if t.strip().startswith("(svītrots") else ("lead" if t.rstrip().endswith(":") else child_kind)
|
||||
item = {"n": n, "text": t, "kind": kd, "notes": [], "kids": []}
|
||||
while len(path) > 1 and not n.startswith(path[-1]["n"] + "."):
|
||||
path.pop()
|
||||
path[-1]["kids"].append(item)
|
||||
path.append(item)
|
||||
|
||||
def render(x):
|
||||
pid = f"{doc}.p" + re.sub(r"([⁰¹²³⁴⁵⁶⁷⁸⁹]+)", lambda m: "s" + m.group(1).translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")), x["n"])
|
||||
return (f'<Point id="{pid}" n="{x["n"]}" kind="{x["kind"]}">' + el("Text", x["text"].strip() or "(tukšs)")
|
||||
+ "".join(el("Note", nn) for nn in x["notes"]) + "".join(render(k) for k in x["kids"]) + "</Point>")
|
||||
parts.append(render(node))
|
||||
parts.append("</Chapter>")
|
||||
out.append("".join(parts))
|
||||
sig_xml = "".join(el("Signatory", s, role="parakstītājs") for s in sigs)
|
||||
ann_xml = "".join(f'<Annex id="{doc}.a{i}">' + el("Title", t.split(" ", 6)[0:6] and " ".join(t.split(" ")[:6])) + el("Text", t) + "</Annex>"
|
||||
for i, t in enumerate(annexes, 1))
|
||||
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
|
||||
f'<InstitutionStatute xmlns="urn:pppa:vpk:nolikums:0.1" schemaVersion="0.1" id="{doc}">'
|
||||
+ el("Institution", inst_name, org=org)
|
||||
+ "<Act>" + el("Title", pase["Nosaukums"] or title) + el("Issuer", pase["Izdevējs"]) + el("Type", pase["Veids"])
|
||||
+ el("Number", pase["Numurs"]) + el("Adopted", iso(pase["Pieņemts"])) + el("InForce", iso(pase["Stājas spēkā"]))
|
||||
+ el("Published", pase["Publicēts"]) + el("Status", pase["Statuss"]) + el("LikumiId", str(likumi_id))
|
||||
+ el("Url", f"https://likumi.lv/ta/id/{likumi_id}") + "</Act>"
|
||||
+ "<Source>" + el("File", txt_rel) + el("SHA256", sha) + el("Retrieved", retrieved) + "</Source>"
|
||||
+ el("LegalBasis", basis) + "".join(out) + sig_xml + ann_xml
|
||||
+ f'<DataVersion number="1" date="{retrieved}">' + el("Change", "Pirmā versija no likumi.lv konsolidētās redakcijas.") + "</DataVersion>"
|
||||
+ "</InstitutionStatute>\n")
|
||||
from lxml import etree
|
||||
tree = etree.fromstring(xml.encode("utf-8"))
|
||||
etree.indent(tree, space=" ")
|
||||
os.makedirs(os.path.join(ROOT, "data"), exist_ok=True)
|
||||
with open(os.path.join(ROOT, "data", stem + ".xml"), "wb") as fh:
|
||||
fh.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
|
||||
return {"chapters": len(chapters), "points": sum(len(c["points"]) for c in chapters), "pase": pase}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pg = open(sys.argv[1], encoding="utf-8").read()
|
||||
print(build(pg, sys.argv[2], sys.argv[3], sys.argv[4], int(sys.argv[5]), sys.argv[6]))
|
||||
86
tools/validate.py
Normal file
86
tools/validate.py
Normal file
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Nolikumi-as-Code pārbaudītājs.
|
||||
|
||||
python3 tools/validate.py [--registry organizacijas.xml]
|
||||
|
||||
Katram nolikumam katalogā nolikumi.yaml:
|
||||
- XML atbilst shēmai schemas/nolikums-0.1.xsd;
|
||||
- datnes nosaukums sākas ar iestādes VPK ID, un VPK ID ir Valsts institūciju reģistrā;
|
||||
- teksta datnes sha256 sakrīt ar Source/SHA256;
|
||||
- katrs XML punkts ir teksta datnē (rinda „<numurs>. <teksts>”), un katra numurēta teksta rinda ir XML punkts;
|
||||
- katalogā likumi.lv id sakrīt ar XML.
|
||||
Reģistrs pēc noklusējuma — ProcessGit Valdibas-Deklaracija-as-Code data/organizacijas.xml.
|
||||
"""
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
import yaml
|
||||
from lxml import etree
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
NS = {"n": "urn:pppa:vpk:nolikums:0.1"}
|
||||
REG_URL = "https://processgit.org/Valsts-Pirmkods/Valdibas-Deklaracija-as-Code/raw/branch/main/data/organizacijas.xml"
|
||||
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from convert import split_num # noqa: E402
|
||||
|
||||
|
||||
def main():
|
||||
src = sys.argv[sys.argv.index("--registry") + 1] if "--registry" in sys.argv else REG_URL
|
||||
reg = etree.parse(src if os.path.exists(src) else
|
||||
urllib.request.urlopen(urllib.request.Request(src, headers={"User-Agent": "nolikumi-as-code-validate/0.1"})))
|
||||
ids = {o.get("id") for o in reg.getroot() if etree.QName(o).localname == "Organization"}
|
||||
xsd = etree.XMLSchema(etree.parse(os.path.join(ROOT, "schemas/nolikums-0.1.xsd")))
|
||||
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
||||
errors, points = [], 0
|
||||
for e in cat["nolikumi"]:
|
||||
path = os.path.join(ROOT, e["file"])
|
||||
tag = os.path.basename(path)
|
||||
doc = etree.parse(path)
|
||||
if not xsd.validate(doc):
|
||||
errors += [f"{tag}: XSD {x.line}: {x.message}" for x in list(xsd.error_log)[:5]]
|
||||
continue
|
||||
r = doc.getroot()
|
||||
org = r.find("n:Institution", NS).get("org")
|
||||
if not tag.startswith(org + "-") or org != e["org"]:
|
||||
errors.append(f"{tag}: datnes nosaukums vai katalogs neatbilst VPK ID {org}")
|
||||
if org not in ids:
|
||||
errors.append(f"{tag}: VPK ID {org} nav reģistrā")
|
||||
if int(r.findtext("n:Act/n:LikumiId", namespaces=NS)) != int(e["likumi_id"]):
|
||||
errors.append(f"{tag}: likumi.lv id katalogā un datnē atšķiras")
|
||||
txt_path = os.path.join(ROOT, r.findtext("n:Source/n:File", namespaces=NS))
|
||||
raw = open(txt_path, "rb").read()
|
||||
if hashlib.sha256(raw).hexdigest() != r.findtext("n:Source/n:SHA256", namespaces=NS):
|
||||
errors.append(f"{tag}: teksta datnes sha256 nesakrīt")
|
||||
lines = raw.decode("utf-8").splitlines()
|
||||
# only the body: stop at the first signatory or annex line
|
||||
stops = {x.text for x in r.findall("n:Signatory", NS)} | {x.text for x in r.findall("n:Annex/n:Text", NS)}
|
||||
cut = next((k for k, ln in enumerate(lines) if ln in stops), len(lines))
|
||||
lines = lines[:cut]
|
||||
numbered = {}
|
||||
for ln in lines:
|
||||
n, t = split_num(ln)
|
||||
if n is not None:
|
||||
numbered.setdefault(n, []).append(t)
|
||||
xml_pts = {}
|
||||
for p in r.iter("{urn:pppa:vpk:nolikums:0.1}Point"):
|
||||
points += 1
|
||||
xml_pts[p.get("n")] = p.findtext("n:Text", namespaces=NS)
|
||||
miss = [n for n, t in xml_pts.items() if not any(t.startswith(x.split(" (")[0][:40]) or x.startswith(t[:40])
|
||||
for x in numbered.get(n, []))]
|
||||
extra = [n for n in numbered if n not in xml_pts]
|
||||
if miss:
|
||||
errors.append(f"{tag}: XML punkti, kuru nav teksta datnē: {', '.join(miss[:8])}")
|
||||
if extra:
|
||||
errors.append(f"{tag}: teksta datnes numurētas rindas, kuru nav XML: {', '.join(extra[:8])}")
|
||||
for x in errors:
|
||||
print("KĻŪDA", x)
|
||||
print(f"nolikumi: {len(cat['nolikumi'])}, punkti: {points}, kļūdas: {len(errors)}")
|
||||
sys.exit(1 if errors else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
53
tools/verify_html.py
Normal file
53
tools/verify_html.py
Normal file
@@ -0,0 +1,53 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Teksta datnes salīdzināšana ar likumi.lv lapu (lai pārliecinātos, ka teksts nav zaudēts vai pārveidots).
|
||||
|
||||
python3 tools/verify_html.py <html kešs>
|
||||
|
||||
Salīdzina burtu un ciparu secību: teksta datne pret lapas redzamo dokumenta tekstu no virsraksta līdz pēdējam
|
||||
parakstītājam (bez tiesību akta pases, saitēm un izvēlnēm). Lapu nolasa ar lxml text_content — citādi nekā convert.py.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
|
||||
import lxml.html
|
||||
import yaml
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def norm(s):
|
||||
return re.sub(r"[^0-9a-zāčēģīķļņšūž⁰¹²³⁴⁵⁶⁷⁸⁹]", "", s.lower().replace("", ""))
|
||||
|
||||
|
||||
def page_text(path):
|
||||
r = lxml.html.fromstring(open(path, encoding="utf-8").read())
|
||||
body = r.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0].getparent()
|
||||
for s in body.iter("sup"):
|
||||
s.text = (s.text or "").translate(str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹"))
|
||||
parts = []
|
||||
for el in body:
|
||||
c = (el.get("class") or "").split()
|
||||
if c and c[0].startswith("TV") and el.tag != "style":
|
||||
for junk in el.xpath(".//*[contains(@class,'info-icon-wrapper') or contains(@class,'panta-doc-npk')]"):
|
||||
junk.getparent().remove(junk)
|
||||
parts.append(el.text_content())
|
||||
return norm(" ".join(parts))
|
||||
|
||||
|
||||
def main(cache):
|
||||
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
||||
bad = 0
|
||||
for e in cat["nolikumi"]:
|
||||
stem = os.path.basename(e["file"])[:-4]
|
||||
t = norm(open(os.path.join(ROOT, "sources", stem + ".txt"), encoding="utf-8").read())
|
||||
h = page_text(os.path.join(cache, f"{e['likumi_id']}.html"))
|
||||
if t != h:
|
||||
bad += 1
|
||||
i = next((k for k in range(min(len(t), len(h))) if t[k] != h[k]), min(len(t), len(h)))
|
||||
print(f"ATŠĶIRAS {stem}: teksts {len(t)}, lapa {len(h)}; no {i}: '{t[i:i+60]}' / '{h[i:i+60]}'")
|
||||
print(f"pārbaudīti {len(cat['nolikumi'])}, atšķiras {bad}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1])
|
||||
Reference in New Issue
Block a user