Nolikumi-as-Code 0.1.0: 93 iestāžu nolikumi kā teksts un XML
Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts — likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/), datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
86
tools/validate.py
Normal file
86
tools/validate.py
Normal file
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Nolikumi-as-Code pārbaudītājs.
|
||||
|
||||
python3 tools/validate.py [--registry organizacijas.xml]
|
||||
|
||||
Katram nolikumam katalogā nolikumi.yaml:
|
||||
- XML atbilst shēmai schemas/nolikums-0.1.xsd;
|
||||
- datnes nosaukums sākas ar iestādes VPK ID, un VPK ID ir Valsts institūciju reģistrā;
|
||||
- teksta datnes sha256 sakrīt ar Source/SHA256;
|
||||
- katrs XML punkts ir teksta datnē (rinda „<numurs>. <teksts>”), un katra numurēta teksta rinda ir XML punkts;
|
||||
- katalogā likumi.lv id sakrīt ar XML.
|
||||
Reģistrs pēc noklusējuma — ProcessGit Valdibas-Deklaracija-as-Code data/organizacijas.xml.
|
||||
"""
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import urllib.request
|
||||
|
||||
import yaml
|
||||
from lxml import etree
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
NS = {"n": "urn:pppa:vpk:nolikums:0.1"}
|
||||
REG_URL = "https://processgit.org/Valsts-Pirmkods/Valdibas-Deklaracija-as-Code/raw/branch/main/data/organizacijas.xml"
|
||||
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
from convert import split_num # noqa: E402
|
||||
|
||||
|
||||
def main():
|
||||
src = sys.argv[sys.argv.index("--registry") + 1] if "--registry" in sys.argv else REG_URL
|
||||
reg = etree.parse(src if os.path.exists(src) else
|
||||
urllib.request.urlopen(urllib.request.Request(src, headers={"User-Agent": "nolikumi-as-code-validate/0.1"})))
|
||||
ids = {o.get("id") for o in reg.getroot() if etree.QName(o).localname == "Organization"}
|
||||
xsd = etree.XMLSchema(etree.parse(os.path.join(ROOT, "schemas/nolikums-0.1.xsd")))
|
||||
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
||||
errors, points = [], 0
|
||||
for e in cat["nolikumi"]:
|
||||
path = os.path.join(ROOT, e["file"])
|
||||
tag = os.path.basename(path)
|
||||
doc = etree.parse(path)
|
||||
if not xsd.validate(doc):
|
||||
errors += [f"{tag}: XSD {x.line}: {x.message}" for x in list(xsd.error_log)[:5]]
|
||||
continue
|
||||
r = doc.getroot()
|
||||
org = r.find("n:Institution", NS).get("org")
|
||||
if not tag.startswith(org + "-") or org != e["org"]:
|
||||
errors.append(f"{tag}: datnes nosaukums vai katalogs neatbilst VPK ID {org}")
|
||||
if org not in ids:
|
||||
errors.append(f"{tag}: VPK ID {org} nav reģistrā")
|
||||
if int(r.findtext("n:Act/n:LikumiId", namespaces=NS)) != int(e["likumi_id"]):
|
||||
errors.append(f"{tag}: likumi.lv id katalogā un datnē atšķiras")
|
||||
txt_path = os.path.join(ROOT, r.findtext("n:Source/n:File", namespaces=NS))
|
||||
raw = open(txt_path, "rb").read()
|
||||
if hashlib.sha256(raw).hexdigest() != r.findtext("n:Source/n:SHA256", namespaces=NS):
|
||||
errors.append(f"{tag}: teksta datnes sha256 nesakrīt")
|
||||
lines = raw.decode("utf-8").splitlines()
|
||||
# only the body: stop at the first signatory or annex line
|
||||
stops = {x.text for x in r.findall("n:Signatory", NS)} | {x.text for x in r.findall("n:Annex/n:Text", NS)}
|
||||
cut = next((k for k, ln in enumerate(lines) if ln in stops), len(lines))
|
||||
lines = lines[:cut]
|
||||
numbered = {}
|
||||
for ln in lines:
|
||||
n, t = split_num(ln)
|
||||
if n is not None:
|
||||
numbered.setdefault(n, []).append(t)
|
||||
xml_pts = {}
|
||||
for p in r.iter("{urn:pppa:vpk:nolikums:0.1}Point"):
|
||||
points += 1
|
||||
xml_pts[p.get("n")] = p.findtext("n:Text", namespaces=NS)
|
||||
miss = [n for n, t in xml_pts.items() if not any(t.startswith(x.split(" (")[0][:40]) or x.startswith(t[:40])
|
||||
for x in numbered.get(n, []))]
|
||||
extra = [n for n in numbered if n not in xml_pts]
|
||||
if miss:
|
||||
errors.append(f"{tag}: XML punkti, kuru nav teksta datnē: {', '.join(miss[:8])}")
|
||||
if extra:
|
||||
errors.append(f"{tag}: teksta datnes numurētas rindas, kuru nav XML: {', '.join(extra[:8])}")
|
||||
for x in errors:
|
||||
print("KĻŪDA", x)
|
||||
print(f"nolikumi: {len(cat['nolikumi'])}, punkti: {points}, kļūdas: {len(errors)}")
|
||||
sys.exit(1 if errors else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user