Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts — likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/), datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
87 lines
4.0 KiB
Python
87 lines
4.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Nolikumi-as-Code pārbaudītājs.
|
|
|
|
python3 tools/validate.py [--registry organizacijas.xml]
|
|
|
|
Katram nolikumam katalogā nolikumi.yaml:
|
|
- XML atbilst shēmai schemas/nolikums-0.1.xsd;
|
|
- datnes nosaukums sākas ar iestādes VPK ID, un VPK ID ir Valsts institūciju reģistrā;
|
|
- teksta datnes sha256 sakrīt ar Source/SHA256;
|
|
- katrs XML punkts ir teksta datnē (rinda „<numurs>. <teksts>”), un katra numurēta teksta rinda ir XML punkts;
|
|
- katalogā likumi.lv id sakrīt ar XML.
|
|
Reģistrs pēc noklusējuma — ProcessGit Valdibas-Deklaracija-as-Code data/organizacijas.xml.
|
|
"""
|
|
import hashlib
|
|
import os
|
|
import re
|
|
import sys
|
|
import urllib.request
|
|
|
|
import yaml
|
|
from lxml import etree
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
NS = {"n": "urn:pppa:vpk:nolikums:0.1"}
|
|
REG_URL = "https://processgit.org/Valsts-Pirmkods/Valdibas-Deklaracija-as-Code/raw/branch/main/data/organizacijas.xml"
|
|
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
from convert import split_num # noqa: E402
|
|
|
|
|
|
def main():
|
|
src = sys.argv[sys.argv.index("--registry") + 1] if "--registry" in sys.argv else REG_URL
|
|
reg = etree.parse(src if os.path.exists(src) else
|
|
urllib.request.urlopen(urllib.request.Request(src, headers={"User-Agent": "nolikumi-as-code-validate/0.1"})))
|
|
ids = {o.get("id") for o in reg.getroot() if etree.QName(o).localname == "Organization"}
|
|
xsd = etree.XMLSchema(etree.parse(os.path.join(ROOT, "schemas/nolikums-0.1.xsd")))
|
|
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
|
errors, points = [], 0
|
|
for e in cat["nolikumi"]:
|
|
path = os.path.join(ROOT, e["file"])
|
|
tag = os.path.basename(path)
|
|
doc = etree.parse(path)
|
|
if not xsd.validate(doc):
|
|
errors += [f"{tag}: XSD {x.line}: {x.message}" for x in list(xsd.error_log)[:5]]
|
|
continue
|
|
r = doc.getroot()
|
|
org = r.find("n:Institution", NS).get("org")
|
|
if not tag.startswith(org + "-") or org != e["org"]:
|
|
errors.append(f"{tag}: datnes nosaukums vai katalogs neatbilst VPK ID {org}")
|
|
if org not in ids:
|
|
errors.append(f"{tag}: VPK ID {org} nav reģistrā")
|
|
if int(r.findtext("n:Act/n:LikumiId", namespaces=NS)) != int(e["likumi_id"]):
|
|
errors.append(f"{tag}: likumi.lv id katalogā un datnē atšķiras")
|
|
txt_path = os.path.join(ROOT, r.findtext("n:Source/n:File", namespaces=NS))
|
|
raw = open(txt_path, "rb").read()
|
|
if hashlib.sha256(raw).hexdigest() != r.findtext("n:Source/n:SHA256", namespaces=NS):
|
|
errors.append(f"{tag}: teksta datnes sha256 nesakrīt")
|
|
lines = raw.decode("utf-8").splitlines()
|
|
# only the body: stop at the first signatory or annex line
|
|
stops = {x.text for x in r.findall("n:Signatory", NS)} | {x.text for x in r.findall("n:Annex/n:Text", NS)}
|
|
cut = next((k for k, ln in enumerate(lines) if ln in stops), len(lines))
|
|
lines = lines[:cut]
|
|
numbered = {}
|
|
for ln in lines:
|
|
n, t = split_num(ln)
|
|
if n is not None:
|
|
numbered.setdefault(n, []).append(t)
|
|
xml_pts = {}
|
|
for p in r.iter("{urn:pppa:vpk:nolikums:0.1}Point"):
|
|
points += 1
|
|
xml_pts[p.get("n")] = p.findtext("n:Text", namespaces=NS)
|
|
miss = [n for n, t in xml_pts.items() if not any(t.startswith(x.split(" (")[0][:40]) or x.startswith(t[:40])
|
|
for x in numbered.get(n, []))]
|
|
extra = [n for n in numbered if n not in xml_pts]
|
|
if miss:
|
|
errors.append(f"{tag}: XML punkti, kuru nav teksta datnē: {', '.join(miss[:8])}")
|
|
if extra:
|
|
errors.append(f"{tag}: teksta datnes numurētas rindas, kuru nav XML: {', '.join(extra[:8])}")
|
|
for x in errors:
|
|
print("KĻŪDA", x)
|
|
print(f"nolikumi: {len(cat['nolikumi'])}, punkti: {points}, kļūdas: {len(errors)}")
|
|
sys.exit(1 if errors else 0)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|