Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts — likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/), datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
63 lines
2.4 KiB
Python
63 lines
2.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Visi nolikumi: katalogs nolikumi.yaml (VPK ID → likumi.lv id) + likumi.lv HTML kešs → sources/*.txt, data/*.xml.
|
|
|
|
python3 tools/build_all.py <html kešs> <organizacijas.xml> <ielādes datums>
|
|
"""
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import unicodedata
|
|
|
|
import yaml
|
|
from lxml import etree
|
|
|
|
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
import convert # noqa: E402
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
DROP = {"latvijas", "republikas", "lr", "valsts", "un", "–", "-"}
|
|
|
|
|
|
def ascii_(s):
|
|
s = unicodedata.normalize("NFKD", s)
|
|
return "".join(c for c in s if not unicodedata.combining(c))
|
|
|
|
|
|
def short_name(name):
|
|
w = [x for x in re.split(r"[^\wĀ-ž]+", name.lower()) if x]
|
|
core = [x for x in w if x not in DROP] or w
|
|
return "-".join(ascii_(x) for x in core[:4])
|
|
|
|
|
|
def stem_for(org, reg):
|
|
o = reg.get(org) or {}
|
|
ab = next((x for x in o.get("abbr", []) if re.fullmatch(r"[A-ZĀ-Ž][A-Za-zĀ-ž]{1,9}", x)), None)
|
|
return f"{org}-{ascii_(ab)}" if ab else f"{org}-{short_name(o.get('name', org))}"
|
|
|
|
|
|
def main(cache, reg_path, retrieved):
|
|
ns = {"v": "urn:pppa:cac:valdiba:0.2"}
|
|
reg = {}
|
|
for o in etree.parse(reg_path).getroot().findall("v:Organization", ns):
|
|
reg[o.get("id")] = {"name": o.findtext("v:Name", namespaces=ns),
|
|
"abbr": [x.text for x in o.findall("v:Abbreviation", ns)]}
|
|
cat = yaml.safe_load(open(os.path.join(ROOT, "nolikumi.yaml"), encoding="utf-8"))
|
|
report = []
|
|
for e in cat["nolikumi"]:
|
|
stem = stem_for(e["org"], reg)
|
|
e["file"] = f"data/{stem}.xml"
|
|
page = open(os.path.join(cache, f"{e['likumi_id']}.html"), encoding="utf-8").read()
|
|
name = e.get("institution") or reg[e["org"]]["name"]
|
|
r = convert.build(page, e["org"], name, stem, e["likumi_id"], retrieved)
|
|
report.append({"org": e["org"], "file": e["file"], "chapters": r["chapters"], "points": r["points"],
|
|
"status": r["pase"]["Statuss"], "title": r["pase"]["Nosaukums"]})
|
|
with open(os.path.join(ROOT, "nolikumi.yaml"), "w", encoding="utf-8") as fh:
|
|
yaml.safe_dump(cat, fh, allow_unicode=True, sort_keys=False, width=200)
|
|
json.dump(report, open("/tmp/build_report.json", "w"), ensure_ascii=False, indent=1)
|
|
print("built", len(report))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main(*sys.argv[1:4])
|