1
0

Nolikumi-as-Code 0.1.0: 93 iestāžu nolikumi kā teksts un XML

Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts —
likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/),
datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti
sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
2026-10-11 16:42:50 +00:00
commit 0c5116a2c6
198 changed files with 29909 additions and 0 deletions

275
tools/convert.py Normal file
View File

@@ -0,0 +1,275 @@
#!/usr/bin/env python3
"""likumi.lv konsolidētā redakcija (HTML) → sources/<fails>.txt un data/<fails>.xml (shēma nolikums-0.1).
python3 tools/convert.py <html> <VPK ID> <iestādes nosaukums> <datnes nosaukums bez paplašinājuma> <likumi.lv id> <ielādes datums>
HTML netiek glabāts repozitorijā: avots ir teksta datne, XML atbilst tai (tools/validate.py pārbauda).
"""
import hashlib
import html
import os
import re
import sys
from xml.sax.saxutils import escape, quoteattr
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
def clean(x):
x = re.sub(r"<br\s*/?>", " ", x or "")
x = re.sub(r"<[^>]+>", "", x)
return re.sub(r"\s+", " ", html.unescape(x)).replace(" ", " ").strip()
def a(k, v):
return f" {k}={quoteattr(str(v))}" if v not in (None, "") else ""
def el(tag, text, **at):
return f"<{tag}{''.join(a(k, v) for k, v in at.items())}>{escape(text)}</{tag}>" if text else ""
def chapter_kind(t):
t = (t or "").lower()
if "vispārīg" in t:
return "general"
if "noslēguma" in t or "pārejas" in t:
return "final"
if "padotībā" in t or "kapitāl" in t:
return "subordinates"
if "tiesiskum" in t or "pārskat" in t:
return "legality"
if "funkcij" in t or "uzdevum" in t or "kompetenc" in t or "tiesīb" in t:
return "competence"
if "pārvald" in t or "struktūr" in t or "amatpersonu" in t or "darba organiz" in t or "vadīb" in t:
return "management"
return "other"
def lead_kind(lead):
"""Kind of the items a lead sentence introduces: "… ir šādas funkcijas:" → function, "… šādi uzdevumi:" → task,
"… šādas tiesības:" → right; "Lai nodrošinātu funkciju izpildi, … :" → task."""
l = lead.lower()
m = re.search(r"šād(?:as|i|u)\s+(funkcij|uzdevum|tiesīb|kompetenc)", l)
if m:
return {"funkcij": "function", "uzdevum": "task", "tiesīb": "right", "kompetenc": "task"}[m.group(1)]
if re.search(r"funkciju (izpild|īstenoš)|nodrošinātu .*funkcij", l):
return "task"
for k, w in (("right", "tiesīb"), ("task", "uzdevum"), ("function", "funkcij")):
if w in l:
return k
return "provision"
def iso(d):
m = re.match(r"(\d{2})\.(\d{2})\.(\d{4})", d or "")
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
SUP = str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹")
def text_of(el):
"""Visible text of an element: <sup> digits as superscript characters, <br> as space, notes kept out by the caller."""
import copy
e = copy.deepcopy(el)
for s in e.iter("sup"):
s.text = (s.text or "").translate(SUP)
for b in e.iter("br"):
b.tail = " " + (b.tail or "")
return re.sub(r"\s+", " ", e.text_content().replace("\u00ad", "").replace("\u00a0", " ")).strip()
def lines_of(el):
import copy
e = copy.deepcopy(el)
for b in e.iter("br"):
b.tail = "\n" + (b.tail or "")
return [re.sub(r"\s+", " ", x).strip() for x in e.text_content().split("\n") if x.strip()]
def block_lines(el):
"""Text lines of an annex block; a table becomes one line per row with cells separated by ' | '."""
rows = el.xpath(".//tr")
if not rows:
return [text_of(el)] if text_of(el) else []
out = []
for tr in rows:
cells = [text_of(td) for td in tr.xpath("./td|./th")]
if any(cells):
out.append(" | ".join(cells))
return out
def parse(page):
import lxml.html
pase_txt = clean(re.sub(r"<script.*?</script>|<style.*?</style>", " ", page, flags=re.S))
k = pase_txt.rfind("Tiesību akta pase")
seg = pase_txt[k:k + 900]
keys = ["Nosaukums", "Statuss", "Izdevējs", "Veids", "Numurs", "Pieņemts", "Stājas spēkā", "Zaudē spēku", "Publicēts"]
def f(name):
m = re.search(re.escape(name) + r":\s*(.*?)\s*(?:" + "|".join(re.escape(x) + ":" for x in keys if x != name) + r"|Satura rādītājs|$)", seg)
return m.group(1).strip() if m else None
pase = {x: f(x) for x in keys}
pase["Statuss"] = " ".join((pase["Statuss"] or "").split()[:2])
if pase["Publicēts"]:
pase["Publicēts"] = re.split(r"\s+(?:OP numurs|Dokumenta valoda|Saistītie dokumenti)", pase["Publicēts"])[0].strip()
root = lxml.html.fromstring(page)
t207 = root.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0]
body = t207.getparent()
head = title = basis = None
chapters, cur, sigs, annexes, tail = [], None, [], [], []
after_points = False
for el in body:
cls = (el.get("class") or "").split()
c = cls[0] if cls else el.tag
if c == "TV206":
head = " ".join(lines_of(el))
elif c == "TV207":
title = text_of(el)
elif c == "TV900" and not chapters:
basis = text_of(el)
elif sigs and c.startswith("TV") and text_of(el):
# after the signatories: annexes (pielikumi) with their own headings, numbered rows and tables
for t in block_lines(el):
annexes.append(t)
tail.append(t)
elif c == "TV212":
t = text_of(el)
m = re.match(r"^(.*?)\s*(\((?:Nodaļas nosaukums|Nodaļa)[^()]*(?:\([^()]*\)[^()]*)*\))$", t)
cur = {"title": m.group(1) if m else t, "note": m.group(2) if m else None, "points": []}
chapters.append(cur)
elif c == "TV213" and el.get("data-num"):
if cur is None:
cur = {"title": None, "note": None, "points": []}
chapters.append(cur)
notes = [text_of(p) for p in el.xpath(".//p[contains(@class,'labojumu_pamats')]")]
paras = [text_of(p) for p in el.xpath("./p[contains(concat(' ', normalize-space(@class), ' '), ' TV213 ')]")]
paras = [p for p in paras if p]
if paras:
cur["points"].append({"num": el.get("data-num"), "paras": paras, "notes": [n for n in notes if n]})
after_points = True
elif after_points and c in ("TV216", "TV217", "TV218", "TV219"):
sigs += lines_of(el)
tail += lines_of(el)
elif after_points and c.startswith("TV") and text_of(el):
annexes.append(text_of(el))
tail.append(text_of(el))
return pase, head, title, basis, chapters, sigs, annexes, tail
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
def split_num(text):
"""Printed point number at the start of a paragraph → (normalised number, rest).
"3.1. x" → 3.1; inserted points with superscript: "4.4.¹ x" and "4.4¹. x" → 4.4¹; "24.¹1. x" and "24.¹ 1. x" → 24¹.1"""
m = re.match(r"^(\d+)", text)
if not m:
return None, text
segs, i = [m.group(1)], m.end()
while True:
rest = text[i:]
sup = re.match(r"^\.?([" + SUPC + r"]+)", rest)
if sup and not segs[-1].endswith(tuple(SUPC)):
segs[-1] += sup.group(1)
i += sup.end()
rest = text[i:]
nxt = re.match(r"^\.?\s?(\d+)(?=[." + SUPC + r"])", rest) # 24¹.1 written as "24.¹1." or "24.¹ 1."
if nxt:
segs.append(nxt.group(1)); i += nxt.end()
continue
nxt = re.match(r"^\.(\d+)(?=[." + SUPC + r"])", rest)
if nxt:
segs.append(nxt.group(1)); i += nxt.end()
continue
break
end = re.match(r"^\.\s*", text[i:]) or (re.match(r"^\s+", text[i:]) if segs[-1][-1] in SUPC else None)
if not end:
return None, text
return ".".join(segs), text[i + end.end():]
def build(page, org, inst_name, stem, likumi_id, retrieved):
pase, head, title, basis, chapters, sigs, annexes, tail = parse(page)
doc = f"lv-nol-{org}"
# ---------------------------------------------------------------- text source
lines = [x for x in (head, title, basis) if x]
for ch in chapters:
if ch["title"]:
lines.append("")
lines.append(ch["title"] + (" " + ch["note"] if ch.get("note") else ""))
for p in ch["points"]:
lines += p["paras"]
lines += p["notes"]
if tail: # signatories and annexes in document order
lines.append("")
lines += tail
txt = "\n".join(lines).strip() + "\n"
txt_rel = f"sources/{stem}.txt"
os.makedirs(os.path.join(ROOT, "sources"), exist_ok=True)
open(os.path.join(ROOT, txt_rel), "w", encoding="utf-8").write(txt)
sha = hashlib.sha256(txt.encode("utf-8")).hexdigest()
# ---------------------------------------------------------------- XML
out = []
for ci, ch in enumerate(chapters, 1):
ck = chapter_kind(ch["title"]) if ch["title"] else "none"
roman = re.match(r"^([IVXLC]+)\.\s", ch["title"] or "")
parts = [f'<Chapter id="{doc}.c{ci}"{a("n", roman.group(1) if roman else None)} kind="{ck}">', el("Title", ch["title"]), el("Note", ch.get("note"))]
for p in ch["points"]:
n0, t0 = split_num(p["paras"][0]) if p["paras"] else (p["num"], "")
n0 = n0 or p["num"]
lead = t0.rstrip().endswith(":") and len(p["paras"]) > 1
deleted = t0.strip().startswith("(svītrots")
k0 = "deleted" if deleted else "lead" if lead else "provision"
child_kind = lead_kind(t0) if (lead and ck == "competence") else "provision"
# nested sub-points by printed number depth
stack = [(n0, [])]
node = {"n": n0, "text": t0, "kind": k0, "notes": p["notes"], "kids": []}
path = [node]
for para in p["paras"][1:]:
n, t = split_num(para)
if n is None: # continuation paragraph without its own number
path[-1]["text"] += " " + para
continue
kd = "deleted" if t.strip().startswith("(svītrots") else ("lead" if t.rstrip().endswith(":") else child_kind)
item = {"n": n, "text": t, "kind": kd, "notes": [], "kids": []}
while len(path) > 1 and not n.startswith(path[-1]["n"] + "."):
path.pop()
path[-1]["kids"].append(item)
path.append(item)
def render(x):
pid = f"{doc}.p" + re.sub(r"([⁰¹²³⁴⁵⁶⁷⁸⁹]+)", lambda m: "s" + m.group(1).translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")), x["n"])
return (f'<Point id="{pid}" n="{x["n"]}" kind="{x["kind"]}">' + el("Text", x["text"].strip() or "(tukšs)")
+ "".join(el("Note", nn) for nn in x["notes"]) + "".join(render(k) for k in x["kids"]) + "</Point>")
parts.append(render(node))
parts.append("</Chapter>")
out.append("".join(parts))
sig_xml = "".join(el("Signatory", s, role="parakstītājs") for s in sigs)
ann_xml = "".join(f'<Annex id="{doc}.a{i}">' + el("Title", t.split(" ", 6)[0:6] and " ".join(t.split(" ")[:6])) + el("Text", t) + "</Annex>"
for i, t in enumerate(annexes, 1))
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
f'<InstitutionStatute xmlns="urn:pppa:vpk:nolikums:0.1" schemaVersion="0.1" id="{doc}">'
+ el("Institution", inst_name, org=org)
+ "<Act>" + el("Title", pase["Nosaukums"] or title) + el("Issuer", pase["Izdevējs"]) + el("Type", pase["Veids"])
+ el("Number", pase["Numurs"]) + el("Adopted", iso(pase["Pieņemts"])) + el("InForce", iso(pase["Stājas spēkā"]))
+ el("Published", pase["Publicēts"]) + el("Status", pase["Statuss"]) + el("LikumiId", str(likumi_id))
+ el("Url", f"https://likumi.lv/ta/id/{likumi_id}") + "</Act>"
+ "<Source>" + el("File", txt_rel) + el("SHA256", sha) + el("Retrieved", retrieved) + "</Source>"
+ el("LegalBasis", basis) + "".join(out) + sig_xml + ann_xml
+ f'<DataVersion number="1" date="{retrieved}">' + el("Change", "Pirmā versija no likumi.lv konsolidētās redakcijas.") + "</DataVersion>"
+ "</InstitutionStatute>\n")
from lxml import etree
tree = etree.fromstring(xml.encode("utf-8"))
etree.indent(tree, space=" ")
os.makedirs(os.path.join(ROOT, "data"), exist_ok=True)
with open(os.path.join(ROOT, "data", stem + ".xml"), "wb") as fh:
fh.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
return {"chapters": len(chapters), "points": sum(len(c["points"]) for c in chapters), "pase": pase}
if __name__ == "__main__":
pg = open(sys.argv[1], encoding="utf-8").read()
print(build(pg, sys.argv[2], sys.argv[3], sys.argv[4], int(sys.argv[5]), sys.argv[6]))