Nolikumi-as-Code 0.1.0: 93 iestāžu nolikumi kā teksts un XML
Shēma nolikums-0.1, katalogs nolikumi.yaml, datu līgums B05. Teksts — likumi.lv konsolidētās redakcijas 2026-10-11 (sources/), dati (data/), datnes nosaukums <VPK ID>-<saīsinājums>. Pārbaude: 0 kļūdas; 92/93 teksti sakrīt ar likumi.lv lapu, 19-0458 — 2 cipari informatīvajā atsaucē. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
This commit is contained in:
275
tools/convert.py
Normal file
275
tools/convert.py
Normal file
@@ -0,0 +1,275 @@
|
||||
#!/usr/bin/env python3
|
||||
"""likumi.lv konsolidētā redakcija (HTML) → sources/<fails>.txt un data/<fails>.xml (shēma nolikums-0.1).
|
||||
|
||||
python3 tools/convert.py <html> <VPK ID> <iestādes nosaukums> <datnes nosaukums bez paplašinājuma> <likumi.lv id> <ielādes datums>
|
||||
|
||||
HTML netiek glabāts repozitorijā: avots ir teksta datne, XML atbilst tai (tools/validate.py pārbauda).
|
||||
"""
|
||||
import hashlib
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from xml.sax.saxutils import escape, quoteattr
|
||||
|
||||
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
|
||||
def clean(x):
|
||||
x = re.sub(r"<br\s*/?>", " ", x or "")
|
||||
x = re.sub(r"<[^>]+>", "", x)
|
||||
return re.sub(r"\s+", " ", html.unescape(x)).replace(" ", " ").strip()
|
||||
|
||||
|
||||
def a(k, v):
|
||||
return f" {k}={quoteattr(str(v))}" if v not in (None, "") else ""
|
||||
|
||||
|
||||
def el(tag, text, **at):
|
||||
return f"<{tag}{''.join(a(k, v) for k, v in at.items())}>{escape(text)}</{tag}>" if text else ""
|
||||
|
||||
|
||||
def chapter_kind(t):
|
||||
t = (t or "").lower()
|
||||
if "vispārīg" in t:
|
||||
return "general"
|
||||
if "noslēguma" in t or "pārejas" in t:
|
||||
return "final"
|
||||
if "padotībā" in t or "kapitāl" in t:
|
||||
return "subordinates"
|
||||
if "tiesiskum" in t or "pārskat" in t:
|
||||
return "legality"
|
||||
if "funkcij" in t or "uzdevum" in t or "kompetenc" in t or "tiesīb" in t:
|
||||
return "competence"
|
||||
if "pārvald" in t or "struktūr" in t or "amatpersonu" in t or "darba organiz" in t or "vadīb" in t:
|
||||
return "management"
|
||||
return "other"
|
||||
|
||||
|
||||
def lead_kind(lead):
|
||||
"""Kind of the items a lead sentence introduces: "… ir šādas funkcijas:" → function, "… šādi uzdevumi:" → task,
|
||||
"… šādas tiesības:" → right; "Lai nodrošinātu funkciju izpildi, … :" → task."""
|
||||
l = lead.lower()
|
||||
m = re.search(r"šād(?:as|i|u)\s+(funkcij|uzdevum|tiesīb|kompetenc)", l)
|
||||
if m:
|
||||
return {"funkcij": "function", "uzdevum": "task", "tiesīb": "right", "kompetenc": "task"}[m.group(1)]
|
||||
if re.search(r"funkciju (izpild|īstenoš)|nodrošinātu .*funkcij", l):
|
||||
return "task"
|
||||
for k, w in (("right", "tiesīb"), ("task", "uzdevum"), ("function", "funkcij")):
|
||||
if w in l:
|
||||
return k
|
||||
return "provision"
|
||||
|
||||
|
||||
def iso(d):
|
||||
m = re.match(r"(\d{2})\.(\d{2})\.(\d{4})", d or "")
|
||||
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
|
||||
|
||||
|
||||
SUP = str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹")
|
||||
|
||||
|
||||
def text_of(el):
|
||||
"""Visible text of an element: <sup> digits as superscript characters, <br> as space, notes kept out by the caller."""
|
||||
import copy
|
||||
e = copy.deepcopy(el)
|
||||
for s in e.iter("sup"):
|
||||
s.text = (s.text or "").translate(SUP)
|
||||
for b in e.iter("br"):
|
||||
b.tail = " " + (b.tail or "")
|
||||
return re.sub(r"\s+", " ", e.text_content().replace("\u00ad", "").replace("\u00a0", " ")).strip()
|
||||
|
||||
|
||||
def lines_of(el):
|
||||
import copy
|
||||
e = copy.deepcopy(el)
|
||||
for b in e.iter("br"):
|
||||
b.tail = "\n" + (b.tail or "")
|
||||
return [re.sub(r"\s+", " ", x).strip() for x in e.text_content().split("\n") if x.strip()]
|
||||
|
||||
|
||||
def block_lines(el):
|
||||
"""Text lines of an annex block; a table becomes one line per row with cells separated by ' | '."""
|
||||
rows = el.xpath(".//tr")
|
||||
if not rows:
|
||||
return [text_of(el)] if text_of(el) else []
|
||||
out = []
|
||||
for tr in rows:
|
||||
cells = [text_of(td) for td in tr.xpath("./td|./th")]
|
||||
if any(cells):
|
||||
out.append(" | ".join(cells))
|
||||
return out
|
||||
|
||||
|
||||
def parse(page):
|
||||
import lxml.html
|
||||
pase_txt = clean(re.sub(r"<script.*?</script>|<style.*?</style>", " ", page, flags=re.S))
|
||||
k = pase_txt.rfind("Tiesību akta pase")
|
||||
seg = pase_txt[k:k + 900]
|
||||
keys = ["Nosaukums", "Statuss", "Izdevējs", "Veids", "Numurs", "Pieņemts", "Stājas spēkā", "Zaudē spēku", "Publicēts"]
|
||||
|
||||
def f(name):
|
||||
m = re.search(re.escape(name) + r":\s*(.*?)\s*(?:" + "|".join(re.escape(x) + ":" for x in keys if x != name) + r"|Satura rādītājs|$)", seg)
|
||||
return m.group(1).strip() if m else None
|
||||
pase = {x: f(x) for x in keys}
|
||||
pase["Statuss"] = " ".join((pase["Statuss"] or "").split()[:2])
|
||||
if pase["Publicēts"]:
|
||||
pase["Publicēts"] = re.split(r"\s+(?:OP numurs|Dokumenta valoda|Saistītie dokumenti)", pase["Publicēts"])[0].strip()
|
||||
root = lxml.html.fromstring(page)
|
||||
t207 = root.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0]
|
||||
body = t207.getparent()
|
||||
head = title = basis = None
|
||||
chapters, cur, sigs, annexes, tail = [], None, [], [], []
|
||||
after_points = False
|
||||
for el in body:
|
||||
cls = (el.get("class") or "").split()
|
||||
c = cls[0] if cls else el.tag
|
||||
if c == "TV206":
|
||||
head = " ".join(lines_of(el))
|
||||
elif c == "TV207":
|
||||
title = text_of(el)
|
||||
elif c == "TV900" and not chapters:
|
||||
basis = text_of(el)
|
||||
elif sigs and c.startswith("TV") and text_of(el):
|
||||
# after the signatories: annexes (pielikumi) with their own headings, numbered rows and tables
|
||||
for t in block_lines(el):
|
||||
annexes.append(t)
|
||||
tail.append(t)
|
||||
elif c == "TV212":
|
||||
t = text_of(el)
|
||||
m = re.match(r"^(.*?)\s*(\((?:Nodaļas nosaukums|Nodaļa)[^()]*(?:\([^()]*\)[^()]*)*\))$", t)
|
||||
cur = {"title": m.group(1) if m else t, "note": m.group(2) if m else None, "points": []}
|
||||
chapters.append(cur)
|
||||
elif c == "TV213" and el.get("data-num"):
|
||||
if cur is None:
|
||||
cur = {"title": None, "note": None, "points": []}
|
||||
chapters.append(cur)
|
||||
notes = [text_of(p) for p in el.xpath(".//p[contains(@class,'labojumu_pamats')]")]
|
||||
paras = [text_of(p) for p in el.xpath("./p[contains(concat(' ', normalize-space(@class), ' '), ' TV213 ')]")]
|
||||
paras = [p for p in paras if p]
|
||||
if paras:
|
||||
cur["points"].append({"num": el.get("data-num"), "paras": paras, "notes": [n for n in notes if n]})
|
||||
after_points = True
|
||||
elif after_points and c in ("TV216", "TV217", "TV218", "TV219"):
|
||||
sigs += lines_of(el)
|
||||
tail += lines_of(el)
|
||||
elif after_points and c.startswith("TV") and text_of(el):
|
||||
annexes.append(text_of(el))
|
||||
tail.append(text_of(el))
|
||||
return pase, head, title, basis, chapters, sigs, annexes, tail
|
||||
|
||||
|
||||
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
||||
|
||||
|
||||
def split_num(text):
|
||||
"""Printed point number at the start of a paragraph → (normalised number, rest).
|
||||
"3.1. x" → 3.1; inserted points with superscript: "4.4.¹ x" and "4.4¹. x" → 4.4¹; "24.¹1. x" and "24.¹ 1. x" → 24¹.1"""
|
||||
m = re.match(r"^(\d+)", text)
|
||||
if not m:
|
||||
return None, text
|
||||
segs, i = [m.group(1)], m.end()
|
||||
while True:
|
||||
rest = text[i:]
|
||||
sup = re.match(r"^\.?([" + SUPC + r"]+)", rest)
|
||||
if sup and not segs[-1].endswith(tuple(SUPC)):
|
||||
segs[-1] += sup.group(1)
|
||||
i += sup.end()
|
||||
rest = text[i:]
|
||||
nxt = re.match(r"^\.?\s?(\d+)(?=[." + SUPC + r"])", rest) # 24¹.1 written as "24.¹1." or "24.¹ 1."
|
||||
if nxt:
|
||||
segs.append(nxt.group(1)); i += nxt.end()
|
||||
continue
|
||||
nxt = re.match(r"^\.(\d+)(?=[." + SUPC + r"])", rest)
|
||||
if nxt:
|
||||
segs.append(nxt.group(1)); i += nxt.end()
|
||||
continue
|
||||
break
|
||||
end = re.match(r"^\.\s*", text[i:]) or (re.match(r"^\s+", text[i:]) if segs[-1][-1] in SUPC else None)
|
||||
if not end:
|
||||
return None, text
|
||||
return ".".join(segs), text[i + end.end():]
|
||||
|
||||
|
||||
def build(page, org, inst_name, stem, likumi_id, retrieved):
|
||||
pase, head, title, basis, chapters, sigs, annexes, tail = parse(page)
|
||||
doc = f"lv-nol-{org}"
|
||||
# ---------------------------------------------------------------- text source
|
||||
lines = [x for x in (head, title, basis) if x]
|
||||
for ch in chapters:
|
||||
if ch["title"]:
|
||||
lines.append("")
|
||||
lines.append(ch["title"] + (" " + ch["note"] if ch.get("note") else ""))
|
||||
for p in ch["points"]:
|
||||
lines += p["paras"]
|
||||
lines += p["notes"]
|
||||
if tail: # signatories and annexes in document order
|
||||
lines.append("")
|
||||
lines += tail
|
||||
txt = "\n".join(lines).strip() + "\n"
|
||||
txt_rel = f"sources/{stem}.txt"
|
||||
os.makedirs(os.path.join(ROOT, "sources"), exist_ok=True)
|
||||
open(os.path.join(ROOT, txt_rel), "w", encoding="utf-8").write(txt)
|
||||
sha = hashlib.sha256(txt.encode("utf-8")).hexdigest()
|
||||
# ---------------------------------------------------------------- XML
|
||||
out = []
|
||||
for ci, ch in enumerate(chapters, 1):
|
||||
ck = chapter_kind(ch["title"]) if ch["title"] else "none"
|
||||
roman = re.match(r"^([IVXLC]+)\.\s", ch["title"] or "")
|
||||
parts = [f'<Chapter id="{doc}.c{ci}"{a("n", roman.group(1) if roman else None)} kind="{ck}">', el("Title", ch["title"]), el("Note", ch.get("note"))]
|
||||
for p in ch["points"]:
|
||||
n0, t0 = split_num(p["paras"][0]) if p["paras"] else (p["num"], "")
|
||||
n0 = n0 or p["num"]
|
||||
lead = t0.rstrip().endswith(":") and len(p["paras"]) > 1
|
||||
deleted = t0.strip().startswith("(svītrots")
|
||||
k0 = "deleted" if deleted else "lead" if lead else "provision"
|
||||
child_kind = lead_kind(t0) if (lead and ck == "competence") else "provision"
|
||||
# nested sub-points by printed number depth
|
||||
stack = [(n0, [])]
|
||||
node = {"n": n0, "text": t0, "kind": k0, "notes": p["notes"], "kids": []}
|
||||
path = [node]
|
||||
for para in p["paras"][1:]:
|
||||
n, t = split_num(para)
|
||||
if n is None: # continuation paragraph without its own number
|
||||
path[-1]["text"] += " " + para
|
||||
continue
|
||||
kd = "deleted" if t.strip().startswith("(svītrots") else ("lead" if t.rstrip().endswith(":") else child_kind)
|
||||
item = {"n": n, "text": t, "kind": kd, "notes": [], "kids": []}
|
||||
while len(path) > 1 and not n.startswith(path[-1]["n"] + "."):
|
||||
path.pop()
|
||||
path[-1]["kids"].append(item)
|
||||
path.append(item)
|
||||
|
||||
def render(x):
|
||||
pid = f"{doc}.p" + re.sub(r"([⁰¹²³⁴⁵⁶⁷⁸⁹]+)", lambda m: "s" + m.group(1).translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")), x["n"])
|
||||
return (f'<Point id="{pid}" n="{x["n"]}" kind="{x["kind"]}">' + el("Text", x["text"].strip() or "(tukšs)")
|
||||
+ "".join(el("Note", nn) for nn in x["notes"]) + "".join(render(k) for k in x["kids"]) + "</Point>")
|
||||
parts.append(render(node))
|
||||
parts.append("</Chapter>")
|
||||
out.append("".join(parts))
|
||||
sig_xml = "".join(el("Signatory", s, role="parakstītājs") for s in sigs)
|
||||
ann_xml = "".join(f'<Annex id="{doc}.a{i}">' + el("Title", t.split(" ", 6)[0:6] and " ".join(t.split(" ")[:6])) + el("Text", t) + "</Annex>"
|
||||
for i, t in enumerate(annexes, 1))
|
||||
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
|
||||
f'<InstitutionStatute xmlns="urn:pppa:vpk:nolikums:0.1" schemaVersion="0.1" id="{doc}">'
|
||||
+ el("Institution", inst_name, org=org)
|
||||
+ "<Act>" + el("Title", pase["Nosaukums"] or title) + el("Issuer", pase["Izdevējs"]) + el("Type", pase["Veids"])
|
||||
+ el("Number", pase["Numurs"]) + el("Adopted", iso(pase["Pieņemts"])) + el("InForce", iso(pase["Stājas spēkā"]))
|
||||
+ el("Published", pase["Publicēts"]) + el("Status", pase["Statuss"]) + el("LikumiId", str(likumi_id))
|
||||
+ el("Url", f"https://likumi.lv/ta/id/{likumi_id}") + "</Act>"
|
||||
+ "<Source>" + el("File", txt_rel) + el("SHA256", sha) + el("Retrieved", retrieved) + "</Source>"
|
||||
+ el("LegalBasis", basis) + "".join(out) + sig_xml + ann_xml
|
||||
+ f'<DataVersion number="1" date="{retrieved}">' + el("Change", "Pirmā versija no likumi.lv konsolidētās redakcijas.") + "</DataVersion>"
|
||||
+ "</InstitutionStatute>\n")
|
||||
from lxml import etree
|
||||
tree = etree.fromstring(xml.encode("utf-8"))
|
||||
etree.indent(tree, space=" ")
|
||||
os.makedirs(os.path.join(ROOT, "data"), exist_ok=True)
|
||||
with open(os.path.join(ROOT, "data", stem + ".xml"), "wb") as fh:
|
||||
fh.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
|
||||
return {"chapters": len(chapters), "points": sum(len(c["points"]) for c in chapters), "pase": pase}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pg = open(sys.argv[1], encoding="utf-8").read()
|
||||
print(build(pg, sys.argv[2], sys.argv[3], sys.argv[4], int(sys.argv[5]), sys.argv[6]))
|
||||
Reference in New Issue
Block a user