Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
276 lines
12 KiB
Python
276 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""likumi.lv konsolidētā redakcija (HTML) → sources/<fails>.txt un data/<fails>.xml (shēma nolikums-0.1).
|
|
|
|
python3 tools/convert.py <html> <VPK ID> <iestādes nosaukums> <datnes nosaukums bez paplašinājuma> <likumi.lv id> <ielādes datums>
|
|
|
|
HTML netiek glabāts repozitorijā: avots ir teksta datne, XML atbilst tai (tools/validate.py pārbauda).
|
|
"""
|
|
import hashlib
|
|
import html
|
|
import os
|
|
import re
|
|
import sys
|
|
from xml.sax.saxutils import escape, quoteattr
|
|
|
|
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
|
|
|
|
def clean(x):
|
|
x = re.sub(r"<br\s*/?>", " ", x or "")
|
|
x = re.sub(r"<[^>]+>", "", x)
|
|
return re.sub(r"\s+", " ", html.unescape(x)).replace(" ", " ").strip()
|
|
|
|
|
|
def a(k, v):
|
|
return f" {k}={quoteattr(str(v))}" if v not in (None, "") else ""
|
|
|
|
|
|
def el(tag, text, **at):
|
|
return f"<{tag}{''.join(a(k, v) for k, v in at.items())}>{escape(text)}</{tag}>" if text else ""
|
|
|
|
|
|
def chapter_kind(t):
|
|
t = (t or "").lower()
|
|
if "vispārīg" in t:
|
|
return "general"
|
|
if "noslēguma" in t or "pārejas" in t:
|
|
return "final"
|
|
if "padotībā" in t or "kapitāl" in t:
|
|
return "subordinates"
|
|
if "tiesiskum" in t or "pārskat" in t:
|
|
return "legality"
|
|
if "funkcij" in t or "uzdevum" in t or "kompetenc" in t or "tiesīb" in t:
|
|
return "competence"
|
|
if "pārvald" in t or "struktūr" in t or "amatpersonu" in t or "darba organiz" in t or "vadīb" in t:
|
|
return "management"
|
|
return "other"
|
|
|
|
|
|
def lead_kind(lead):
|
|
"""Kind of the items a lead sentence introduces: "… ir šādas funkcijas:" → function, "… šādi uzdevumi:" → task,
|
|
"… šādas tiesības:" → right; "Lai nodrošinātu funkciju izpildi, … :" → task."""
|
|
l = lead.lower()
|
|
m = re.search(r"šād(?:as|i|u)\s+(funkcij|uzdevum|tiesīb|kompetenc)", l)
|
|
if m:
|
|
return {"funkcij": "function", "uzdevum": "task", "tiesīb": "right", "kompetenc": "task"}[m.group(1)]
|
|
if re.search(r"funkciju (izpild|īstenoš)|nodrošinātu .*funkcij", l):
|
|
return "task"
|
|
for k, w in (("right", "tiesīb"), ("task", "uzdevum"), ("function", "funkcij")):
|
|
if w in l:
|
|
return k
|
|
return "provision"
|
|
|
|
|
|
def iso(d):
|
|
m = re.match(r"(\d{2})\.(\d{2})\.(\d{4})", d or "")
|
|
return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None
|
|
|
|
|
|
SUP = str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹")
|
|
|
|
|
|
def text_of(el):
|
|
"""Visible text of an element: <sup> digits as superscript characters, <br> as space, notes kept out by the caller."""
|
|
import copy
|
|
e = copy.deepcopy(el)
|
|
for s in e.iter("sup"):
|
|
s.text = (s.text or "").translate(SUP)
|
|
for b in e.iter("br"):
|
|
b.tail = " " + (b.tail or "")
|
|
return re.sub(r"\s+", " ", e.text_content().replace("\u00ad", "").replace("\u00a0", " ")).strip()
|
|
|
|
|
|
def lines_of(el):
|
|
import copy
|
|
e = copy.deepcopy(el)
|
|
for b in e.iter("br"):
|
|
b.tail = "\n" + (b.tail or "")
|
|
return [re.sub(r"\s+", " ", x).strip() for x in e.text_content().split("\n") if x.strip()]
|
|
|
|
|
|
def block_lines(el):
|
|
"""Text lines of an annex block; a table becomes one line per row with cells separated by ' | '."""
|
|
rows = el.xpath(".//tr")
|
|
if not rows:
|
|
return [text_of(el)] if text_of(el) else []
|
|
out = []
|
|
for tr in rows:
|
|
cells = [text_of(td) for td in tr.xpath("./td|./th")]
|
|
if any(cells):
|
|
out.append(" | ".join(cells))
|
|
return out
|
|
|
|
|
|
def parse(page):
|
|
import lxml.html
|
|
pase_txt = clean(re.sub(r"<script.*?</script>|<style.*?</style>", " ", page, flags=re.S))
|
|
k = pase_txt.rfind("Tiesību akta pase")
|
|
seg = pase_txt[k:k + 900]
|
|
keys = ["Nosaukums", "Statuss", "Izdevējs", "Atbildīgā iestāde", "Veids", "Numurs", "Pieņemts", "Stājas spēkā", "Zaudē spēku", "Publicēts"]
|
|
|
|
def f(name):
|
|
m = re.search(re.escape(name) + r":\s*(.*?)\s*(?:" + "|".join(re.escape(x) + ":" for x in keys if x != name) + r"|Satura rādītājs|$)", seg)
|
|
return m.group(1).strip() if m else None
|
|
pase = {x: f(x) for x in keys}
|
|
pase["Statuss"] = " ".join((pase["Statuss"] or "").split()[:2])
|
|
if pase["Publicēts"]:
|
|
pase["Publicēts"] = re.split(r"\s+(?:OP numurs|Dokumenta valoda|Saistītie dokumenti)", pase["Publicēts"])[0].strip()
|
|
root = lxml.html.fromstring(page)
|
|
t207 = root.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0]
|
|
body = t207.getparent()
|
|
head = title = basis = None
|
|
chapters, cur, sigs, annexes, tail = [], None, [], [], []
|
|
after_points = False
|
|
for el in body:
|
|
cls = (el.get("class") or "").split()
|
|
c = cls[0] if cls else el.tag
|
|
if c == "TV206":
|
|
head = " ".join(lines_of(el))
|
|
elif c == "TV207":
|
|
title = text_of(el)
|
|
elif c == "TV900" and not chapters:
|
|
basis = text_of(el)
|
|
elif sigs and c.startswith("TV") and text_of(el):
|
|
# after the signatories: annexes (pielikumi) with their own headings, numbered rows and tables
|
|
for t in block_lines(el):
|
|
annexes.append(t)
|
|
tail.append(t)
|
|
elif c == "TV212":
|
|
t = text_of(el)
|
|
m = re.match(r"^(.*?)\s*(\((?:Nodaļas nosaukums|Nodaļa)[^()]*(?:\([^()]*\)[^()]*)*\))$", t)
|
|
cur = {"title": m.group(1) if m else t, "note": m.group(2) if m else None, "points": []}
|
|
chapters.append(cur)
|
|
elif c == "TV213" and el.get("data-num"):
|
|
if cur is None:
|
|
cur = {"title": None, "note": None, "points": []}
|
|
chapters.append(cur)
|
|
notes = [text_of(p) for p in el.xpath(".//p[contains(@class,'labojumu_pamats')]")]
|
|
paras = [text_of(p) for p in el.xpath("./p[contains(concat(' ', normalize-space(@class), ' '), ' TV213 ')]")]
|
|
paras = [p for p in paras if p]
|
|
if paras:
|
|
cur["points"].append({"num": el.get("data-num"), "paras": paras, "notes": [n for n in notes if n]})
|
|
after_points = True
|
|
elif after_points and c in ("TV216", "TV217", "TV218", "TV219"):
|
|
sigs += lines_of(el)
|
|
tail += lines_of(el)
|
|
elif after_points and c.startswith("TV") and text_of(el):
|
|
annexes.append(text_of(el))
|
|
tail.append(text_of(el))
|
|
return pase, head, title, basis, chapters, sigs, annexes, tail
|
|
|
|
|
|
SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹"
|
|
|
|
|
|
def split_num(text):
|
|
"""Printed point number at the start of a paragraph → (normalised number, rest).
|
|
"3.1. x" → 3.1; inserted points with superscript: "4.4.¹ x" and "4.4¹. x" → 4.4¹; "24.¹1. x" and "24.¹ 1. x" → 24¹.1"""
|
|
m = re.match(r"^(\d+)", text)
|
|
if not m:
|
|
return None, text
|
|
segs, i = [m.group(1)], m.end()
|
|
while True:
|
|
rest = text[i:]
|
|
sup = re.match(r"^\.?([" + SUPC + r"]+)", rest)
|
|
if sup and not segs[-1].endswith(tuple(SUPC)):
|
|
segs[-1] += sup.group(1)
|
|
i += sup.end()
|
|
rest = text[i:]
|
|
nxt = re.match(r"^\.?\s?(\d+)(?=[." + SUPC + r"])", rest) # 24¹.1 written as "24.¹1." or "24.¹ 1."
|
|
if nxt:
|
|
segs.append(nxt.group(1)); i += nxt.end()
|
|
continue
|
|
nxt = re.match(r"^\.(\d+)(?=[." + SUPC + r"])", rest)
|
|
if nxt:
|
|
segs.append(nxt.group(1)); i += nxt.end()
|
|
continue
|
|
break
|
|
end = re.match(r"^\.\s*", text[i:]) or (re.match(r"^\s+", text[i:]) if segs[-1][-1] in SUPC else None)
|
|
if not end:
|
|
return None, text
|
|
return ".".join(segs), text[i + end.end():]
|
|
|
|
|
|
def build(page, org, inst_name, stem, likumi_id, retrieved):
|
|
pase, head, title, basis, chapters, sigs, annexes, tail = parse(page)
|
|
doc = f"lv-nol-{org}"
|
|
# ---------------------------------------------------------------- text source
|
|
lines = [x for x in (head, title, basis) if x]
|
|
for ch in chapters:
|
|
if ch["title"]:
|
|
lines.append("")
|
|
lines.append(ch["title"] + (" " + ch["note"] if ch.get("note") else ""))
|
|
for p in ch["points"]:
|
|
lines += p["paras"]
|
|
lines += p["notes"]
|
|
if tail: # signatories and annexes in document order
|
|
lines.append("")
|
|
lines += tail
|
|
txt = "\n".join(lines).strip() + "\n"
|
|
txt_rel = f"sources/{stem}.txt"
|
|
os.makedirs(os.path.join(ROOT, "sources"), exist_ok=True)
|
|
open(os.path.join(ROOT, txt_rel), "w", encoding="utf-8").write(txt)
|
|
sha = hashlib.sha256(txt.encode("utf-8")).hexdigest()
|
|
# ---------------------------------------------------------------- XML
|
|
out = []
|
|
for ci, ch in enumerate(chapters, 1):
|
|
ck = chapter_kind(ch["title"]) if ch["title"] else "none"
|
|
roman = re.match(r"^([IVXLC]+)\.\s", ch["title"] or "")
|
|
parts = [f'<Chapter id="{doc}.c{ci}"{a("n", roman.group(1) if roman else None)} kind="{ck}">', el("Title", ch["title"]), el("Note", ch.get("note"))]
|
|
for p in ch["points"]:
|
|
n0, t0 = split_num(p["paras"][0]) if p["paras"] else (p["num"], "")
|
|
n0 = n0 or p["num"]
|
|
lead = t0.rstrip().endswith(":") and len(p["paras"]) > 1
|
|
deleted = t0.strip().startswith("(svītrots")
|
|
k0 = "deleted" if deleted else "lead" if lead else "provision"
|
|
child_kind = lead_kind(t0) if (lead and ck == "competence") else "provision"
|
|
# nested sub-points by printed number depth
|
|
stack = [(n0, [])]
|
|
node = {"n": n0, "text": t0, "kind": k0, "notes": p["notes"], "kids": []}
|
|
path = [node]
|
|
for para in p["paras"][1:]:
|
|
n, t = split_num(para)
|
|
if n is None: # continuation paragraph without its own number
|
|
path[-1]["text"] += " " + para
|
|
continue
|
|
kd = "deleted" if t.strip().startswith("(svītrots") else ("lead" if t.rstrip().endswith(":") else child_kind)
|
|
item = {"n": n, "text": t, "kind": kd, "notes": [], "kids": []}
|
|
while len(path) > 1 and not n.startswith(path[-1]["n"] + "."):
|
|
path.pop()
|
|
path[-1]["kids"].append(item)
|
|
path.append(item)
|
|
|
|
def render(x):
|
|
pid = f"{doc}.p" + re.sub(r"([⁰¹²³⁴⁵⁶⁷⁸⁹]+)", lambda m: "s" + m.group(1).translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")), x["n"])
|
|
return (f'<Point id="{pid}" n="{x["n"]}" kind="{x["kind"]}">' + el("Text", x["text"].strip() or "(tukšs)")
|
|
+ "".join(el("Note", nn) for nn in x["notes"]) + "".join(render(k) for k in x["kids"]) + "</Point>")
|
|
parts.append(render(node))
|
|
parts.append("</Chapter>")
|
|
out.append("".join(parts))
|
|
sig_xml = "".join(el("Signatory", s, role="parakstītājs") for s in sigs)
|
|
ann_xml = "".join(f'<Annex id="{doc}.a{i}">' + el("Title", t.split(" ", 6)[0:6] and " ".join(t.split(" ")[:6])) + el("Text", t) + "</Annex>"
|
|
for i, t in enumerate(annexes, 1))
|
|
xml = ('<?xml version="1.0" encoding="UTF-8"?>\n'
|
|
f'<InstitutionStatute xmlns="urn:pppa:vpk:nolikums:0.1" schemaVersion="0.1" id="{doc}">'
|
|
+ el("Institution", inst_name, org=org)
|
|
+ "<Act>" + el("Title", pase["Nosaukums"] or title) + el("Issuer", pase["Izdevējs"]) + el("ResponsibleInstitution", pase["Atbildīgā iestāde"]) + el("Type", pase["Veids"])
|
|
+ el("Number", pase["Numurs"]) + el("Adopted", iso(pase["Pieņemts"])) + el("InForce", iso(pase["Stājas spēkā"]))
|
|
+ el("Published", pase["Publicēts"]) + el("Status", pase["Statuss"]) + el("LikumiId", str(likumi_id))
|
|
+ el("Url", f"https://likumi.lv/ta/id/{likumi_id}") + "</Act>"
|
|
+ "<Source>" + el("File", txt_rel) + el("SHA256", sha) + el("Retrieved", retrieved) + "</Source>"
|
|
+ el("LegalBasis", basis) + "".join(out) + sig_xml + ann_xml
|
|
+ f'<DataVersion number="1" date="{retrieved}">' + el("Change", "Pirmā versija no likumi.lv konsolidētās redakcijas.") + "</DataVersion>"
|
|
+ "</InstitutionStatute>\n")
|
|
from lxml import etree
|
|
tree = etree.fromstring(xml.encode("utf-8"))
|
|
etree.indent(tree, space=" ")
|
|
os.makedirs(os.path.join(ROOT, "data"), exist_ok=True)
|
|
with open(os.path.join(ROOT, "data", stem + ".xml"), "wb") as fh:
|
|
fh.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True))
|
|
return {"chapters": len(chapters), "points": sum(len(c["points"]) for c in chapters), "pase": pase}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
pg = open(sys.argv[1], encoding="utf-8").read()
|
|
print(build(pg, sys.argv[2], sys.argv[3], sys.argv[4], int(sys.argv[5]), sys.argv[6]))
|