#!/usr/bin/env python3 """likumi.lv konsolidētā redakcija (HTML) → sources/.txt un data/.xml (shēma nolikums-0.1). python3 tools/convert.py HTML netiek glabāts repozitorijā: avots ir teksta datne, XML atbilst tai (tools/validate.py pārbauda). """ import hashlib import html import os import re import sys from xml.sax.saxutils import escape, quoteattr ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) def clean(x): x = re.sub(r"", " ", x or "") x = re.sub(r"<[^>]+>", "", x) return re.sub(r"\s+", " ", html.unescape(x)).replace(" ", " ").strip() def a(k, v): return f" {k}={quoteattr(str(v))}" if v not in (None, "") else "" def el(tag, text, **at): return f"<{tag}{''.join(a(k, v) for k, v in at.items())}>{escape(text)}" if text else "" def chapter_kind(t): t = (t or "").lower() if "vispārīg" in t: return "general" if "noslēguma" in t or "pārejas" in t: return "final" if "padotībā" in t or "kapitāl" in t: return "subordinates" if "tiesiskum" in t or "pārskat" in t: return "legality" if "funkcij" in t or "uzdevum" in t or "kompetenc" in t or "tiesīb" in t: return "competence" if "pārvald" in t or "struktūr" in t or "amatpersonu" in t or "darba organiz" in t or "vadīb" in t: return "management" return "other" def lead_kind(lead): """Kind of the items a lead sentence introduces: "… ir šādas funkcijas:" → function, "… šādi uzdevumi:" → task, "… šādas tiesības:" → right; "Lai nodrošinātu funkciju izpildi, … :" → task.""" l = lead.lower() m = re.search(r"šād(?:as|i|u)\s+(funkcij|uzdevum|tiesīb|kompetenc)", l) if m: return {"funkcij": "function", "uzdevum": "task", "tiesīb": "right", "kompetenc": "task"}[m.group(1)] if re.search(r"funkciju (izpild|īstenoš)|nodrošinātu .*funkcij", l): return "task" for k, w in (("right", "tiesīb"), ("task", "uzdevum"), ("function", "funkcij")): if w in l: return k return "provision" def iso(d): m = re.match(r"(\d{2})\.(\d{2})\.(\d{4})", d or "") return f"{m.group(3)}-{m.group(2)}-{m.group(1)}" if m else None SUP = str.maketrans("0123456789", "⁰¹²³⁴⁵⁶⁷⁸⁹") def text_of(el): """Visible text of an element: digits as superscript characters,
as space, notes kept out by the caller.""" import copy e = copy.deepcopy(el) for s in e.iter("sup"): s.text = (s.text or "").translate(SUP) for b in e.iter("br"): b.tail = " " + (b.tail or "") return re.sub(r"\s+", " ", e.text_content().replace("\u00ad", "").replace("\u00a0", " ")).strip() def lines_of(el): import copy e = copy.deepcopy(el) for b in e.iter("br"): b.tail = "\n" + (b.tail or "") return [re.sub(r"\s+", " ", x).strip() for x in e.text_content().split("\n") if x.strip()] def block_lines(el): """Text lines of an annex block; a table becomes one line per row with cells separated by ' | '.""" rows = el.xpath(".//tr") if not rows: return [text_of(el)] if text_of(el) else [] out = [] for tr in rows: cells = [text_of(td) for td in tr.xpath("./td|./th")] if any(cells): out.append(" | ".join(cells)) return out def parse(page): import lxml.html pase_txt = clean(re.sub(r"|", " ", page, flags=re.S)) k = pase_txt.rfind("Tiesību akta pase") seg = pase_txt[k:k + 900] keys = ["Nosaukums", "Statuss", "Izdevējs", "Atbildīgā iestāde", "Veids", "Numurs", "Pieņemts", "Stājas spēkā", "Zaudē spēku", "Publicēts"] def f(name): m = re.search(re.escape(name) + r":\s*(.*?)\s*(?:" + "|".join(re.escape(x) + ":" for x in keys if x != name) + r"|Satura rādītājs|$)", seg) return m.group(1).strip() if m else None pase = {x: f(x) for x in keys} pase["Statuss"] = " ".join((pase["Statuss"] or "").split()[:2]) if pase["Publicēts"]: pase["Publicēts"] = re.split(r"\s+(?:OP numurs|Dokumenta valoda|Saistītie dokumenti)", pase["Publicēts"])[0].strip() root = lxml.html.fromstring(page) t207 = root.xpath("//div[contains(concat(' ', normalize-space(@class), ' '), ' TV207 ')]")[0] body = t207.getparent() head = title = basis = None chapters, cur, sigs, annexes, tail = [], None, [], [], [] after_points = False for el in body: cls = (el.get("class") or "").split() c = cls[0] if cls else el.tag if c == "TV206": head = " ".join(lines_of(el)) elif c == "TV207": title = text_of(el) elif c == "TV900" and not chapters: basis = text_of(el) elif sigs and c.startswith("TV") and text_of(el): # after the signatories: annexes (pielikumi) with their own headings, numbered rows and tables for t in block_lines(el): annexes.append(t) tail.append(t) elif c == "TV212": t = text_of(el) m = re.match(r"^(.*?)\s*(\((?:Nodaļas nosaukums|Nodaļa)[^()]*(?:\([^()]*\)[^()]*)*\))$", t) cur = {"title": m.group(1) if m else t, "note": m.group(2) if m else None, "points": []} chapters.append(cur) elif c == "TV213" and el.get("data-num"): if cur is None: cur = {"title": None, "note": None, "points": []} chapters.append(cur) notes = [text_of(p) for p in el.xpath(".//p[contains(@class,'labojumu_pamats')]")] paras = [text_of(p) for p in el.xpath("./p[contains(concat(' ', normalize-space(@class), ' '), ' TV213 ')]")] paras = [p for p in paras if p] if paras: cur["points"].append({"num": el.get("data-num"), "paras": paras, "notes": [n for n in notes if n]}) after_points = True elif after_points and c in ("TV216", "TV217", "TV218", "TV219"): sigs += lines_of(el) tail += lines_of(el) elif after_points and c.startswith("TV") and text_of(el): annexes.append(text_of(el)) tail.append(text_of(el)) return pase, head, title, basis, chapters, sigs, annexes, tail SUPC = "⁰¹²³⁴⁵⁶⁷⁸⁹" def split_num(text): """Printed point number at the start of a paragraph → (normalised number, rest). "3.1. x" → 3.1; inserted points with superscript: "4.4.¹ x" and "4.4¹. x" → 4.4¹; "24.¹1. x" and "24.¹ 1. x" → 24¹.1""" m = re.match(r"^(\d+)", text) if not m: return None, text segs, i = [m.group(1)], m.end() while True: rest = text[i:] sup = re.match(r"^\.?([" + SUPC + r"]+)", rest) if sup and not segs[-1].endswith(tuple(SUPC)): segs[-1] += sup.group(1) i += sup.end() rest = text[i:] nxt = re.match(r"^\.?\s?(\d+)(?=[." + SUPC + r"])", rest) # 24¹.1 written as "24.¹1." or "24.¹ 1." if nxt: segs.append(nxt.group(1)); i += nxt.end() continue nxt = re.match(r"^\.(\d+)(?=[." + SUPC + r"])", rest) if nxt: segs.append(nxt.group(1)); i += nxt.end() continue break end = re.match(r"^\.\s*", text[i:]) or (re.match(r"^\s+", text[i:]) if segs[-1][-1] in SUPC else None) if not end: return None, text return ".".join(segs), text[i + end.end():] def build(page, org, inst_name, stem, likumi_id, retrieved): pase, head, title, basis, chapters, sigs, annexes, tail = parse(page) doc = f"lv-nol-{org}" # ---------------------------------------------------------------- text source lines = [x for x in (head, title, basis) if x] for ch in chapters: if ch["title"]: lines.append("") lines.append(ch["title"] + (" " + ch["note"] if ch.get("note") else "")) for p in ch["points"]: lines += p["paras"] lines += p["notes"] if tail: # signatories and annexes in document order lines.append("") lines += tail txt = "\n".join(lines).strip() + "\n" txt_rel = f"sources/{stem}.txt" os.makedirs(os.path.join(ROOT, "sources"), exist_ok=True) open(os.path.join(ROOT, txt_rel), "w", encoding="utf-8").write(txt) sha = hashlib.sha256(txt.encode("utf-8")).hexdigest() # ---------------------------------------------------------------- XML out = [] for ci, ch in enumerate(chapters, 1): ck = chapter_kind(ch["title"]) if ch["title"] else "none" roman = re.match(r"^([IVXLC]+)\.\s", ch["title"] or "") parts = [f'', el("Title", ch["title"]), el("Note", ch.get("note"))] for p in ch["points"]: n0, t0 = split_num(p["paras"][0]) if p["paras"] else (p["num"], "") n0 = n0 or p["num"] lead = t0.rstrip().endswith(":") and len(p["paras"]) > 1 deleted = t0.strip().startswith("(svītrots") k0 = "deleted" if deleted else "lead" if lead else "provision" child_kind = lead_kind(t0) if (lead and ck == "competence") else "provision" # nested sub-points by printed number depth stack = [(n0, [])] node = {"n": n0, "text": t0, "kind": k0, "notes": p["notes"], "kids": []} path = [node] for para in p["paras"][1:]: n, t = split_num(para) if n is None: # continuation paragraph without its own number path[-1]["text"] += " " + para continue kd = "deleted" if t.strip().startswith("(svītrots") else ("lead" if t.rstrip().endswith(":") else child_kind) item = {"n": n, "text": t, "kind": kd, "notes": [], "kids": []} while len(path) > 1 and not n.startswith(path[-1]["n"] + "."): path.pop() path[-1]["kids"].append(item) path.append(item) def render(x): pid = f"{doc}.p" + re.sub(r"([⁰¹²³⁴⁵⁶⁷⁸⁹]+)", lambda m: "s" + m.group(1).translate(str.maketrans("⁰¹²³⁴⁵⁶⁷⁸⁹", "0123456789")), x["n"]) return (f'' + el("Text", x["text"].strip() or "(tukšs)") + "".join(el("Note", nn) for nn in x["notes"]) + "".join(render(k) for k in x["kids"]) + "") parts.append(render(node)) parts.append("") out.append("".join(parts)) sig_xml = "".join(el("Signatory", s, role="parakstītājs") for s in sigs) ann_xml = "".join(f'' + el("Title", t.split(" ", 6)[0:6] and " ".join(t.split(" ")[:6])) + el("Text", t) + "" for i, t in enumerate(annexes, 1)) xml = ('\n' f'' + el("Institution", inst_name, org=org) + "" + el("Title", pase["Nosaukums"] or title) + el("Issuer", pase["Izdevējs"]) + el("ResponsibleInstitution", pase["Atbildīgā iestāde"]) + el("Type", pase["Veids"]) + el("Number", pase["Numurs"]) + el("Adopted", iso(pase["Pieņemts"])) + el("InForce", iso(pase["Stājas spēkā"])) + el("Published", pase["Publicēts"]) + el("Status", pase["Statuss"]) + el("LikumiId", str(likumi_id)) + el("Url", f"https://likumi.lv/ta/id/{likumi_id}") + "" + "" + el("File", txt_rel) + el("SHA256", sha) + el("Retrieved", retrieved) + "" + el("LegalBasis", basis) + "".join(out) + sig_xml + ann_xml + f'' + el("Change", "Pirmā versija no likumi.lv konsolidētās redakcijas.") + "" + "\n") from lxml import etree tree = etree.fromstring(xml.encode("utf-8")) etree.indent(tree, space=" ") os.makedirs(os.path.join(ROOT, "data"), exist_ok=True) with open(os.path.join(ROOT, "data", stem + ".xml"), "wb") as fh: fh.write(etree.tostring(tree, xml_declaration=True, encoding="UTF-8", pretty_print=True)) return {"chapters": len(chapters), "points": sum(len(c["points"]) for c in chapters), "pase": pase} if __name__ == "__main__": pg = open(sys.argv[1], encoding="utf-8").read() print(build(pg, sys.argv[2], sys.argv[3], sys.argv[4], int(sys.argv[5]), sys.argv[6]))