475 numurētie punkti (mērķi, 131 indikators, 124 uzdevumi ar VPK ID, telpiskās attīstības virzieni), 154 pamatojuma ieraksti, 34 zemsvītras piezīmes. Pārbaude pret avota PDF ar citu nolasītāju — izturēta. Katalogs planosanas-dokumenti.yaml, rīki parse/build/verify/validate. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016679RwmHsuTFfxt26wP6rk
614 lines
29 KiB
Python
614 lines
29 KiB
Python
#!/usr/bin/env python3
|
|
"""NAP2027 PDF → strukturēts JSON (starpposms; XML veido build_xml.py).
|
|
|
|
python3 tools/parse_nap.py sources/nap2027/NAP2027.pdf build/nap2027.json
|
|
|
|
Lasa vārdus ar koordinātām (pdfplumber), noņem lapu galvenes un kājenes, atdala zemsvītras piezīmes,
|
|
atpazīst nodaļas, numurētos punktus [1]–[475], indikatoru un uzdevumu tabulas (ailes pēc galvenes koordinātām)
|
|
un pielikuma pamatojuma punktus.
|
|
"""
|
|
import json
|
|
import re
|
|
import sys
|
|
from collections import defaultdict
|
|
|
|
import pdfplumber
|
|
|
|
HEAD_TOP, FOOT_TOP = 52, 775
|
|
MARK = re.compile(r"^\[(\d{1,3})\]$")
|
|
|
|
|
|
def norm(s):
|
|
s = re.sub(r"\s+", " ", s or "").strip()
|
|
s = re.sub(r"\s+([,.;:)”])", r"\1", s)
|
|
s = re.sub(r"([“(])\s+", r"\1", s)
|
|
return repair(s)
|
|
|
|
|
|
from collections import Counter as _Counter
|
|
VOCAB = _Counter()
|
|
|
|
|
|
def load_vocab(pdf_path):
|
|
"""Word-form frequencies as pdftotext reads the body text — used to repair words split by kerning or cell hyphenation."""
|
|
import subprocess
|
|
raw = subprocess.run(["pdftotext", pdf_path, "-"], capture_output=True, text=True).stdout
|
|
VOCAB.update(w.strip(".,;:()“”\"'") for w in raw.split())
|
|
|
|
|
|
def _better(whole, *parts):
|
|
"""Join when the whole word is attested more often than any of its fragments."""
|
|
n = VOCAB.get(whole, 0) + VOCAB.get(whole.lower(), 0)
|
|
return n >= 2 and any(VOCAB.get(p, 0) < n for p in parts)
|
|
|
|
|
|
def repair(s):
|
|
if not s or not VOCAB:
|
|
return s
|
|
toks, out = s.split(" "), []
|
|
for t in toks:
|
|
if out:
|
|
a, core = out[-1], t.rstrip(".,;:)”")
|
|
ca = a.lstrip("“(")
|
|
if "/" in a or "http" in a or "www." in a:
|
|
out.append(t)
|
|
continue
|
|
# three fragments "fundamentāl a s"
|
|
if len(out) >= 2 and len(a) <= 2 and a.isalpha() and core[:1].islower() and \
|
|
_better(out[-2].lstrip("“(") + a + core, out[-2].lstrip("“("), a, core):
|
|
out.pop()
|
|
out[-1] = out[-1] + a + t
|
|
continue
|
|
# "paš - valdības"
|
|
if a == "-" and len(out) >= 2 and core[:1].islower() and _better(out[-2].lstrip("“(") + core, out[-2].lstrip("“("), core):
|
|
out.pop()
|
|
out[-1] = out[-1] + t
|
|
continue
|
|
# "aizsar- dzības"
|
|
if ca.endswith("-") and len(ca) > 2 and ca[-2].islower() and core[:1].islower() and \
|
|
(_better(ca[:-1] + core, ca[:-1], core) or VOCAB.get(core, 0) == 0):
|
|
out[-1] = a[:-1] + t
|
|
continue
|
|
# dropped capital "O glekļa", kerning "pašvaldī bas"
|
|
if ca[-1:].isalpha() and core[:1].isalpha() and core[:1].islower() and \
|
|
(_better(ca + core, ca, core) or (len(ca) == 1 and ca.isupper() and VOCAB.get((ca + core).lower(), 0) >= 2) or
|
|
(len(ca) == 1 and ca.isupper() and VOCAB.get(ca + core, 0) >= 1 and VOCAB.get(core, 0) == 0)):
|
|
out[-1] = a + t
|
|
continue
|
|
out.append(t)
|
|
return " ".join(out)
|
|
|
|
|
|
def fix_urls(s):
|
|
"""A URL broken across lines: rejoin the piece after - / _ . = ? & %."""
|
|
prev = None
|
|
while prev != s:
|
|
prev = s
|
|
s = re.sub(r"((?:https?://|www\.)\S*[-/_.%=?&])\s+(?=[\w%])", r"\1", s)
|
|
return s
|
|
|
|
|
|
def squash(s):
|
|
return re.sub(r"\s+", "", s or "")
|
|
|
|
|
|
def page_lines(pdf):
|
|
"""-> list of lines: {page, top, x0, words:[...], text, size, bold}; footnotes separately."""
|
|
out, notes = [], []
|
|
for pi, p in enumerate(pdf.pages):
|
|
ws = p.extract_words(extra_attrs=["size", "fontname"], keep_blank_chars=False)
|
|
ws = [w for w in ws if HEAD_TOP < w["top"] < FOOT_TOP]
|
|
# footnotes: small text at the bottom of the page
|
|
marks = [w["top"] for w in ws if w["size"] < 6.5 and w["text"].isdigit() and w["x0"] < 90 and w["top"] > 550]
|
|
fz = min(marks) - 1.5 if marks else 9999
|
|
infoot = lambda w: w["size"] < 9.5 and w["top"] >= fz
|
|
foot = [w for w in ws if infoot(w)]
|
|
body = [w for w in ws if not infoot(w) and w["size"] >= 7.5]
|
|
smalls = [w for w in ws if not infoot(w) and w["size"] < 7.5]
|
|
if foot:
|
|
fsub = [w for w in foot if w["size"] < 6.5 and w["x0"] >= 90]
|
|
foot = [dict(w) for w in foot if not (w["size"] < 6.5 and w["x0"] >= 90)]
|
|
for sw in fsub:
|
|
near = [m for m in foot if abs(m["top"] - sw["top"]) < 8 and -0.5 <= sw["x0"] - m["x1"] < 3]
|
|
if near:
|
|
m = min(near, key=lambda m: sw["x0"] - m["x1"])
|
|
m["text"] += sw["text"].translate(str.maketrans("0123456789", "₀₁₂₃₄₅₆₇₈₉"))
|
|
foot.sort(key=lambda w: (w["top"], w["x0"]))
|
|
rows_f = []
|
|
for w in foot:
|
|
if rows_f and abs(w["top"] - rows_f[-1][0]["top"]) < 2.5:
|
|
rows_f[-1].append(w)
|
|
else:
|
|
rows_f.append([w])
|
|
cur = None
|
|
for r in rows_f:
|
|
r.sort(key=lambda w: w["x0"])
|
|
if r[0]["size"] < 6.5 and r[0]["text"].isdigit() and r[0]["x0"] < 90:
|
|
cur = {"page": pi + 1, "number": int(r[0]["text"]), "text": " ".join(w["text"] for w in r[1:])}
|
|
notes.append(cur)
|
|
elif cur:
|
|
cur["text"] += " " + " ".join(w["text"] for w in r)
|
|
# merge glyphs split by kerning on one line
|
|
body.sort(key=lambda w: (round(w["top"] / 2.5), w["x0"]))
|
|
merged = []
|
|
for w in body:
|
|
if merged and abs(merged[-1]["top"] - w["top"]) < 2.5 and 0 <= w["x0"] - merged[-1]["x1"] < 0.9:
|
|
m = merged[-1]
|
|
m["text"] += w["text"]
|
|
m["x1"] = w["x1"]
|
|
m["bold"] = m["bold"] and "Bold" in w["fontname"]
|
|
else:
|
|
merged.append(dict(text=w["text"], x0=w["x0"], x1=w["x1"], top=w["top"], size=w["size"], page=pi + 1,
|
|
bold="Bold" in w["fontname"], italic="Italic" in w["fontname"]))
|
|
split = []
|
|
for w in merged:
|
|
m = re.match(r"^(\[\d{1,3}\])(.+)$", w["text"])
|
|
if m:
|
|
split.append(dict(w, text=m.group(1), x1=w["x0"] + 20))
|
|
split.append(dict(w, text=m.group(2), x0=w["x0"] + 20.5))
|
|
else:
|
|
split.append(w)
|
|
merged = split
|
|
# footnote numbers glued to a word at the same font size ("attīstībā16.")
|
|
page_notes = {n["number"] for n in notes if n["page"] == pi + 1}
|
|
for m_ in merged:
|
|
g = re.match(r"^(.*[a-zāčēģīķļņšūž”)])(\d{1,2})([.,;:]?)$", m_["text"])
|
|
if g and int(g.group(2)) in page_notes:
|
|
m_["text"] = g.group(1) + g.group(3)
|
|
m_.setdefault("refs", []).append(int(g.group(2)))
|
|
# small glyphs: subscripts (CO₂) join the word; superscript digits are footnote references
|
|
SUBS = str.maketrans("0123456789", "₀₁₂₃₄₅₆₇₈₉")
|
|
for sw in smalls:
|
|
near = [m for m in merged if abs(m["top"] - sw["top"]) < 8 and -0.5 <= sw["x0"] - m["x1"] < 3]
|
|
if not near:
|
|
continue
|
|
m = min(near, key=lambda m: sw["x0"] - m["x1"])
|
|
if sw["top"] > m["top"] + 1.5:
|
|
m["text"] += sw["text"].translate(SUBS)
|
|
m["x1"] = sw["x1"]
|
|
elif sw["text"].isdigit():
|
|
m.setdefault("refs", []).append(int(sw["text"]))
|
|
rows = defaultdict(list)
|
|
for w in merged:
|
|
rows[round(w["top"] / 2.5)].append(w)
|
|
keys = sorted(rows)
|
|
# join rows whose tops are within 2.5 pt (rounding boundary)
|
|
grouped = []
|
|
for k in keys:
|
|
if grouped and abs(rows[k][0]["top"] - grouped[-1][0]["top"]) < 2.6:
|
|
grouped[-1].extend(rows[k])
|
|
else:
|
|
grouped.append(list(rows[k]))
|
|
for g in grouped:
|
|
g.sort(key=lambda w: w["x0"])
|
|
out.append({"page": pi + 1, "top": min(w["top"] for w in g), "x0": g[0]["x0"], "words": g,
|
|
"text": " ".join(w["text"] for w in g), "size": max(w["size"] for w in g),
|
|
"bold": all(w["bold"] for w in g)})
|
|
return out, notes
|
|
|
|
|
|
SECTION_HEADS = [
|
|
("introduction", "IEVADS"), ("vision", "VĪZIJA PAR LATVIJAS NĀKOTNI 2027. GADĀ"), ("framework", "NAP2027 IETVARS"),
|
|
("strategicGoals", "NAP2027 STRATĒĢISKIE MĒRĶI"), ("spatial", "NAP2027 telpiskās attīstības perspektīva"),
|
|
("implementation", "NAP2027 īstenošanas, finansēšanas, uzraudzības un novērtēšanas process"), ("annex", "Pielikums"),
|
|
]
|
|
SUB = {
|
|
"PRIORITĀTESMĒRĶIS": "priorityGoal", "RĪCĪBASVIRZIENAMĒRĶIS": "actionLineGoal", "RĪCĪBASVIRZIENAMĒRĶI": "actionLineGoal",
|
|
"Stratēģiskomērķuindikatori": "indicators", "Rīcībasvirzienamērķaindikatori": "indicators",
|
|
"Rīcībasvirzienamērķuindikatori": "indicators", "Rīcībasvirzienauzdevumi": "tasks",
|
|
}
|
|
IND_COLS = ["no", "name", "unit", "baseYear", "baseValue", "target2024", "target2027", "source"]
|
|
TASK_COLS = ["no", "text", "responsible", "coResponsible", "funding", "indicators"]
|
|
|
|
|
|
def header_centers(lines, kind):
|
|
"""Column centres from the header words of one table (the lines between the table title and the first row)."""
|
|
ws = [w for ln in lines for w in ln["words"]]
|
|
def c(w):
|
|
return (w["x0"] + w["x1"]) / 2
|
|
def first(txt, n=0):
|
|
hits = sorted([w for w in ws if w["text"].startswith(txt)], key=lambda w: w["x0"])
|
|
return hits[n] if len(hits) > n else None
|
|
if kind == "indicators":
|
|
hs = [first("Nr"), first("Progresa") or first("Rādītājs") or first("rādītājs") or first("Indikators"), first("Mēr"), first("Bāzes", 0),
|
|
first("Bāzes", 1), first("Mērķa", 0), first("Mērķa", 1), first("Datu")]
|
|
else:
|
|
hs = [first("Nr"), first("Uzdevums"), (first("Atbildīgā") or first("Atbildī")), (first("Līdzatbildīgās") or first("Līdz")), first("Finanšu"), first("Indikators")]
|
|
if any(h is None for h in hs):
|
|
return None
|
|
return [c(h) for h in hs]
|
|
|
|
|
|
def gutters(words, centers, min_gap=2.0):
|
|
"""Column boundaries from the empty vertical strips between the words of one table page.
|
|
Between two neighbouring header centres the widest empty strip is the gutter; without one, the midpoint."""
|
|
iv = sorted((w["x0"], w["x1"]) for w in words)
|
|
gaps, end = [], None
|
|
for a, b in iv:
|
|
if end is not None and a - end >= min_gap:
|
|
gaps.append((end, a))
|
|
end = b if end is None else max(end, b)
|
|
bounds = []
|
|
for i in range(len(centers) - 1):
|
|
lo, hi = centers[i], centers[i + 1]
|
|
cand = [(g[1] - g[0], (g[0] + g[1]) / 2) for g in gaps if lo < (g[0] + g[1]) / 2 < hi]
|
|
bounds.append(max(cand)[1] if cand else (lo + hi) / 2)
|
|
return bounds
|
|
|
|
|
|
def assign(words, centers, names, bounds=None):
|
|
bounds = bounds or [(centers[i] + centers[i + 1]) / 2 for i in range(len(centers) - 1)]
|
|
cells = {n: [] for n in names}
|
|
for w in words:
|
|
cx = (w["x0"] + w["x1"]) / 2
|
|
i = sum(1 for b in bounds if cx > b)
|
|
cells[names[i]].append(w)
|
|
return cells
|
|
|
|
|
|
def cell_text(ws):
|
|
ws = sorted(ws, key=lambda w: (w.get("page", 0), round(w["top"]), w["x0"]))
|
|
t = norm(" ".join(w["text"] for w in ws))
|
|
return re.sub(r"(\d{4}/\d{1,3}) (\d)", r"\1\2", t) # "2018/2 019" broken inside a narrow cell
|
|
|
|
|
|
def split_by_gap(ws, gap=17.0):
|
|
"""Indicator names in a task row are separated by an empty line."""
|
|
ws = sorted(ws, key=lambda w: (w.get("page", 0), round(w["top"]), w["x0"]))
|
|
groups, last_top, last_page = [], None, None
|
|
for w in ws:
|
|
if last_top is None or w["top"] - last_top > gap or w.get("page") != last_page:
|
|
groups.append([])
|
|
groups[-1].append(w)
|
|
last_top, last_page = w["top"], w.get("page")
|
|
return [norm(" ".join(x["text"] for x in g)) for g in groups if g]
|
|
|
|
|
|
def table_pages(lines):
|
|
"""First pass: (kind, page) → words of table rows (≈10 pt lines between a table title and body-size text)."""
|
|
acc, mode, started, tid = defaultdict(list), None, False, 0
|
|
for ln in lines:
|
|
sq = squash(ln["text"])
|
|
if sq in SUB and ln["x0"] < 100:
|
|
mode = {"indicators": "indicator", "tasks": "task"}.get(SUB[sq])
|
|
started = False
|
|
tid += mode is not None
|
|
continue
|
|
if mode and ln["size"] >= 11:
|
|
mode = None
|
|
if mode and MARK.match(ln["words"][0]["text"]):
|
|
started = True
|
|
if mode and started and ln["size"] < 11 and not ln["text"].startswith("*"):
|
|
acc[(tid, ln["page"])] += ln["words"]
|
|
return acc
|
|
|
|
|
|
def parse(path):
|
|
pdf = pdfplumber.open(path)
|
|
load_vocab(path)
|
|
lines, notes = page_lines(pdf)
|
|
page_words = table_pages(lines)
|
|
page_bounds = {}
|
|
tid = 0
|
|
sections, items, evidence = [], [], []
|
|
sec = {"id": "front", "kind": "front", "title": "Titullapa un saīsinājumi", "parent": None}
|
|
sections.append(sec)
|
|
pr = rv = None
|
|
npr = nrv = 0
|
|
mode, centers, pending_goal = "text", None, None
|
|
centers_by = {}
|
|
area_buf = []
|
|
cur = None # current item being filled
|
|
hdr_buf = []
|
|
funding_buf = None
|
|
note_open = False
|
|
abbrs = []
|
|
problem, problem_open, problem_words = None, False, []
|
|
spatial_note, spatial_note_page = False, None
|
|
i = 0
|
|
stats = defaultdict(int)
|
|
stats_pages = []
|
|
|
|
def close():
|
|
nonlocal cur
|
|
if cur is None:
|
|
return
|
|
refs = sorted({r for w in cur.get("_words", []) for r in w.get("refs", [])})
|
|
if refs:
|
|
cur["footnoteRefs"] = refs
|
|
if cur["kind"] in ("indicator", "task"):
|
|
cols = IND_COLS if cur["kind"] == "indicator" else TASK_COLS
|
|
ws, ctr = cur.pop("_words"), cur.pop("_centers")
|
|
ws = [w for w in ws if not (w["text"] in ("(", "[") and w["x0"] < 125) and w["text"] != "["]
|
|
tid_ = cur.pop("_tid")
|
|
cells = {n: [] for n in cols}
|
|
for pg in sorted({w["page"] for w in ws}):
|
|
pw = [w for w in ws if w["page"] == pg]
|
|
key = (tid_, pg)
|
|
if key not in page_bounds:
|
|
page_bounds[key] = gutters(page_words.get(key) or pw, ctr)
|
|
b = page_bounds[key]
|
|
for k, v in assign(pw, ctr, cols, b).items():
|
|
cells[k] += v
|
|
if cur["kind"] == "indicator":
|
|
for k in cols[1:]:
|
|
cur[k] = cell_text(cells[k]) or None
|
|
else:
|
|
cur["text"] = cell_text(cells["text"])
|
|
cur["responsible"] = cell_text(cells["responsible"]) or None
|
|
cur["coResponsible"] = cell_text(cells["coResponsible"]) or None
|
|
cur["funding"] = cell_text(cells["funding"]) or None
|
|
cur["indicatorNames"] = split_by_gap(cells["indicators"])
|
|
else:
|
|
ws = cur.pop("_words")
|
|
bold = [w for w in ws if w["bold"]]
|
|
cur["text"] = norm(" ".join(w["text"] for w in ws))
|
|
cur["boldLead"] = norm(" ".join(w["text"] for w in ws[:len(ws)] if w["bold"])) if bold else None
|
|
if "_area" in cur:
|
|
cur["area"] = cur.pop("_area")[0] if cur["_area"] else None
|
|
items.append(cur)
|
|
cur = None
|
|
|
|
while i < len(lines):
|
|
ln = lines[i]
|
|
t, sq = ln["text"], squash(ln["text"])
|
|
# ---------------------------------------------------------------- big headings
|
|
if ln["size"] >= 13.5 and ln["page"] > 3:
|
|
j, title = i + 1, t
|
|
while j < len(lines) and lines[j]["size"] >= 13.5 and abs(lines[j]["size"] - ln["size"]) < 0.5 \
|
|
and lines[j]["page"] == ln["page"] and lines[j]["top"] - lines[j - 1]["top"] < 26:
|
|
title += " " + lines[j]["text"]
|
|
j += 1
|
|
title = norm(title)
|
|
close()
|
|
if funding_buf:
|
|
funding_buf = None
|
|
m = re.match(r"^Prioritāte “(.+)”$", title)
|
|
m2 = re.match(r"^Rīcības virziens “(.+)”$", title)
|
|
if sec.get("kind") == "annex" or (sections and any(s["kind"] == "annex" for s in sections)):
|
|
pass
|
|
if m:
|
|
npr += 1
|
|
pr = {"id": f"pr{npr}", "kind": "priority", "title": m.group(1), "parent": None}
|
|
sections.append(pr)
|
|
sec, rv, mode = pr, None, "text"
|
|
elif m2:
|
|
nrv += 1
|
|
rv = {"id": f"{pr['id']}.rv{nrv}", "kind": "actionLine", "title": m2.group(1), "parent": pr["id"], "funding": None}
|
|
sections.append(rv)
|
|
sec, mode = rv, "text"
|
|
elif title == "NAP2027 prioritāšu pamatojuma avoti":
|
|
mode = "annex"
|
|
else:
|
|
kind = next((k for k, h in SECTION_HEADS if squash(h) == squash(title)), None)
|
|
if kind:
|
|
sec = {"id": kind, "kind": kind, "title": title, "parent": None}
|
|
sections.append(sec)
|
|
pr = rv = None
|
|
mode = "annex" if kind == "annex" else "text"
|
|
elif ln["page"] > 4:
|
|
stats["unknown_heading"] += 1
|
|
i = j
|
|
continue
|
|
# ---------------------------------------------------------------- annex: evidence points per action line
|
|
if mode == "annex":
|
|
m = re.match(r"^Prioritāte “(.+)”$", norm(t))
|
|
m2 = re.match(r"^Rīcības virziens “(.+)”?$", norm(t))
|
|
if m:
|
|
pr = next((s for s in sections if s["kind"] == "priority" and squash(s["title"]) == squash(m.group(1))), None)
|
|
rv = None
|
|
problem, problem_open = None, False
|
|
i += 1
|
|
continue
|
|
if m2 and ln["x0"] < 90:
|
|
title = norm(t)
|
|
while not title.endswith("”") and i + 1 < len(lines):
|
|
i += 1
|
|
title = norm(title + " " + lines[i]["text"])
|
|
name = re.match(r"^Rīcības virziens “(.+)”$", title).group(1)
|
|
rv = next((s for s in sections if s["kind"] == "actionLine" and squash(s["title"]) == squash(name)), None)
|
|
if rv is None:
|
|
stats["annex_unknown_line"] += 1
|
|
problem, problem_open = None, False
|
|
i += 1
|
|
continue
|
|
m3 = re.match(r"^(\d{1,3})\.\s", t)
|
|
if m3 and ln["x0"] < 90:
|
|
problem_open = False
|
|
evidence.append({"number": int(m3.group(1)), "section": (rv or pr or {}).get("id"),
|
|
"_words": ln["words"][1:], "page": ln["page"], "problem": problem})
|
|
elif ln["bold"] and ln["x0"] < 90:
|
|
# bold problem statement that groups the following evidence points ("…:")
|
|
if problem_open and problem:
|
|
problem, problem_words = norm(problem + " " + t), problem_words + ln["words"]
|
|
else:
|
|
problem, problem_words = norm(t), list(ln["words"])
|
|
problem_open = not t.rstrip().endswith(":")
|
|
elif problem_open and ln["x0"] < 90:
|
|
# unnumbered evidence paragraph: bold lead without ":" continues as body text
|
|
evidence.append({"number": None, "section": (rv or pr or {}).get("id"), "_words": problem_words + ln["words"],
|
|
"page": ln["page"], "problem": None})
|
|
problem, problem_open, problem_words = None, False, []
|
|
elif evidence:
|
|
evidence[-1]["_words"] += ln["words"]
|
|
i += 1
|
|
continue
|
|
# ---------------------------------------------------------------- sub-headings
|
|
if sq in SUB and ln["x0"] < 100:
|
|
close()
|
|
kind = SUB[sq]
|
|
if kind in ("priorityGoal", "actionLineGoal"):
|
|
pending_goal, mode = kind, "text"
|
|
else:
|
|
mode, hdr_buf = kind, []
|
|
tid += 1
|
|
# header lines until the first row marker
|
|
j = i + 1
|
|
while j < len(lines) and not MARK.match(lines[j]["words"][0]["text"]):
|
|
hdr_buf.append(lines[j])
|
|
j += 1
|
|
c = header_centers(hdr_buf, kind)
|
|
if c:
|
|
centers_by[kind] = c
|
|
else:
|
|
stats["header_fallback_" + kind] += 1
|
|
stats.setdefault("header_fallback_pages", []).append(ln["page"])
|
|
centers = centers_by.get(kind)
|
|
i = j
|
|
continue
|
|
i += 1
|
|
continue
|
|
# ---------------------------------------------------------------- indicative funding of an action line
|
|
if sq.startswith("Rīcībasvirzienapasākumu") or funding_buf is not None:
|
|
close()
|
|
funding_buf = (funding_buf or "") + " " + t
|
|
if "EUR" in t:
|
|
m = re.search(r"apjoms\s+([\d\s,]+)\s*milj\.\s*EUR", norm(funding_buf))
|
|
if rv is not None:
|
|
rv["funding"] = {"text": norm(funding_buf), "millionEur": m.group(1).replace(" ", "") if m else None}
|
|
funding_buf = None
|
|
mode = "text"
|
|
i += 1
|
|
continue
|
|
# ---------------------------------------------------------------- spatial perspective: area | items
|
|
if sec.get("kind") == "spatial" and (t.startswith("Pilsētu un lauku mijiedarbības dimensijas") or spatial_note):
|
|
if MARK.search(t) or ln["page"] != spatial_note_page and spatial_note:
|
|
spatial_note = False
|
|
else:
|
|
close()
|
|
if not spatial_note:
|
|
sec.setdefault("notes", []).append({"area": area_buf[0] if area_buf else None, "text": norm(t)})
|
|
spatial_note, spatial_note_page = True, ln["page"]
|
|
else:
|
|
sec["notes"][-1]["text"] = norm(sec["notes"][-1]["text"] + ("\n" if t.startswith(("–", "−")) else " ") + t)
|
|
i += 1
|
|
continue
|
|
if sec.get("kind") == "spatial":
|
|
mk_i = next((k for k, w in enumerate(ln["words"]) if MARK.match(w["text"])), None)
|
|
if mk_i is not None and ln["words"][mk_i]["x0"] > 180:
|
|
close()
|
|
left = [w for w in ln["words"][:mk_i] if w["x0"] < 195]
|
|
if left:
|
|
area_buf[:] = [norm(" ".join(w["text"] for w in left))]
|
|
cur = {"number": int(MARK.match(ln["words"][mk_i]["text"]).group(1)), "kind": "spatialDirection",
|
|
"section": "spatial", "page": ln["page"], "_words": ln["words"][mk_i + 1:], "_area": area_buf}
|
|
i += 1
|
|
continue
|
|
if cur is not None and cur.get("_area") is not None:
|
|
left = [w for w in ln["words"] if w["x0"] < 195]
|
|
right = [w for w in ln["words"] if w["x0"] >= 195]
|
|
if left and not right and len(left) <= 5:
|
|
area_buf[0] = norm(area_buf[0] + " " + " ".join(w["text"] for w in left)) if area_buf else norm(" ".join(w["text"] for w in left))
|
|
i += 1
|
|
continue
|
|
if left and right and right[0]["x0"] > 190 and len(left) <= 4:
|
|
area_buf[0] = norm(area_buf[0] + " " + " ".join(w["text"] for w in left))
|
|
cur["_words"] += right
|
|
i += 1
|
|
continue
|
|
# ---------------------------------------------------------------- table notes ("* ...", 9 pt)
|
|
if (t.startswith("*") and ln["size"] < 9.5) or (note_open and ln["size"] < 9.5 and not MARK.match(ln["words"][0]["text"])):
|
|
if t.startswith("*"):
|
|
close()
|
|
sec.setdefault("notes", []).append(norm(t.lstrip("* ")))
|
|
else:
|
|
sec["notes"][-1] = norm(sec["notes"][-1] + " " + t)
|
|
note_open = True
|
|
i += 1
|
|
continue
|
|
note_open = False
|
|
# ---------------------------------------------------------------- thematic sub-sections inside an action line
|
|
if rv is not None and 11.5 < ln["size"] < 13.5 and t.startswith("“") and norm(t).endswith("”") and ln["x0"] < 90:
|
|
close()
|
|
n_th = sum(1 for x in sections if x.get("parent") == rv["id"]) + 1
|
|
sec = {"id": f"{rv['id']}.t{n_th}", "kind": "theme", "title": norm(t).strip("“”").strip(), "parent": rv["id"]}
|
|
sections.append(sec)
|
|
mode = "text"
|
|
i += 1
|
|
continue
|
|
# ---------------------------------------------------------------- numbered items
|
|
first = ln["words"][0]
|
|
mk = MARK.match(first["text"])
|
|
in_table = mode in ("indicators", "tasks")
|
|
if in_table and not mk and ln["size"] >= 11:
|
|
# body-size text ends the table (table cells are ≈10 pt)
|
|
if True:
|
|
close()
|
|
mode = "text"
|
|
in_table = False
|
|
if mk:
|
|
close()
|
|
n = int(mk.group(1))
|
|
if in_table and ln["size"] < 11:
|
|
kind = "indicator" if mode == "indicators" else "task"
|
|
cur = {"number": n, "kind": kind, "section": (sec if sec.get("kind") == "theme" else (rv or pr or sec))["id"], "page": ln["page"],
|
|
"_words": ln["words"][1:], "_centers": centers, "_tid": tid}
|
|
else:
|
|
if in_table:
|
|
mode = "text"
|
|
k = "text"
|
|
if pending_goal:
|
|
k = pending_goal
|
|
cur = {"number": n, "kind": k, "section": (sec if sec.get("kind") == "theme" else (rv or pr or sec))["id"], "page": ln["page"], "_words": ln["words"][1:]}
|
|
if pending_goal:
|
|
pending_goal = None if pending_goal == "priorityGoal" else pending_goal
|
|
i += 1
|
|
continue
|
|
if cur is not None:
|
|
cur["_words"] += [w for w in ln["words"] if not (w["text"] == "[" )]
|
|
elif sec["kind"] == "front" and ln["page"] in (3, 4) and t != "IZMANTOTIE SAĪSINĀJUMI":
|
|
m = re.match(r"^(.+?)\s+–\s+(.+)$", t)
|
|
if m and ln["x0"] < 90 and len(m.group(1)) <= 40:
|
|
abbrs.append({"abbr": m.group(1).strip(), "meaning": m.group(2).strip()})
|
|
elif abbrs:
|
|
abbrs[-1]["meaning"] = norm(abbrs[-1]["meaning"] + " " + t)
|
|
elif sec["kind"] == "front":
|
|
pass # title page and table of contents
|
|
else:
|
|
stats["orphan_line"] += 1
|
|
stats.setdefault("orphans", []).append([ln["page"], (sec or {}).get("id"), " ".join(w["text"] for w in ln["words"])[:110]])
|
|
i += 1
|
|
close()
|
|
|
|
# goals: after RĪCĪBAS VIRZIENA MĒRĶIS(-I) only bold items are goals; the first non-bold item is context
|
|
for it in items:
|
|
if it["kind"] == "actionLineGoal" and not (it.get("boldLead") and len(it["boldLead"]) >= 0.6 * len(it["text"])):
|
|
it["kind"] = "text"
|
|
# once a non-goal follows, later items in the section are context
|
|
seen_text = set()
|
|
for it in items:
|
|
if it["kind"] == "text":
|
|
seen_text.add(it["section"])
|
|
elif it["kind"] == "actionLineGoal" and it["section"] in seen_text:
|
|
it["kind"] = "text"
|
|
# strategic goals: paragraphs with a bold lead in the strategic goals section
|
|
for it in items:
|
|
if it["section"] == "strategicGoals" and it["kind"] == "text" and it.get("boldLead"):
|
|
it["kind"] = "strategicGoal"
|
|
it["title"] = it["boldLead"]
|
|
for e in evidence:
|
|
refs = sorted({r for w in e["_words"] for r in w.get("refs", [])})
|
|
if refs:
|
|
e["footnoteRefs"] = refs
|
|
e["text"] = norm(" ".join(w["text"] for w in e.pop("_words")))
|
|
urls = re.findall(r"(?:https?://|www\.)\S+", re.sub(r"(?<=[-/_.%=?&])\s+(?=\S)", "", e["text"]))
|
|
clean = []
|
|
for u in urls:
|
|
u = u.rstrip(".,;")
|
|
while u.endswith(")") and u.count(")") > u.count("("):
|
|
u = u[:-1]
|
|
clean.append(u.rstrip(".,;"))
|
|
e["urls"] = clean
|
|
return {"abbreviations": abbrs, "sections": sections, "items": items, "evidence": evidence, "footnotes": [dict(n, text=fix_urls(norm(n["text"]))) for n in notes], "stats": dict(stats)}
|
|
|
|
|
|
if __name__ == "__main__":
|
|
doc = parse(sys.argv[1])
|
|
out = sys.argv[2]
|
|
with open(out, "w", encoding="utf-8") as f:
|
|
json.dump(doc, f, ensure_ascii=False, indent=1)
|
|
from collections import Counter
|
|
print("sections", Counter(s["kind"] for s in doc["sections"]))
|
|
print("items", len(doc["items"]), Counter(i["kind"] for i in doc["items"]))
|
|
print("evidence", len(doc["evidence"]), "footnotes", len(doc["footnotes"]), "stats", doc["stats"])
|