"""Shared readers for budget-law annex files (xlsx via openpyxl, xls via xlrd, docx via stdlib).""" import zipfile, re, xml.etree.ElementTree as ET def sheets(path): """Yield (sheet_name, rows) with rows as lists of stripped strings/numbers; hidden BEx sheets included.""" if path.lower().endswith('.xlsx'): import openpyxl wb = openpyxl.load_workbook(path, read_only=True, data_only=True) for ws in wb.worksheets: rows = [] for r in ws.iter_rows(values_only=True): rows.append([("" if v is None else (v.strip() if isinstance(v, str) else v)) for v in r]) yield ws.title, rows elif path.lower().endswith('.xls'): import xlrd wb = xlrd.open_workbook(path) for ws in wb.sheets(): rows = [] for i in range(ws.nrows): rows.append([(c.strip() if isinstance(c, str) else c) for c in ws.row_values(i)]) yield ws.name, rows def main_sheet(path): """The printed annex sheet: the one whose name ends with 'piel' or is the first non-BEx sheet with most rows.""" best = None for name, rows in sheets(path): if name.lower().endswith('piel') or re.match(r'^\d+\.?\s*piel', name.lower()): return name, rows if name.startswith(('BEx', 'HEADER', 'FOOTER', 'ZQZ', 'var', 'list', 'parms')): continue if best is None or len(rows) > len(best[1]): best = (name, rows) return best W = '{http://schemas.openxmlformats.org/wordprocessingml/2006/main}' def docx_tables(path): """List of tables, each a list of rows of cell texts.""" x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml')) out = [] for t in x.iter(W + 'tbl'): rows = [] for tr in t.findall(W + 'tr'): rows.append([''.join(n.text or '' for n in tc.iter(W + 't')).strip() for tc in tr.findall(W + 'tc')]) out.append(rows) return out def docx_paras(path): x = ET.fromstring(zipfile.ZipFile(path).read('word/document.xml')) return [''.join(n.text or '' for n in p.iter(W + 't')).strip() for p in x.iter(W + 'p')] def num(v): """Parse an amount cell: numbers, '1 234 567', '-1 234', '–'. Returns float or None.""" if isinstance(v, (int, float)): return float(v) s = str(v).replace(' ', ' ').replace(' ', '').replace('–', '-').replace('−', '-').replace(',', '.') return float(s) if re.fullmatch(r'-?\d+(\.\d+)?', s) else None