import sys import json import re import pdfplumber sys.stdout.reconfigure(encoding="utf-8") VOTE_CHARS = set("〇○×✕✖◯") # 賛成・反対の記号候補 def norm(c): """セル文字列を正規化(空白・改行を除去)""" if c is None: return "" return "".join(str(c).split()) HEADER_WORDS = {"区分", "番号", "件名", "議決結果", "賛成", "反対", "会派", "議員", "賛否"} def is_vote(c): s = norm(c) return len(s) == 1 and s in VOTE_CHARS def is_name(c): s = norm(c) if not (2 <= len(s) <= 8): return False if is_vote(c) or s in HEADER_WORDS: return False return not re.search(r"[0-90-9]", s) def is_label(c): s = norm(c) if not (1 <= len(s) <= 5): return False if is_vote(c) or s in HEADER_WORDS: return False return not re.search(r"[0-90-9]", s) def main(path): members = [] # [(col, 会派, 議員名)] rows = [] # [{区分, 番号, 件名, 議決結果, votes:{col: '〇'/'×'}}] carries = {} # 区分の繰越 with pdfplumber.open(path) as pdf: for page in pdf.pages: for table in page.extract_tables(): if not table: continue ncols = max(len(r) for r in table) grid = [list(r) + [None] * (ncols - len(r)) for r in table] # 議員名の行=2〜8文字の名前らしいセルが最も多く並ぶ行 best = None for i, row in enumerate(grid[:6]): n = sum(1 for c in row if is_name(c)) if n >= 5 and (best is None or n > best[1]): best = (i, n) if best is None: continue member_row_idx = best[0] # 会派名は、その上の行(無ければ同じ行の短いラベル)から拾う label_row = grid[member_row_idx - 1] if member_row_idx > 0 else grid[member_row_idx] local = {} group = "" for c in range(ncols): lab = norm(label_row[c]) if lab and is_label(label_row[c]): group = lab if is_name(grid[member_row_idx][c]): local[c] = (group, norm(grid[member_row_idx][c])) if local and not members: members = [(c, g, n) for c, (g, n) in sorted(local.items())] # データ行(〇×がある行) for row in grid[member_row_idx + 1:]: votes = {c: norm(row[c]) for c in local if is_vote(row[c])} if len(votes) < 2: continue vals = [norm(x) for x in row] disp = vals[0] if vals[0] else carries.get("区分", "") if vals[0]: carries["区分"] = vals[0] rows.append({ "区分": disp, "番号": next((v for v in vals[1:4] if v and "第" in v), ""), "件名": next((v for v in vals[1:6] if len(v) >= 8), ""), "議決結果": next((v for v in vals if v in ("認 定", "認定", "原案可決", "可決", "不採択", "採択", "否決", "原案否決")), ""), "votes": votes, "raw": vals[:4], }) print("議員(会派 / 名前):") for c, g, n in members: print(f" col{c:>3}: {g:<6} {n}") print("\n議案行数:", len(rows)) for r in rows[:6]: vs = "".join(r["votes"].get(c, "・") for c, _, _ in members) print(f" {r['区分'][:6]:<6} {r['番号']:<8} {r['議決結果']:<8} {r['件名'][:28]:<28} {vs}") with open(".sanpi-parsed.json", "w", encoding="utf-8") as f: json.dump({"members": members, "rows": rows}, f, ensure_ascii=False, indent=1) print("\n.sanpi-parsed.json に保存しました") main(".tmp-hyo.pdf")