diff options
| -rw-r--r-- | scripts/parse-sanpi-hyo.py | 72 |
1 files changed, 56 insertions, 16 deletions
diff --git a/scripts/parse-sanpi-hyo.py b/scripts/parse-sanpi-hyo.py index 1e6a496..4af544e 100644 --- a/scripts/parse-sanpi-hyo.py +++ b/scripts/parse-sanpi-hyo.py @@ -63,37 +63,77 @@ def row_chars(cs, x0, x1, y0, y1): return sorted(sel, key=lambda c: (round(c["y"] / 4), c["x"])) +def _text_lines(sel, tol=3.0): + """文字を y 座標で「行」にまとめ、行内は x 順に並べて返す。""" + lines = [] + for c in sorted(sel, key=lambda c: c["y"]): + if lines and c["y"] - lines[-1][-1]["y"] <= tol: + lines[-1].append(c) + else: + lines.append([c]) + return [sorted(ln, key=lambda c: c["x"]) for ln in lines] + + +def _clusters_by_x(cs, gap=12.0): + """文字を x 座標でまとめ、列(横位置の塊)に分ける。""" + out = [] + for c in sorted(cs, key=lambda c: c["x"]): + if out and c["x"] - out[-1][-1]["x"] <= gap: + out[-1].append(c) + else: + out.append([c]) + return out + + def split_row(sel): """行の帯の文字から、号数・件名・議決結果を切り出す。 - PDF の件名はセルの中で中央揃えの複数行になるため、単純に連結すると - 左の号数欄・右の議決結果欄が件名の途中に挟まる。そこで議決結果は - 「いちばん右にある議決語」とし、その語だけを件名から取り除く。 + PDF は列ごとに独立したテキストになっているため、単純に連結すると号数や + 議決結果が件名の途中に挟まる(号数は左右に分かれた件名の行の「間」に + 来ることがある)。そこで、 + - 号数(第N号)は行ごとに探し、いちばん左にあるものを採る + - 議決結果は、いちばん右にある議決語 + - 件名は、号数より右の文字を x でまとめ、最初(左)の塊を採る + とする。号数・議決結果の列や、左端の「議員提出議案」等の欄や票の列は + 自然に外れる。 """ tch = [c["c"].translate(FULL2HALF) for c in sel] tstr = "".join(tch) - drop = set() - no = "" - no_m = NO_RE.search(tstr) - if no_m: - no = no_m.group(0) - drop.update(range(*no_m.span())) + no, no_hi, best_x = "", None, None + for ln in _text_lines(sel): + s = "".join(c["c"].translate(FULL2HALF) for c in ln) + for m in re.finditer(r"第[0-9]+号", s): + x = ln[m.start()]["x"] + if best_x is None or x < best_x: + best_x, no, no_hi = x, m.group(0), ln[m.end() - 1]["x"] - kekka = "" - best = None + kekka, best = "", None for m in KEKKA_RE.finditer(tstr): x = sel[m.start()]["x"] if best is None or x > best[0]: best = (x, m) if best: kekka = best[1].group(0) - drop.update(range(*best[1].span())) - for m in LEGEND_RE.finditer(tstr): - drop.update(range(*m.span())) + if no_hi is None: + # 号数が見つからない行は、従来どおり全体を連結して組み立てる + drop = set() + if best: + drop.update(range(*best[1].span())) + for m in LEGEND_RE.finditer(tstr): + drop.update(range(*m.span())) + name = "".join(tch[i] for i in range(len(tch)) if i not in drop) + return no, kekka, name.strip() - name = "".join(tch[i] for i in range(len(tch)) if i not in drop) + # 件名: 号数より右の文字を x でまとめ、最初(左)の塊を読む順に連結する。 + # 議決結果や票の列は右側の別の塊になるので、件名には混ざらない。 + # (件名自体に「同意」などの議決語が入っていても、それは消さない) + right = [c for c in sel if c["x"] > no_hi + 2] + clusters = _clusters_by_x(right) + title = clusters[0] if clusters else [] + title.sort(key=lambda c: (round(c["y"]), c["x"])) + name = "".join(c["c"].translate(FULL2HALF) for c in title) return no, kekka, name.strip() @@ -183,7 +223,7 @@ def main(path, label): for n, y in enumerate(ys): lo = (ymid[n - 1] + ymid[n]) / 2 if n > 0 else ymid[0] - 16 hi = (ymid[n] + ymid[n + 1]) / 2 if n + 1 < len(ys) else ymid[n] + 16 - sel = row_chars(cs, 74, left_end, lo, hi) + sel = row_chars(cs, 0, left_end, lo, hi) no, kekka, name = split_row(sel) if os.environ.get("GEN_DEBUG"): print(f" [row n={n} lo={lo:.0f} hi={hi:.0f}] no={no!r} kekka={kekka!r} name={name!r}") |
