aboutsummaryrefslogtreecommitdiffhomepage
path: root/scripts/parse-sanpi-hyo.py
diff options
context:
space:
mode:
Diffstat (limited to 'scripts/parse-sanpi-hyo.py')
-rw-r--r--scripts/parse-sanpi-hyo.py72
1 files changed, 56 insertions, 16 deletions
diff --git a/scripts/parse-sanpi-hyo.py b/scripts/parse-sanpi-hyo.py
index 1e6a496..4af544e 100644
--- a/scripts/parse-sanpi-hyo.py
+++ b/scripts/parse-sanpi-hyo.py
@@ -63,37 +63,77 @@ def row_chars(cs, x0, x1, y0, y1):
return sorted(sel, key=lambda c: (round(c["y"] / 4), c["x"]))
+def _text_lines(sel, tol=3.0):
+ """文字を y 座標で「行」にまとめ、行内は x 順に並べて返す。"""
+ lines = []
+ for c in sorted(sel, key=lambda c: c["y"]):
+ if lines and c["y"] - lines[-1][-1]["y"] <= tol:
+ lines[-1].append(c)
+ else:
+ lines.append([c])
+ return [sorted(ln, key=lambda c: c["x"]) for ln in lines]
+
+
+def _clusters_by_x(cs, gap=12.0):
+ """文字を x 座標でまとめ、列(横位置の塊)に分ける。"""
+ out = []
+ for c in sorted(cs, key=lambda c: c["x"]):
+ if out and c["x"] - out[-1][-1]["x"] <= gap:
+ out[-1].append(c)
+ else:
+ out.append([c])
+ return out
+
+
def split_row(sel):
"""行の帯の文字から、号数・件名・議決結果を切り出す。
- PDF の件名はセルの中で中央揃えの複数行になるため、単純に連結すると
- 左の号数欄・右の議決結果欄が件名の途中に挟まる。そこで議決結果は
- 「いちばん右にある議決語」とし、その語だけを件名から取り除く。
+ PDF は列ごとに独立したテキストになっているため、単純に連結すると号数や
+ 議決結果が件名の途中に挟まる(号数は左右に分かれた件名の行の「間」に
+ 来ることがある)。そこで、
+ - 号数(第N号)は行ごとに探し、いちばん左にあるものを採る
+ - 議決結果は、いちばん右にある議決語
+ - 件名は、号数より右の文字を x でまとめ、最初(左)の塊を採る
+ とする。号数・議決結果の列や、左端の「議員提出議案」等の欄や票の列は
+ 自然に外れる。
"""
tch = [c["c"].translate(FULL2HALF) for c in sel]
tstr = "".join(tch)
- drop = set()
- no = ""
- no_m = NO_RE.search(tstr)
- if no_m:
- no = no_m.group(0)
- drop.update(range(*no_m.span()))
+ no, no_hi, best_x = "", None, None
+ for ln in _text_lines(sel):
+ s = "".join(c["c"].translate(FULL2HALF) for c in ln)
+ for m in re.finditer(r"第[0-9]+号", s):
+ x = ln[m.start()]["x"]
+ if best_x is None or x < best_x:
+ best_x, no, no_hi = x, m.group(0), ln[m.end() - 1]["x"]
- kekka = ""
- best = None
+ kekka, best = "", None
for m in KEKKA_RE.finditer(tstr):
x = sel[m.start()]["x"]
if best is None or x > best[0]:
best = (x, m)
if best:
kekka = best[1].group(0)
- drop.update(range(*best[1].span()))
- for m in LEGEND_RE.finditer(tstr):
- drop.update(range(*m.span()))
+ if no_hi is None:
+ # 号数が見つからない行は、従来どおり全体を連結して組み立てる
+ drop = set()
+ if best:
+ drop.update(range(*best[1].span()))
+ for m in LEGEND_RE.finditer(tstr):
+ drop.update(range(*m.span()))
+ name = "".join(tch[i] for i in range(len(tch)) if i not in drop)
+ return no, kekka, name.strip()
- name = "".join(tch[i] for i in range(len(tch)) if i not in drop)
+ # 件名: 号数より右の文字を x でまとめ、最初(左)の塊を読む順に連結する。
+ # 議決結果や票の列は右側の別の塊になるので、件名には混ざらない。
+ # (件名自体に「同意」などの議決語が入っていても、それは消さない)
+ right = [c for c in sel if c["x"] > no_hi + 2]
+ clusters = _clusters_by_x(right)
+ title = clusters[0] if clusters else []
+ title.sort(key=lambda c: (round(c["y"]), c["x"]))
+ name = "".join(c["c"].translate(FULL2HALF) for c in title)
return no, kekka, name.strip()
@@ -183,7 +223,7 @@ def main(path, label):
for n, y in enumerate(ys):
lo = (ymid[n - 1] + ymid[n]) / 2 if n > 0 else ymid[0] - 16
hi = (ymid[n] + ymid[n + 1]) / 2 if n + 1 < len(ys) else ymid[n] + 16
- sel = row_chars(cs, 74, left_end, lo, hi)
+ sel = row_chars(cs, 0, left_end, lo, hi)
no, kekka, name = split_row(sel)
if os.environ.get("GEN_DEBUG"):
print(f" [row n={n} lo={lo:.0f} hi={hi:.0f}] no={no!r} kekka={kekka!r} name={name!r}")