From 3c63cfc2d4e8277ba9490141cc17b4570ba4d23a Mon Sep 17 00:00:00 2001 From: Yasutake Yohei <61961825+yasutakeyohei@users.noreply.github.com> Date: Tue, 6 Oct 2026 13:59:27 +0900 Subject: psi-extract: PSIレポートから監査項目を抜き出すツールを追加 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 失敗・要改善の項目の見出しだけ、または指定した項目の本文だけを取り出す。 レポートの丸読みによるコンテキストオーバーを避けるため、既定で1件1200文字に切り詰める。 --- scripts/psi-extract.py | 141 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 141 insertions(+) create mode 100644 scripts/psi-extract.py (limited to 'scripts/psi-extract.py') diff --git a/scripts/psi-extract.py b/scripts/psi-extract.py new file mode 100644 index 0000000..5d99dd7 --- /dev/null +++ b/scripts/psi-extract.py @@ -0,0 +1,141 @@ +"""PageSpeed Insights のレポートから、必要な監査項目だけを抜き出す。 + +PSI のレポートを丸ごと読むと、それだけで会話がコンテキストオーバーになる。 +このスクリプトは、失敗・要改善の項目の見出しだけ、あるいは指定した項目の +本文(説明と表)だけを取り出し、既定で1件1200文字に切り詰めて出力する。 + +- 引数にレポートだけを渡すと、失敗・要改善の監査項目を見出しだけ一覧表示する +- 見出しの一部分を続けて渡すと、その項目の本文だけを表示する(部分一致) +- レポートは MHTML(.mhtml)でも HTML(.html)でもよい +- PSI のレポートはモバイル版とデスクトップ版が並ぶため、同じ見出しは1件だけ出す +- pass の項目は出力しない(コンテキスト節約のため) + +使い方: + python scripts/psi-extract.py + python scripts/psi-extract.py 画像 DOM + python scripts/psi-extract.py 画像 --full + python scripts/psi-extract.py 画像 --max 3000 +""" +import email +import html +import re +import sys +from pathlib import Path + +sys.stdout.reconfigure(encoding="utf-8") + +# 1件あたりの出力上限(文字数)。 +DEFAULT_MAX = 1200 + +AUDIT_MARK = re.compile(r'class="lh-audit lh-audit--[a-z]+ lh-audit--(fail|average|pass)"') +TITLE = re.compile(r'class="lh-audit__title">(.*?)', re.S) +AUDIT_SPLIT = re.compile(r'
str: + return html.unescape(re.sub(r"<[^>]+>", "", raw)).strip() + + +def audit_list(page: str) -> list[tuple[str, str]]: + """(種別, 見出し) を、見出しの重複を除いて返す。""" + rows: list[tuple[str, str]] = [] + seen: set[str] = set() + for m in AUDIT_MARK.finditer(page): + kind = m.group(1) + if kind == "pass": + continue + t = TITLE.search(page, m.end()) + if not t: + continue + name = text_of(t.group(1)) + if name and name not in seen: + seen.add(name) + rows.append((kind, name)) + return rows + + +def audit_blocks(page: str) -> list[tuple[str, str]]: + """(見出し, ブロック) を、見出しの重複を除いて返す。""" + rows: list[tuple[str, str]] = [] + seen: set[str] = set() + for block in AUDIT_SPLIT.split(page)[1:]: + m = TITLE.search(block) + if not m: + continue + name = text_of(m.group(1)) + if name and name not in seen: + seen.add(name) + rows.append((name, block)) + return rows + + +def block_text(block: str) -> str: + """ブロックからタグを落とし、読みやすいテキストにする。""" + # AUDIT_SPLIT で分割したため、先頭に開始タグの残骸が残る。それを落とす。 + block = re.sub(r"^[^>]*>", "", block, count=1) + text = re.sub(r"<(script|style)[^>]*>.*?", "", block, flags=re.S) + text = re.sub(r"", "\n", text) + text = re.sub(r"", "\n", text) + text = re.sub(r"", " | ", text) + text = re.sub(r"<[^>]+>", "", text) + text = html.unescape(text) + text = re.sub(r"[ \t]+\n", "\n", text) + text = re.sub(r"\n{3,}", "\n\n", text) + return text.strip() + + +def truncate(text: str, limit: int | None) -> str: + if limit is None or len(text) <= limit: + return text + return f"{text[:limit]}\n…(残り{len(text) - limit}文字を省略)" + + +def main() -> None: + args = sys.argv[1:] + if not args or args[0] in {"-h", "--help"}: + print(__doc__.strip()) + return + + report = Path(args.pop(0)) + limit = None if "--full" in args else DEFAULT_MAX + if "--max" in args: + i = args.index("--max") + limit = int(args[i + 1]) + del args[i : i + 2] + targets = [a for a in args if not a.startswith("--")] + + page = read_html(report) + + if not targets: + print("# 失敗・要改善の監査項目") + for kind, name in audit_list(page): + print(f"[{kind}] {name}") + return + + found = 0 + for name, block in audit_blocks(page): + if not any(t in name for t in targets): + continue + print(f"\n===== {name} =====") + print(truncate(block_text(block), limit)) + found += 1 + if not found: + print("該当する監査項目が見つかりませんでした。", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": + main() -- cgit v1.3.1