diff options
Diffstat (limited to 'scripts')
| -rw-r--r-- | scripts/psi-extract.py | 141 |
1 files changed, 141 insertions, 0 deletions
diff --git a/scripts/psi-extract.py b/scripts/psi-extract.py new file mode 100644 index 0000000..5d99dd7 --- /dev/null +++ b/scripts/psi-extract.py @@ -0,0 +1,141 @@ +"""PageSpeed Insights のレポートから、必要な監査項目だけを抜き出す。 + +PSI のレポートを丸ごと読むと、それだけで会話がコンテキストオーバーになる。 +このスクリプトは、失敗・要改善の項目の見出しだけ、あるいは指定した項目の +本文(説明と表)だけを取り出し、既定で1件1200文字に切り詰めて出力する。 + +- 引数にレポートだけを渡すと、失敗・要改善の監査項目を見出しだけ一覧表示する +- 見出しの一部分を続けて渡すと、その項目の本文だけを表示する(部分一致) +- レポートは MHTML(.mhtml)でも HTML(.html)でもよい +- PSI のレポートはモバイル版とデスクトップ版が並ぶため、同じ見出しは1件だけ出す +- pass の項目は出力しない(コンテキスト節約のため) + +使い方: + python scripts/psi-extract.py <report.mhtml|report.html> + python scripts/psi-extract.py <report> 画像 DOM + python scripts/psi-extract.py <report> 画像 --full + python scripts/psi-extract.py <report> 画像 --max 3000 +""" +import email +import html +import re +import sys +from pathlib import Path + +sys.stdout.reconfigure(encoding="utf-8") + +# 1件あたりの出力上限(文字数)。 +DEFAULT_MAX = 1200 + +AUDIT_MARK = re.compile(r'class="lh-audit lh-audit--[a-z]+ lh-audit--(fail|average|pass)"') +TITLE = re.compile(r'class="lh-audit__title"><span>(.*?)</span>', re.S) +AUDIT_SPLIT = re.compile(r'<div class="lh-audit ') + + +def read_html(path: Path) -> str: + """MHTML または HTML から HTML 本文を取り出す。""" + if path.suffix.lower() in {".mhtml", ".mht"}: + message = email.message_from_bytes(path.read_bytes()) + for part in message.walk(): + if part.get_content_type() == "text/html": + payload = part.get_payload(decode=True) or b"" + charset = part.get_content_charset() or "utf-8" + return payload.decode(charset, "replace") + raise SystemExit(f"{path}: HTML パートが見つかりませんでした") + return path.read_text(encoding="utf-8", errors="replace") + + +def text_of(raw: str) -> str: + return html.unescape(re.sub(r"<[^>]+>", "", raw)).strip() + + +def audit_list(page: str) -> list[tuple[str, str]]: + """(種別, 見出し) を、見出しの重複を除いて返す。""" + rows: list[tuple[str, str]] = [] + seen: set[str] = set() + for m in AUDIT_MARK.finditer(page): + kind = m.group(1) + if kind == "pass": + continue + t = TITLE.search(page, m.end()) + if not t: + continue + name = text_of(t.group(1)) + if name and name not in seen: + seen.add(name) + rows.append((kind, name)) + return rows + + +def audit_blocks(page: str) -> list[tuple[str, str]]: + """(見出し, ブロック) を、見出しの重複を除いて返す。""" + rows: list[tuple[str, str]] = [] + seen: set[str] = set() + for block in AUDIT_SPLIT.split(page)[1:]: + m = TITLE.search(block) + if not m: + continue + name = text_of(m.group(1)) + if name and name not in seen: + seen.add(name) + rows.append((name, block)) + return rows + + +def block_text(block: str) -> str: + """ブロックからタグを落とし、読みやすいテキストにする。""" + # AUDIT_SPLIT で分割したため、先頭に開始タグの残骸が残る。それを落とす。 + block = re.sub(r"^[^>]*>", "", block, count=1) + text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", block, flags=re.S) + text = re.sub(r"<br\s*/?>", "\n", text) + text = re.sub(r"</(?:tr|p|div|li|h[1-6])>", "\n", text) + text = re.sub(r"</t[dh]>", " | ", text) + text = re.sub(r"<[^>]+>", "", text) + text = html.unescape(text) + text = re.sub(r"[ \t]+\n", "\n", text) + text = re.sub(r"\n{3,}", "\n\n", text) + return text.strip() + + +def truncate(text: str, limit: int | None) -> str: + if limit is None or len(text) <= limit: + return text + return f"{text[:limit]}\n…(残り{len(text) - limit}文字を省略)" + + +def main() -> None: + args = sys.argv[1:] + if not args or args[0] in {"-h", "--help"}: + print(__doc__.strip()) + return + + report = Path(args.pop(0)) + limit = None if "--full" in args else DEFAULT_MAX + if "--max" in args: + i = args.index("--max") + limit = int(args[i + 1]) + del args[i : i + 2] + targets = [a for a in args if not a.startswith("--")] + + page = read_html(report) + + if not targets: + print("# 失敗・要改善の監査項目") + for kind, name in audit_list(page): + print(f"[{kind}] {name}") + return + + found = 0 + for name, block in audit_blocks(page): + if not any(t in name for t in targets): + continue + print(f"\n===== {name} =====") + print(truncate(block_text(block), limit)) + found += 1 + if not found: + print("該当する監査項目が見つかりませんでした。", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": + main() |
