aboutsummaryrefslogtreecommitdiffhomepage
diff options
context:
space:
mode:
-rw-r--r--scripts/psi-extract.py141
1 files changed, 141 insertions, 0 deletions
diff --git a/scripts/psi-extract.py b/scripts/psi-extract.py
new file mode 100644
index 0000000..5d99dd7
--- /dev/null
+++ b/scripts/psi-extract.py
@@ -0,0 +1,141 @@
+"""PageSpeed Insights のレポートから、必要な監査項目だけを抜き出す。
+
+PSI のレポートを丸ごと読むと、それだけで会話がコンテキストオーバーになる。
+このスクリプトは、失敗・要改善の項目の見出しだけ、あるいは指定した項目の
+本文(説明と表)だけを取り出し、既定で1件1200文字に切り詰めて出力する。
+
+- 引数にレポートだけを渡すと、失敗・要改善の監査項目を見出しだけ一覧表示する
+- 見出しの一部分を続けて渡すと、その項目の本文だけを表示する(部分一致)
+- レポートは MHTML(.mhtml)でも HTML(.html)でもよい
+- PSI のレポートはモバイル版とデスクトップ版が並ぶため、同じ見出しは1件だけ出す
+- pass の項目は出力しない(コンテキスト節約のため)
+
+使い方:
+ python scripts/psi-extract.py <report.mhtml|report.html>
+ python scripts/psi-extract.py <report> 画像 DOM
+ python scripts/psi-extract.py <report> 画像 --full
+ python scripts/psi-extract.py <report> 画像 --max 3000
+"""
+import email
+import html
+import re
+import sys
+from pathlib import Path
+
+sys.stdout.reconfigure(encoding="utf-8")
+
+# 1件あたりの出力上限(文字数)。
+DEFAULT_MAX = 1200
+
+AUDIT_MARK = re.compile(r'class="lh-audit lh-audit--[a-z]+ lh-audit--(fail|average|pass)"')
+TITLE = re.compile(r'class="lh-audit__title"><span>(.*?)</span>', re.S)
+AUDIT_SPLIT = re.compile(r'<div class="lh-audit ')
+
+
+def read_html(path: Path) -> str:
+ """MHTML または HTML から HTML 本文を取り出す。"""
+ if path.suffix.lower() in {".mhtml", ".mht"}:
+ message = email.message_from_bytes(path.read_bytes())
+ for part in message.walk():
+ if part.get_content_type() == "text/html":
+ payload = part.get_payload(decode=True) or b""
+ charset = part.get_content_charset() or "utf-8"
+ return payload.decode(charset, "replace")
+ raise SystemExit(f"{path}: HTML パートが見つかりませんでした")
+ return path.read_text(encoding="utf-8", errors="replace")
+
+
+def text_of(raw: str) -> str:
+ return html.unescape(re.sub(r"<[^>]+>", "", raw)).strip()
+
+
+def audit_list(page: str) -> list[tuple[str, str]]:
+ """(種別, 見出し) を、見出しの重複を除いて返す。"""
+ rows: list[tuple[str, str]] = []
+ seen: set[str] = set()
+ for m in AUDIT_MARK.finditer(page):
+ kind = m.group(1)
+ if kind == "pass":
+ continue
+ t = TITLE.search(page, m.end())
+ if not t:
+ continue
+ name = text_of(t.group(1))
+ if name and name not in seen:
+ seen.add(name)
+ rows.append((kind, name))
+ return rows
+
+
+def audit_blocks(page: str) -> list[tuple[str, str]]:
+ """(見出し, ブロック) を、見出しの重複を除いて返す。"""
+ rows: list[tuple[str, str]] = []
+ seen: set[str] = set()
+ for block in AUDIT_SPLIT.split(page)[1:]:
+ m = TITLE.search(block)
+ if not m:
+ continue
+ name = text_of(m.group(1))
+ if name and name not in seen:
+ seen.add(name)
+ rows.append((name, block))
+ return rows
+
+
+def block_text(block: str) -> str:
+ """ブロックからタグを落とし、読みやすいテキストにする。"""
+ # AUDIT_SPLIT で分割したため、先頭に開始タグの残骸が残る。それを落とす。
+ block = re.sub(r"^[^>]*>", "", block, count=1)
+ text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", block, flags=re.S)
+ text = re.sub(r"<br\s*/?>", "\n", text)
+ text = re.sub(r"</(?:tr|p|div|li|h[1-6])>", "\n", text)
+ text = re.sub(r"</t[dh]>", " | ", text)
+ text = re.sub(r"<[^>]+>", "", text)
+ text = html.unescape(text)
+ text = re.sub(r"[ \t]+\n", "\n", text)
+ text = re.sub(r"\n{3,}", "\n\n", text)
+ return text.strip()
+
+
+def truncate(text: str, limit: int | None) -> str:
+ if limit is None or len(text) <= limit:
+ return text
+ return f"{text[:limit]}\n…(残り{len(text) - limit}文字を省略)"
+
+
+def main() -> None:
+ args = sys.argv[1:]
+ if not args or args[0] in {"-h", "--help"}:
+ print(__doc__.strip())
+ return
+
+ report = Path(args.pop(0))
+ limit = None if "--full" in args else DEFAULT_MAX
+ if "--max" in args:
+ i = args.index("--max")
+ limit = int(args[i + 1])
+ del args[i : i + 2]
+ targets = [a for a in args if not a.startswith("--")]
+
+ page = read_html(report)
+
+ if not targets:
+ print("# 失敗・要改善の監査項目")
+ for kind, name in audit_list(page):
+ print(f"[{kind}] {name}")
+ return
+
+ found = 0
+ for name, block in audit_blocks(page):
+ if not any(t in name for t in targets):
+ continue
+ print(f"\n===== {name} =====")
+ print(truncate(block_text(block), limit))
+ found += 1
+ if not found:
+ print("該当する監査項目が見つかりませんでした。", file=sys.stderr)
+ sys.exit(1)
+
+
+if __name__ == "__main__":
+ main()