1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
|
"""PageSpeed Insights のレポートから、必要な監査項目だけを抜き出す。
PSI のレポートを丸ごと読むと、それだけで会話がコンテキストオーバーになる。
このスクリプトは、失敗・要改善の項目の見出しだけ、あるいは指定した項目の
本文(説明と表)だけを取り出し、既定で1件1200文字に切り詰めて出力する。
- 引数にレポートだけを渡すと、失敗・要改善の監査項目を見出しだけ一覧表示する
- 見出しの一部分を続けて渡すと、その項目の本文だけを表示する(部分一致)
- レポートは MHTML(.mhtml)でも HTML(.html)でもよい
- PSI のレポートはモバイル版とデスクトップ版が並ぶため、同じ見出しは1件だけ出す
- pass の項目は出力しない(コンテキスト節約のため)
使い方:
python scripts/psi-extract.py <report.mhtml|report.html>
python scripts/psi-extract.py <report> 画像 DOM
python scripts/psi-extract.py <report> 画像 --full
python scripts/psi-extract.py <report> 画像 --max 3000
"""
import email
import html
import re
import sys
from pathlib import Path
sys.stdout.reconfigure(encoding="utf-8")
# 1件あたりの出力上限(文字数)。
DEFAULT_MAX = 1200
AUDIT_MARK = re.compile(r'class="lh-audit lh-audit--[a-z]+ lh-audit--(fail|average|pass)"')
TITLE = re.compile(r'class="lh-audit__title"><span>(.*?)</span>', re.S)
AUDIT_SPLIT = re.compile(r'<div class="lh-audit ')
def read_html(path: Path) -> str:
"""MHTML または HTML から HTML 本文を取り出す。"""
if path.suffix.lower() in {".mhtml", ".mht"}:
message = email.message_from_bytes(path.read_bytes())
for part in message.walk():
if part.get_content_type() == "text/html":
payload = part.get_payload(decode=True) or b""
charset = part.get_content_charset() or "utf-8"
return payload.decode(charset, "replace")
raise SystemExit(f"{path}: HTML パートが見つかりませんでした")
return path.read_text(encoding="utf-8", errors="replace")
def text_of(raw: str) -> str:
return html.unescape(re.sub(r"<[^>]+>", "", raw)).strip()
def audit_list(page: str) -> list[tuple[str, str]]:
"""(種別, 見出し) を、見出しの重複を除いて返す。"""
rows: list[tuple[str, str]] = []
seen: set[str] = set()
for m in AUDIT_MARK.finditer(page):
kind = m.group(1)
if kind == "pass":
continue
t = TITLE.search(page, m.end())
if not t:
continue
name = text_of(t.group(1))
if name and name not in seen:
seen.add(name)
rows.append((kind, name))
return rows
def audit_blocks(page: str) -> list[tuple[str, str]]:
"""(見出し, ブロック) を、見出しの重複を除いて返す。"""
rows: list[tuple[str, str]] = []
seen: set[str] = set()
for block in AUDIT_SPLIT.split(page)[1:]:
m = TITLE.search(block)
if not m:
continue
name = text_of(m.group(1))
if name and name not in seen:
seen.add(name)
rows.append((name, block))
return rows
def block_text(block: str) -> str:
"""ブロックからタグを落とし、読みやすいテキストにする。"""
# AUDIT_SPLIT で分割したため、先頭に開始タグの残骸が残る。それを落とす。
block = re.sub(r"^[^>]*>", "", block, count=1)
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", block, flags=re.S)
text = re.sub(r"<br\s*/?>", "\n", text)
text = re.sub(r"</(?:tr|p|div|li|h[1-6])>", "\n", text)
text = re.sub(r"</t[dh]>", " | ", text)
text = re.sub(r"<[^>]+>", "", text)
text = html.unescape(text)
text = re.sub(r"[ \t]+\n", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
def truncate(text: str, limit: int | None) -> str:
if limit is None or len(text) <= limit:
return text
return f"{text[:limit]}\n…(残り{len(text) - limit}文字を省略)"
def main() -> None:
args = sys.argv[1:]
if not args or args[0] in {"-h", "--help"}:
print(__doc__.strip())
return
report = Path(args.pop(0))
limit = None if "--full" in args else DEFAULT_MAX
if "--max" in args:
i = args.index("--max")
limit = int(args[i + 1])
del args[i : i + 2]
targets = [a for a in args if not a.startswith("--")]
page = read_html(report)
if not targets:
print("# 失敗・要改善の監査項目")
for kind, name in audit_list(page):
print(f"[{kind}] {name}")
return
found = 0
for name, block in audit_blocks(page):
if not any(t in name for t in targets):
continue
print(f"\n===== {name} =====")
print(truncate(block_text(block), limit))
found += 1
if not found:
print("該当する監査項目が見つかりませんでした。", file=sys.stderr)
sys.exit(1)
if __name__ == "__main__":
main()
|