Files

56 lines
2.0 KiB
Python

#!/usr/bin/env python3
"""Extracts MAT's "Problem Suspect 1" paragraph from a heap_Leak_Suspects.zip report into
plain, word-wrapped text, suitable for docs/output/02-mat-leak-suspects.txt.
Usage: extract-mat-summary.py <module-dir>
Expects <module-dir>/docs/output/heap_Leak_Suspects.zip to exist (produced by mat-report.sh).
"""
import sys
import re
import html
import textwrap
import zipfile
import tempfile
import os
def main(mod_dir: str) -> None:
zpath = os.path.join(mod_dir, "docs", "output", "heap_Leak_Suspects.zip")
out_path = os.path.join(mod_dir, "docs", "output", "02-mat-leak-suspects.txt")
with tempfile.TemporaryDirectory() as td:
with zipfile.ZipFile(zpath) as zf:
zf.extract("index.html", td)
content = open(os.path.join(td, "index.html"), encoding="utf-8", errors="ignore").read()
starts = [m.start() for m in re.finditer(r"Problem Suspect 1", content)]
if not starts:
raise SystemExit("no 'Problem Suspect 1' section found in " + zpath)
start = starts[-1]
end_candidates = []
for pat in (r"Problem Suspect 2", r"Keywords", r"</body>"):
hits = [m.start() for m in re.finditer(pat, content) if m.start() > start]
if hits:
end_candidates.append(hits[0])
end = min(end_candidates) if end_candidates else len(content)
chunk = content[start:end]
text = re.sub("<[^>]+>", " ", chunk)
text = html.unescape(text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r" *\n *", "\n", text).strip()
wrapped = "\n".join(textwrap.wrap(text, width=96))
header = (
"Eclipse MAT 1.17.0 batch report (org.eclipse.mat.api:suspects) on heap.hprof "
"-- Problem Suspect 1\n" + "=" * 96 + "\n\n"
)
with open(out_path, "w") as f:
f.write(header + wrapped + "\n")
print("wrote", out_path)
if __name__ == "__main__":
if len(sys.argv) != 2:
raise SystemExit("usage: extract-mat-summary.py <module-dir>")
main(sys.argv[1])