#!/usr/bin/env python3 """Extracts MAT's "Problem Suspect 1" paragraph from a heap_Leak_Suspects.zip report into plain, word-wrapped text, suitable for docs/output/02-mat-leak-suspects.txt. Usage: extract-mat-summary.py Expects /docs/output/heap_Leak_Suspects.zip to exist (produced by mat-report.sh). """ import sys import re import html import textwrap import zipfile import tempfile import os def main(mod_dir: str) -> None: zpath = os.path.join(mod_dir, "docs", "output", "heap_Leak_Suspects.zip") out_path = os.path.join(mod_dir, "docs", "output", "02-mat-leak-suspects.txt") with tempfile.TemporaryDirectory() as td: with zipfile.ZipFile(zpath) as zf: zf.extract("index.html", td) content = open(os.path.join(td, "index.html"), encoding="utf-8", errors="ignore").read() starts = [m.start() for m in re.finditer(r"Problem Suspect 1", content)] if not starts: raise SystemExit("no 'Problem Suspect 1' section found in " + zpath) start = starts[-1] end_candidates = [] for pat in (r"Problem Suspect 2", r"Keywords", r""): hits = [m.start() for m in re.finditer(pat, content) if m.start() > start] if hits: end_candidates.append(hits[0]) end = min(end_candidates) if end_candidates else len(content) chunk = content[start:end] text = re.sub("<[^>]+>", " ", chunk) text = html.unescape(text) text = re.sub(r"[ \t]+", " ", text) text = re.sub(r" *\n *", "\n", text).strip() wrapped = "\n".join(textwrap.wrap(text, width=96)) header = ( "Eclipse MAT 1.17.0 batch report (org.eclipse.mat.api:suspects) on heap.hprof " "-- Problem Suspect 1\n" + "=" * 96 + "\n\n" ) with open(out_path, "w") as f: f.write(header + wrapped + "\n") print("wrote", out_path) if __name__ == "__main__": if len(sys.argv) != 2: raise SystemExit("usage: extract-mat-summary.py ") main(sys.argv[1])