← Files PDF to Editable PowerPointARCHIVED FILE
skills/pdf-to-editable-pptx/scripts/audit_conversion.py
4.23 KB · Oct 5, 2026 · 18:33 UTC
#!/usr/bin/env python3
"""Audit basic PDF-to-PPTX fidelity and editability signals."""
from __future__ import annotations
import argparse
import json
import re
import sys
from difflib import SequenceMatcher
from pathlib import Path
def normalize(text: str) -> str:
return re.sub(r"\s+", " ", text).strip()
def pdf_pages(path: Path) -> list[str]:
try:
import fitz # PyMuPDF
with fitz.open(path) as doc:
return [page.get_text("text") for page in doc]
except ImportError:
try:
from pypdf import PdfReader
return [(page.extract_text() or "") for page in PdfReader(str(path)).pages]
except ImportError as exc:
raise RuntimeError("Install PyMuPDF or pypdf to inspect the source PDF") from exc
def pptx_slides(path: Path) -> list[dict]:
try:
from pptx import Presentation
from pptx.enum.shapes import MSO_SHAPE_TYPE
except ImportError as exc:
raise RuntimeError("Install python-pptx to inspect the output presentation") from exc
prs = Presentation(str(path))
result = []
slide_area = int(prs.slide_width) * int(prs.slide_height)
for index, slide in enumerate(prs.slides, start=1):
texts: list[str] = []
text_shapes = 0
pictures = 0
largest_picture_ratio = 0.0
for shape in slide.shapes:
if getattr(shape, "has_text_frame", False):
value = getattr(shape, "text", "")
if normalize(value):
text_shapes += 1
texts.append(value)
if shape.shape_type == MSO_SHAPE_TYPE.PICTURE:
pictures += 1
ratio = (int(shape.width) * int(shape.height)) / slide_area if slide_area else 0
largest_picture_ratio = max(largest_picture_ratio, ratio)
result.append(
{
"slide": index,
"text": "\n".join(texts),
"text_shapes": text_shapes,
"pictures": pictures,
"largest_picture_area_ratio": round(largest_picture_ratio, 4),
"likely_flattened": largest_picture_ratio >= 0.9 and text_shapes == 0,
}
)
return result
def build_report(pdf_text: list[str], slides: list[dict]) -> dict:
comparisons = []
for i in range(max(len(pdf_text), len(slides))):
source = normalize(pdf_text[i]) if i < len(pdf_text) else ""
output = normalize(slides[i]["text"]) if i < len(slides) else ""
ratio = SequenceMatcher(None, source, output, autojunk=False).ratio() if source or output else 1.0
entry = {
"page": i + 1,
"source_characters": len(source),
"output_characters": len(output),
"text_similarity": round(ratio, 6),
}
if i < len(slides):
entry.update({k: v for k, v in slides[i].items() if k != "text"})
comparisons.append(entry)
flattened = [x["page"] for x in comparisons if x.get("likely_flattened")]
mismatched = [x["page"] for x in comparisons if x["source_characters"] and x["text_similarity"] < 0.995]
return {
"pdf_pages": len(pdf_text),
"pptx_slides": len(slides),
"counts_match": len(pdf_text) == len(slides),
"likely_flattened_pages": flattened,
"text_review_pages": mismatched,
"pages": comparisons,
"note": "Rendered visual comparison is still required.",
}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("pdf", type=Path)
parser.add_argument("pptx", type=Path)
parser.add_argument("--json", type=Path, dest="json_path")
args = parser.parse_args()
for path in (args.pdf, args.pptx):
if not path.is_file():
parser.error(f"file not found: {path}")
try:
report = build_report(pdf_pages(args.pdf), pptx_slides(args.pptx))
except RuntimeError as exc:
print(f"error: {exc}", file=sys.stderr)
return 2
payload = json.dumps(report, ensure_ascii=False, indent=2)
print(payload)
if args.json_path:
args.json_path.write_text(payload + "\n", encoding="utf-8")
return 0 if report["counts_match"] else 1
if __name__ == "__main__":
raise SystemExit(main())
SHA-256: 41a61107bbebc32d633316173fb60cfd9fa9afaf5e175aaa8bbf700d481c426e