From a78d51f687ff35e1df7c84236b6ac5082b2910ef Mon Sep 17 00:00:00 2001 From: Anionex <1005128408@qq.com> Date: Wed, 5 Aug 2026 23:49:46 +0800 Subject: [PATCH] feat: add pdf_pages skill script to describe PDF pages via the vision model A self-contained script in the spirit of pixel_diff.py: renders selected PDF pages with poppler (pdftoppm/pdfinfo) and asks the glance CLI (installed by the toolkit install) to describe each page, so the script has no dependency on this repo's Python modules. Prints one Markdown '## Page N / M' section per page. Supports 1-based page ranges (-p 1-3,5), verbatim OCR (--ocr), custom questions (-q), render DPI, and keeping the rendered PNGs (--keep). Fails with a clear error instead of guessing when poppler or glance is missing. Adds skills/vision-tools/scripts/pdf_pages.py, tests/test_pdf_pages.py, and a pdf_pages section in the vision-tools SKILL.md. Robustness: --keep renders into a fresh per-run subdirectory with hard uniqueness (exist_ok=False plus collision retry) so stale pages cannot leak in; sparse page ranges render per contiguous run instead of the whole span; pdfinfo/pdftoppm/glance calls have timeouts with mitigation hints; --dpi is validated; mkdir failures surface as clean errors; empty page lists are rejected before rendering; tests guard on poppler and glance availability and report skipped groups. Tests cover page parsing, contiguous runs, glance command construction, real poppler rendering, timeouts, missing pages, keep failure, keep collisions, mode selection, exit codes, and missing-dependency errors. --- skills/vision-tools/SKILL.md | 31 +- skills/vision-tools/scripts/pdf_pages.py | 256 ++++++++++ tests/test_pdf_pages.py | 574 +++++++++++++++++++++++ 3 files changed, 859 insertions(+), 2 deletions(-) create mode 100644 skills/vision-tools/scripts/pdf_pages.py create mode 100644 tests/test_pdf_pages.py diff --git a/skills/vision-tools/SKILL.md b/skills/vision-tools/SKILL.md index 9ae0424..0c9b394 100644 --- a/skills/vision-tools/SKILL.md +++ b/skills/vision-tools/SKILL.md @@ -1,6 +1,6 @@ --- name: vision-tools -description: Local vision CLIs: glance (describe/ask/OCR an image), ground (locate a target, pixel box), detect (element inventory), trace (image to SVG geometry). Use for any task involving an image — questions, text, locating elements, comparing, rebuilding as HTML/SVG, digitizing a sketch or diagram, reading values off a chart, operating a GUI from screenshots — and to re-check an image yourself when a description you were given lacks a detail. +description: Local vision CLIs: glance (describe/ask/OCR an image), pdf_pages (describe a PDF page by page), ground (locate a target, pixel box), detect (element inventory), trace (image to SVG geometry). Use for any task involving an image or PDF — questions, text, locating elements, comparing, rebuilding as HTML/SVG, digitizing a sketch or diagram, reading values off a chart, operating a GUI from screenshots — and to re-check an image yourself when a description you were given lacks a detail. --- # vision-tools @@ -14,6 +14,7 @@ Pick the tool by the question you are answering: | Question | Tool | |---|---| | "What does this image show / say?" | `glance` | +| "What's in this PDF, page by page?" | `scripts/pdf_pages.py` | | "Where is X?" — a thing you can name | `ground` | | "Where are all the Xs?" — every instance of a kind | `detect` | | "What is its exact shape, size, offset?" | `trace` | @@ -53,6 +54,31 @@ badge or a small shift is a rounding error to a vision model and exact to `scripts/pixel_diff.py`. Diff first to get the box, then `glance --region` that box to read what the change actually is. +## `scripts/pdf_pages.py` — read a PDF page by page + +Renders a PDF's pages to PNG with poppler and describes each one with the +vision model in a single pass, so a text-only agent can read a deck or +document without an extra PDF-to-image step. The script path below is +relative to this skill's directory, like the other scripts here; the PDF +argument itself is resolved from your current working directory: + +```bash +python3 scripts/pdf_pages.py deck.pdf # describe every page +python3 scripts/pdf_pages.py deck.pdf -p 1-3,5,7-9 # only those pages (1-based) +python3 scripts/pdf_pages.py deck.pdf --ocr # verbatim transcription per page +python3 scripts/pdf_pages.py deck.pdf -q "What is the headline on each page?" +python3 scripts/pdf_pages.py deck.pdf --dpi 150 # higher render resolution +python3 scripts/pdf_pages.py deck.pdf --keep work/pages/ # keep rendered PNGs (one fresh subdir per run) +``` + +Output is Markdown with one `## Page N / M` section per page, so you can +quote page numbers back to the user. The script is self-contained like +`pixel_diff.py`: it renders pages with poppler (`pdftoppm` + `pdfinfo`, +install with e.g. `brew install poppler`) and asks the `glance` CLI to +describe each page. It reports a clear error instead of guessing when +either is missing. Page ranges are 1-based and must fit the PDF's page +count. + ## ground — locate a named target ```bash @@ -190,7 +216,8 @@ sequence, and how to tell you got it right. ## Notes -- Only PNG / JPEG / GIF / WebP images are supported. +- Only PNG / JPEG / GIF / WebP images are supported. PDFs are handled by + `scripts/pdf_pages.py`, which renders pages to PNG first (requires poppler). - If a command is not found, the optional tools were not installed — report this to the user instead of improvising a replacement. - If the vision API fails, relay the error faithfully; never fabricate diff --git a/skills/vision-tools/scripts/pdf_pages.py b/skills/vision-tools/scripts/pdf_pages.py new file mode 100644 index 0000000..2a7a5c0 --- /dev/null +++ b/skills/vision-tools/scripts/pdf_pages.py @@ -0,0 +1,256 @@ +#!/usr/bin/env python3 +"""pdf_pages: render a PDF's pages and describe each one via the glance CLI. + +A thin local script in the same spirit as pixel_diff.py: it orchestrates +existing tools instead of depending on this repo's Python modules. It +renders the selected pages with poppler's pdftoppm and lets the glance CLI +(already installed by the toolkit install) describe each page, so the +script itself stays self-contained. + +Run from this skill's directory (paths are relative to it, like the other +scripts here): + + python3 scripts/pdf_pages.py deck.pdf -p 1-3,5,7-9 --ocr + +Requires poppler (pdftoppm + pdfinfo) and the glance CLI on PATH; fails +with a clear error instead of guessing when either is missing. +""" + +from __future__ import annotations + +import argparse +import re +import shutil +import subprocess +import sys +import tempfile +import time +from contextlib import ExitStack +from pathlib import Path + + +class ScriptError(RuntimeError): + """A safe, user-facing script failure.""" + + +_PAGE_SPEC_TOKEN = re.compile(r"^\s*(\d+)(?:\s*-\s*(\d+))?\s*$") + +DEFAULT_DESCRIBE_PROMPT = "Describe the contents of this page in detail." + + +def require_poppler() -> tuple[str, str]: + pdftoppm = shutil.which("pdftoppm") + pdfinfo = shutil.which("pdfinfo") + if not pdftoppm or not pdfinfo: + raise ScriptError( + "pdf_pages requires poppler's pdftoppm and pdfinfo; " + "install poppler first, e.g. 'brew install poppler'" + ) + return pdftoppm, pdfinfo + + +def require_glance() -> str: + glance = shutil.which("glance") + if not glance: + raise ScriptError( + "pdf_pages requires the glance CLI on PATH; " + "install the agent-vision-toolkit CLIs first (see AGENT_INSTALL.md)" + ) + return glance + + +def pdf_page_count(pdf_path: Path, pdfinfo: str) -> int: + try: + proc = subprocess.run([pdfinfo, str(pdf_path)], capture_output=True, text=True, timeout=30) + except subprocess.TimeoutExpired as exc: + raise ScriptError(f"pdfinfo timed out on {pdf_path}") from exc + if proc.returncode != 0: + raise ScriptError(f"pdfinfo failed on {pdf_path}: {proc.stderr.strip() or proc.stdout.strip()}") + for line in proc.stdout.splitlines(): + if line.lower().startswith("pages:"): + try: + return int(line.split(":", 1)[1].strip()) + except ValueError: + break + raise ScriptError(f"pdfinfo did not report a page count for {pdf_path}") + + +def parse_page_spec(spec: str | None, total: int) -> list[int]: + """Parse a 1-based page range like '1-3,5,7-9' into an ordered, deduplicated page list.""" + if spec is None: + return list(range(1, total + 1)) + pages: list[int] = [] + for token in spec.split(","): + match = _PAGE_SPEC_TOKEN.match(token) + if not match: + raise ScriptError(f"Invalid page range {token!r} (expected forms like '3', '1-3', or '1,3-5')") + start = int(match.group(1)) + end = int(match.group(2) or match.group(1)) + if start > end: + raise ScriptError(f"Invalid page range {token!r}: start is after end") + if start < 1 or end > total: + raise ScriptError(f"Page range {token!r} is out of range for a {total}-page PDF") + pages.extend(range(start, end + 1)) + return sorted(set(pages)) + + +def _contiguous_runs(pages: list[int]) -> list[tuple[int, int]]: + """Split a sorted page list into (start, end) contiguous runs.""" + if not pages: + return [] + runs: list[tuple[int, int]] = [] + start = prev = pages[0] + for number in pages[1:]: + if number == prev + 1: + prev = number + else: + runs.append((start, prev)) + start = prev = number + runs.append((start, prev)) + return runs + + +def render_pages(pdf_path: Path, pages: list[int], dpi: int, out_dir: Path, + pdftoppm: str) -> list[Path]: + # Render each contiguous run separately so a sparse selection like + # -p 1,600 never renders the 598 pages in between. + produced: dict[int, Path] = {} + for start, end in _contiguous_runs(pages): + prefix = out_dir / "page" + try: + proc = subprocess.run( + [pdftoppm, "-png", "-r", str(dpi), "-f", str(start), "-l", str(end), + str(pdf_path), str(prefix)], + capture_output=True, + text=True, + timeout=120, + ) + except subprocess.TimeoutExpired as exc: + raise ScriptError( + f"pdftoppm timed out on {pdf_path} (pages {start}-{end}); " + "try a smaller --pages range or a lower --dpi" + ) from exc + if proc.returncode != 0: + raise ScriptError( + f"pdftoppm failed on {pdf_path} (pages {start}-{end}): " + f"{proc.stderr.strip() or proc.stdout.strip()}" + ) + for image in out_dir.glob("page-*.png"): + try: + produced[int(image.stem.split("-")[1])] = image + except (IndexError, ValueError): + continue + missing = [n for n in pages if n not in produced] + if missing: + raise ScriptError(f"pdftoppm output is missing pages {missing}") + return [produced[n] for n in pages] + + +def build_page_prompt(page_no: int, total: int, mode: str, query: str | None) -> str: + context = f"Page {page_no} of {total} in this PDF document." + if mode == "query": + return f"{context}\n{query}" + return f"{context} {DEFAULT_DESCRIBE_PROMPT}" + + +def glance_command(image_path: Path, mode: str, prompt: str | None, + ocr_extra: str | None) -> list[str]: + if mode == "ocr": + command = ["glance", str(image_path), "--ocr"] + if ocr_extra: + command.append(ocr_extra) + return command + return ["glance", str(image_path), "-q", prompt] + + +def describe_pages(image_paths: list[Path], pages: list[int], total: int, + mode: str, query: str | None, ocr_extra: str | None, + glance: str) -> str: + sections = [] + for page_no, image_path in zip(pages, image_paths): + prompt = build_page_prompt(page_no, total, mode, query) if mode != "ocr" else None + command = glance_command(image_path, mode, prompt, ocr_extra) + command[0] = glance + try: + proc = subprocess.run(command, capture_output=True, text=True, timeout=600) + except subprocess.TimeoutExpired as exc: + raise ScriptError(f"glance timed out on page {page_no}") from exc + if proc.returncode != 0: + raise ScriptError( + f"glance failed on page {page_no}: {proc.stderr.strip() or proc.stdout.strip()}" + ) + answer = proc.stdout.strip() + if not answer: + raise ScriptError(f"glance returned no output for page {page_no}") + sections.append(f"## Page {page_no} / {total}\n\n{answer}\n") + return "\n".join(sections) + + +def main() -> None: + parser = argparse.ArgumentParser( + prog="pdf_pages", + description="Render PDF pages to images and describe each page via the glance CLI", + ) + parser.add_argument("pdf", type=Path, help="path to the PDF file") + parser.add_argument("-p", "--pages", metavar="SPEC", + help="1-based page range to process, e.g. '1-3,5,7-9' (default: all pages)") + group = parser.add_mutually_exclusive_group() + group.add_argument("-q", "--query", help="ask a question about every selected page") + group.add_argument("--ocr", nargs="?", const="", metavar="EXTRA", + help="transcribe every selected page's text verbatim") + parser.add_argument("--dpi", type=int, default=100, + help="render resolution in dots per inch (default: 100)") + parser.add_argument("--keep", type=Path, + help="keep rendered page PNGs in this directory instead of a temporary one") + args = parser.parse_args() + if not 1 <= args.dpi <= 600: + parser.exit(2, f"pdf_pages: --dpi must be between 1 and 600\n") + try: + pdftoppm, pdfinfo = require_poppler() + glance = require_glance() + pdf = args.pdf.expanduser() + if not pdf.is_file(): + raise ScriptError(f"PDF not found: {pdf}") + total = pdf_page_count(pdf, pdfinfo) + pages = parse_page_spec(args.pages, total) + if not pages: + raise ScriptError(f"PDF has no pages to process: {pdf}") + mode = "ocr" if args.ocr is not None else ("query" if args.query else "describe") + with ExitStack() as stack: + if args.keep: + keep_base = Path(args.keep).expanduser() + try: + keep_base.mkdir(parents=True, exist_ok=True) + except OSError as exc: + raise ScriptError(f"Cannot use --keep directory {keep_base}: {exc}") from exc + # A fresh subdirectory per run keeps stale renders from an + # earlier PDF out of this run's page glob; exist_ok=False + # makes that uniqueness a hard guarantee. + stamp = f"{pdf.stem}-{time.strftime('%Y%m%d-%H%M%S')}-{time.time_ns() % 1_000_000:06d}" + try: + for attempt in range(3): + out_dir = keep_base / (stamp if attempt == 0 else f"{stamp}-{attempt}") + try: + out_dir.mkdir() + except FileExistsError: + continue + break + else: + raise ScriptError( + f"Cannot create a fresh --keep subdirectory under {keep_base} " + f"(names {stamp}, {stamp}-1, {stamp}-2 all exist)" + ) + except OSError as exc: + raise ScriptError(f"Cannot use --keep directory {keep_base}: {exc}") from exc + else: + out_dir = Path(stack.enter_context(tempfile.TemporaryDirectory(prefix="pdf_pages-"))) + images = render_pages(pdf, pages, args.dpi, out_dir, pdftoppm) + if args.keep: + print(f"Rendered pages kept in {out_dir}", file=sys.stderr) + print(describe_pages(images, pages, total, mode, args.query, args.ocr, glance)) + except ScriptError as exc: + parser.exit(1, f"pdf_pages: {exc}\n") + + +if __name__ == "__main__": + main() diff --git a/tests/test_pdf_pages.py b/tests/test_pdf_pages.py new file mode 100644 index 0000000..9290215 --- /dev/null +++ b/tests/test_pdf_pages.py @@ -0,0 +1,574 @@ +#!/usr/bin/env python3 +"""Unit test: pdf_pages rendering and per-page glance orchestration.""" + +import contextlib +import importlib.machinery +import importlib.util +import io +import os +import shutil +import subprocess +import sys +import tempfile +import types +from pathlib import Path + +REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +SCRIPT = os.path.join(REPO, "skills", "vision-tools", "scripts", "pdf_pages.py") + +SKIPPED: list[str] = [] + + +def load_module(): + spec = importlib.util.spec_from_loader( + "pdf_pages_mod", + importlib.machinery.SourceFileLoader("pdf_pages_mod", SCRIPT)) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +def note_skip(label: str, reason: str) -> None: + SKIPPED.append(f"{label} ({reason})") + print(f"SKIP {label}: {reason}") + + +def check(cond: bool, message: str) -> None: + if not cond: + raise AssertionError(message) + + +def poppler_available() -> bool: + return bool(shutil.which("pdftoppm") and shutil.which("pdfinfo")) + + +def fake_glance_run(real_run, calls=None, stdout="page description", returncode=0): + """Wrap subprocess.run: fake glance calls, pass poppler calls through.""" + def wrapper(cmd, **kwargs): + if str(cmd[0]).endswith("glance"): + if calls is not None: + calls.append(list(cmd)) + return types.SimpleNamespace(returncode=returncode, stdout=stdout, stderr="") + return real_run(cmd, **kwargs) + return wrapper + + +def test_parse_page_spec(mod) -> None: + check(mod.parse_page_spec(None, 10) == list(range(1, 11)), "None means all pages") + check(mod.parse_page_spec("1-3", 10) == [1, 2, 3], "simple range") + check(mod.parse_page_spec("1,3,5-7", 10) == [1, 3, 5, 6, 7], "mixed spec") + check(mod.parse_page_spec("5,1-2", 10) == [1, 2, 5], "dedupe and sort") + check(mod.parse_page_spec("2-2", 10) == [2], "single-token range") + for bad in ("a", "1-", "-3", "3-1", "0", "11", "1-11", "1,,2"): + try: + mod.parse_page_spec(bad, 10) + except mod.ScriptError: + pass + else: + raise AssertionError(f"spec {bad!r} must be rejected") + print("PARSE PAGE SPEC PASS") + + +def test_contiguous_runs(mod) -> None: + check(mod._contiguous_runs([]) == [], "empty input") + check(mod._contiguous_runs([1, 2, 3]) == [(1, 3)], "one run") + check(mod._contiguous_runs([1, 3]) == [(1, 1), (3, 3)], "singletons") + check(mod._contiguous_runs([1, 2, 5, 6, 9]) == [(1, 2), (5, 6), (9, 9)], "mixed") + print("CONTIGUOUS RUNS PASS") + + +def test_build_page_prompt(mod) -> None: + query = mod.build_page_prompt(2, 10, "query", "What is the title?") + check("What is the title?" in query and "Page 2 of 10" in query, "query prompt") + describe = mod.build_page_prompt(2, 10, "describe", None) + check("Describe" in describe and "Page 2 of 10" in describe, "describe prompt") + print("BUILD PAGE PROMPT PASS") + + +def test_glance_command(mod) -> None: + png = Path("page-1.png") + ocr = mod.glance_command(png, "ocr", None, None) + check(ocr == ["glance", "page-1.png", "--ocr"], "ocr command shape") + ocr_extra = mod.glance_command(png, "ocr", None, "ignore footers") + check(ocr_extra == ["glance", "page-1.png", "--ocr", "ignore footers"], "ocr extra appended") + query = mod.glance_command(png, "query", "Page 1 of 5. What title?", None) + check(query == ["glance", "page-1.png", "-q", "Page 1 of 5. What title?"], "query command shape") + print("GLANCE COMMAND PASS") + + +def test_render_pages(mod) -> None: + if not poppler_available(): + note_skip("RENDER", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("RENDER", "Pillow not installed") + return + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "fixture.pdf" + pages = [Image.new("RGB", (100, 140), (200, 30, 30)), + Image.new("RGB", (100, 140), (30, 200, 30)), + Image.new("RGB", (100, 140), (30, 30, 200))] + pages[0].save(pdf, save_all=True, append_images=pages[1:]) + out_dir = Path(raw) / "out" + out_dir.mkdir() + rendered = mod.render_pages(pdf, [1, 3], 80, out_dir, shutil.which("pdftoppm")) + check(len(rendered) == 2, "two pages rendered") + check([int(p.stem.split("-")[1]) for p in rendered] == [1, 3], "page numbers match") + check(not (out_dir / "page-2.png").exists(), + "sparse selection must not render pages in between") + with Image.open(rendered[0]) as image: + check(abs(image.size[0] - round(100 * 80 / 72)) <= 1 + and abs(image.size[1] - round(140 * 80 / 72)) <= 1, + f"size {image.size} should follow the DPI") + print("RENDER PAGES PASS") + + +def test_describe_pages_flow(mod) -> None: + calls = [] + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run, calls=calls) + try: + with tempfile.TemporaryDirectory() as raw: + pngs = [] + for number in (1, 2): + path = Path(raw) / f"page-{number}.png" + path.write_bytes(b"x") + pngs.append(path) + text = mod.describe_pages(pngs, [1, 2], 5, "describe", None, None, "/path/glance") + finally: + mod.subprocess.run = original_run + check(len(calls) == 2, "one glance call per page") + check(all(cmd[0] == "/path/glance" and cmd[2] == "-q" for cmd in calls), + "calls run through the resolved glance path") + check("Page 1 of 5" in calls[0][3] and "Page 2 of 5" in calls[1][3], "page context in prompts") + check("## Page 1 / 5" in text and "## Page 2 / 5" in text, "markdown sections") + check("page description" in text, "answers included") + print("DESCRIBE PAGES FLOW PASS") + + +def test_describe_glance_failure(mod) -> None: + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run, returncode=1, stdout="") + try: + with tempfile.TemporaryDirectory() as raw: + png = Path(raw) / "page-1.png" + png.write_bytes(b"x") + try: + mod.describe_pages([png], [1], 3, "describe", None, None, "glance") + except mod.ScriptError as exc: + check("glance failed on page 1" in str(exc), "glance failure surfaces per page") + else: + raise AssertionError("failing glance must raise ScriptError") + finally: + mod.subprocess.run = original_run + print("DESCRIBE GLANCE FAILURE PASS") + + +def test_main_flow(mod) -> None: + if not shutil.which("glance"): + note_skip("MAIN FLOW", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("MAIN FLOW", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("MAIN FLOW", "Pillow not installed") + return + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run) + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "fixture.pdf" + images = [Image.new("RGB", (60, 60), (10, 10, 10)) for _ in range(3)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf), "-p", "1,3", "--dpi", "80"] + buffer = io.StringIO() + try: + with contextlib.redirect_stdout(buffer): + mod.main() + finally: + sys.argv = old_argv + finally: + mod.subprocess.run = original_run + text = buffer.getvalue() + check("## Page 1 / 3" in text and "## Page 3 / 3" in text, "sections for selected pages") + check("page description" in text, "descriptions included") + check("## Page 2 / 3" not in text, "unselected page skipped") + print("MAIN FLOW PASS") + + +def test_pdf_page_count(mod) -> None: + try: + from PIL import Image + except ImportError: + note_skip("PAGE COUNT", "Pillow not installed") + return + if not poppler_available(): + note_skip("PAGE COUNT", "poppler (pdftoppm/pdfinfo) not installed") + return + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "fixture.pdf" + images = [Image.new("RGB", (30, 30), (0, 0, 0)) for _ in range(3)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + check(mod.pdf_page_count(pdf, shutil.which("pdfinfo")) == 3, "page count parsed") + broken = Path(raw) / "broken.pdf" + broken.write_bytes(b"%PDF-1.3\n%%EOF") + try: + mod.pdf_page_count(broken, shutil.which("pdfinfo")) + except mod.ScriptError: + pass + else: + raise AssertionError("broken PDF must raise ScriptError") + fake = Path(raw) / "fake-pdfinfo" + fake.write_text("#!/bin/sh\necho 'Title: whatever'\nexit 0\n") + fake.chmod(0o755) + try: + mod.pdf_page_count(pdf, str(fake)) + except mod.ScriptError: + pass + else: + raise AssertionError("pdfinfo without a Pages: line must raise ScriptError") + print("PDF PAGE COUNT PASS") + + +def test_render_pages_failure(mod) -> None: + original_run = mod.subprocess.run + try: + mod.subprocess.run = lambda *args, **kwargs: types.SimpleNamespace( + returncode=1, stdout="", stderr="boom") + with tempfile.TemporaryDirectory() as raw: + out_dir = Path(raw) + try: + mod.render_pages(Path(raw) / "x.pdf", [1], 80, out_dir, "pdftoppm") + except mod.ScriptError as exc: + check("boom" in str(exc), "pdftoppm stderr surfaces") + else: + raise AssertionError("failing pdftoppm must raise ScriptError") + finally: + mod.subprocess.run = original_run + print("RENDER PAGES FAILURE PASS") + + +def test_subprocess_timeouts(mod) -> None: + captured: dict = {} + + def timeout_run(*args, **kwargs): + captured["timeout"] = kwargs.get("timeout") + raise subprocess.TimeoutExpired(cmd=args[0], timeout=captured["timeout"]) + + original_run = mod.subprocess.run + mod.subprocess.run = timeout_run + try: + try: + mod.pdf_page_count(Path("x.pdf"), "pdfinfo") + except mod.ScriptError as exc: + check("timed out" in str(exc), "pdfinfo timeout surfaces as ScriptError") + else: + raise AssertionError("pdfinfo timeout must raise ScriptError") + check(captured["timeout"] == 30, "pdfinfo timeout is 30s") + with tempfile.TemporaryDirectory() as raw: + try: + mod.render_pages(Path("x.pdf"), [1, 2], 80, Path(raw), "pdftoppm") + except mod.ScriptError as exc: + check("timed out" in str(exc) and "--dpi" in str(exc), + "pdftoppm timeout carries a mitigation hint") + else: + raise AssertionError("pdftoppm timeout must raise ScriptError") + check(captured["timeout"] == 120, "pdftoppm timeout is 120s") + with tempfile.TemporaryDirectory() as raw: + png = Path(raw) / "page-1.png" + png.write_bytes(b"x") + try: + mod.describe_pages([png], [1], 3, "describe", None, None, "glance") + except mod.ScriptError as exc: + check("glance timed out on page 1" in str(exc), "glance timeout surfaces") + else: + raise AssertionError("glance timeout must raise ScriptError") + check(captured["timeout"] == 600, "glance timeout is 600s") + finally: + mod.subprocess.run = original_run + print("SUBPROCESS TIMEOUTS PASS") + + +def test_render_missing_pages(mod) -> None: + original_run = mod.subprocess.run + try: + mod.subprocess.run = lambda *args, **kwargs: types.SimpleNamespace( + returncode=0, stdout="", stderr="") + with tempfile.TemporaryDirectory() as raw: + try: + mod.render_pages(Path(raw) / "x.pdf", [1, 2], 80, Path(raw), "pdftoppm") + except mod.ScriptError as exc: + check("missing pages [1, 2]" in str(exc), "missing-page check fires") + else: + raise AssertionError("no output files must raise ScriptError") + finally: + mod.subprocess.run = original_run + print("RENDER MISSING PAGES PASS") + + +def test_require_missing(mod) -> None: + original_which = mod.shutil.which + try: + mod.shutil.which = lambda name: None + for require, name in ((mod.require_poppler, "poppler"), (mod.require_glance, "glance")): + try: + require() + except mod.ScriptError as exc: + check(name in str(exc), f"missing {name} surfaces clearly") + else: + raise AssertionError(f"missing {name} must raise ScriptError") + finally: + mod.shutil.which = original_which + print("REQUIRE MISSING PASS") + + +def test_dpi_invalid(mod) -> None: + for bad in ("0", "601"): + old_argv = sys.argv + sys.argv = ["pdf_pages.py", "deck.pdf", "--dpi", bad] + try: + with contextlib.redirect_stdout(io.StringIO()): + mod.main() + except SystemExit as exc: + check(exc.code == 2, f"--dpi {bad} exits 2") + else: + raise AssertionError(f"--dpi {bad} must exit nonzero") + finally: + sys.argv = old_argv + print("DPI INVALID PASS") + + +def test_main_keep_flow(mod) -> None: + if not shutil.which("glance"): + note_skip("KEEP FLOW", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("KEEP FLOW", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("KEEP FLOW", "Pillow not installed") + return + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run) + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "deck.pdf" + images = [Image.new("RGB", (40, 40), (0, 0, 0)) for _ in range(2)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + keep_dir = Path(raw) / "keep" + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf), "--keep", str(keep_dir), "--dpi", "80"] + buffer = io.StringIO() + try: + with contextlib.redirect_stdout(buffer): + mod.main() + finally: + sys.argv = old_argv + check("page description" in buffer.getvalue(), "keep flow describes pages") + subs = [p for p in keep_dir.iterdir() if p.is_dir()] + check(len(subs) == 1, "one fresh subdirectory per run") + pages = sorted(subs[0].glob("page-*.png")) + check(len(pages) == 2, "rendered pages kept inside the fresh subdirectory") + finally: + mod.subprocess.run = original_run + print("MAIN KEEP FLOW PASS") + + +def test_keep_dir_failure(mod) -> None: + if not shutil.which("glance"): + note_skip("KEEP FAILURE", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("KEEP FAILURE", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("KEEP FAILURE", "Pillow not installed") + return + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run) + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "deck.pdf" + images = [Image.new("RGB", (40, 40), (0, 0, 0)) for _ in range(2)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + keep_dir = Path(raw) / "keep" + keep_dir.mkdir() + os.chmod(keep_dir, 0o555) + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf), "--keep", str(keep_dir), "--dpi", "80"] + try: + with contextlib.redirect_stdout(io.StringIO()): + mod.main() + except SystemExit as exc: + check(exc.code == 1, "unusable --keep directory exits 1") + else: + raise AssertionError("unusable --keep directory must fail cleanly") + finally: + sys.argv = old_argv + os.chmod(keep_dir, 0o755) + finally: + mod.subprocess.run = original_run + print("KEEP DIR FAILURE PASS") + + +def test_keep_collision(mod) -> None: + if not shutil.which("glance"): + note_skip("KEEP COLLISION", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("KEEP COLLISION", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("KEEP COLLISION", "Pillow not installed") + return + original_run = mod.subprocess.run + original_strftime = mod.time.strftime + original_ns = mod.time.time_ns + mod.subprocess.run = fake_glance_run(original_run) + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "deck.pdf" + images = [Image.new("RGB", (40, 40), (0, 0, 0)) for _ in range(2)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + keep = Path(raw) / "keep" + keep.mkdir() + mod.time.strftime = lambda fmt: "20260806-000000" + mod.time.time_ns = lambda: 123456 + (keep / "deck-20260806-000000-123456").mkdir() # pre-existing collision + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf), "--keep", str(keep), "--dpi", "80"] + try: + with contextlib.redirect_stdout(io.StringIO()): + mod.main() + finally: + sys.argv = old_argv + fallback = keep / "deck-20260806-000000-123456-1" + check(fallback.is_dir(), "collision falls back to a -1 suffix") + check(len(list(fallback.glob("page-*.png"))) == 2, + "pages rendered into the collision fallback directory") + finally: + mod.subprocess.run = original_run + mod.time.strftime = original_strftime + mod.time.time_ns = original_ns + print("KEEP COLLISION PASS") + + +def test_main_modes(mod) -> None: + if not shutil.which("glance"): + note_skip("MODES", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("MODES", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("MODES", "Pillow not installed") + return + calls = [] + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run, calls=calls, stdout="mode description") + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "deck.pdf" + images = [Image.new("RGB", (40, 40), (0, 0, 0)) for _ in range(2)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + for argv in (["--ocr"], ["--ocr", "ignore footers"], ["-q", "What title?"]): + calls.clear() + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf)] + argv + ["--dpi", "80"] + try: + with contextlib.redirect_stdout(io.StringIO()): + mod.main() + finally: + sys.argv = old_argv + check(len(calls) == 2, f"one glance call per page for {argv}") + if argv[0] == "--ocr": + check(all("--ocr" in cmd for cmd in calls), f"ocr flag passed for {argv}") + if len(argv) > 1: + check(all(cmd[-1] == "ignore footers" for cmd in calls), + "ocr extra reaches glance") + else: + check(all(cmd[2] == "-q" and "What title?" in cmd[3] for cmd in calls), + "query text reaches every page prompt") + finally: + mod.subprocess.run = original_run + print("MAIN MODES PASS") + + +def test_main_exit_code(mod) -> None: + if not shutil.which("glance"): + note_skip("EXIT CODE", "glance CLI not on PATH") + return + if not poppler_available(): + note_skip("EXIT CODE", "poppler (pdftoppm/pdfinfo) not installed") + return + try: + from PIL import Image + except ImportError: + note_skip("EXIT CODE", "Pillow not installed") + return + original_run = mod.subprocess.run + mod.subprocess.run = fake_glance_run(original_run) + try: + with tempfile.TemporaryDirectory() as raw: + pdf = Path(raw) / "deck.pdf" + images = [Image.new("RGB", (40, 40), (0, 0, 0)) for _ in range(2)] + images[0].save(pdf, save_all=True, append_images=images[1:]) + old_argv = sys.argv + sys.argv = ["pdf_pages.py", str(pdf), "-p", "99", "--dpi", "80"] + try: + with contextlib.redirect_stdout(io.StringIO()): + mod.main() + except SystemExit as exc: + check(exc.code == 1, "script errors exit 1") + else: + raise AssertionError("out-of-range pages must exit nonzero") + finally: + sys.argv = old_argv + finally: + mod.subprocess.run = original_run + print("MAIN EXIT CODE PASS") + + +def main() -> None: + mod = load_module() + test_parse_page_spec(mod) + test_contiguous_runs(mod) + test_build_page_prompt(mod) + test_glance_command(mod) + test_render_pages(mod) + test_describe_pages_flow(mod) + test_describe_glance_failure(mod) + test_main_flow(mod) + test_pdf_page_count(mod) + test_render_pages_failure(mod) + test_subprocess_timeouts(mod) + test_render_missing_pages(mod) + test_require_missing(mod) + test_dpi_invalid(mod) + test_main_keep_flow(mod) + test_keep_dir_failure(mod) + test_keep_collision(mod) + test_main_modes(mod) + test_main_exit_code(mod) + if SKIPPED: + print("PDF PAGES TEST PASS (SKIPPED: " + "; ".join(SKIPPED) + ")") + else: + print("PDF PAGES TEST PASS") + + +if __name__ == "__main__": + main()