diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 548b40f22..b08e4df79 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -50,7 +50,7 @@ jobs: runs-on: ubuntu-latest strategy: matrix: - python-version: ["3.10", "3.12"] + python-version: ["3.14"] steps: - uses: actions/checkout@v6 diff --git a/README.md b/README.md index 0459f57cf..34afb8a87 100644 --- a/README.md +++ b/README.md @@ -127,7 +127,7 @@ Every system ran on the same harness with the same model and budgets, scored by | Requirement | Minimum | Check | Install | |---|---|---|---| -| Python | 3.10+ | `python --version` | [python.org](https://www.python.org/downloads/) | +| Python | 3.14+ | `python --version` | [python.org](https://www.python.org/downloads/) | | uv *(recommended)* | any | `uv --version` | `curl -LsSf https://astral.sh/uv/install.sh \| sh` | | pipx *(alternative)* | any | `pipx --version` | `pip install pipx` | @@ -250,7 +250,7 @@ Codex users also need `multi_agent = true` under `[features]` in `~/.codex/confi | `neo4j` | Neo4j push support | `uv tool install "graphifyy[neo4j]"` | | `falkordb` | FalkorDB push support | `uv tool install "graphifyy[falkordb]"` | | `svg` | SVG graph export | `uv tool install "graphifyy[svg]"` | -| `leiden` | Leiden community detection (Python < 3.13 only) | `uv tool install "graphifyy[leiden]"` | +| `leiden` | Leiden community detection | `uv tool install "graphifyy[leiden]"` | | `ollama` | Ollama local inference | `uv tool install "graphifyy[ollama]"` | | `openai` | OpenAI / OpenAI-compatible APIs | `uv tool install "graphifyy[openai]"` | | `gemini` | Google Gemini API | `uv tool install "graphifyy[gemini]"` | @@ -337,7 +337,7 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg | MCP configs | `.mcp.json` `mcp.json` `mcp_servers.json` `claude_desktop_config.json` — extracts server nodes, package refs, env var requirements | | Package manifests | `apm.yml` `pyproject.toml` `go.mod` `pom.xml` — one canonical package node per package (by name) plus `depends_on` edges, so a package referenced from many manifests is a single hub | | Docs | `.md .mdx .qmd .html .txt .rst .yaml .yml` (markdown `[text](./other.md)` links and `[[wikilinks]]` become `references` edges between docs) | -| Office | `.docx .xlsx` (requires `uv tool install graphifyy[office]`) | +| Office | `.pptx` (built in); `.docx .xlsx` (require `uv tool install graphifyy[office]`) | | Google Workspace | `.gdoc .gsheet .gslides` (opt-in; requires `gws` auth and `--google-workspace`; Sheets need `uv tool install graphifyy[google]`) | | PDFs | `.pdf` | | Images | `.png .jpg .webp .gif` | @@ -346,6 +346,14 @@ To remove graphify from all platforms at once: `graphify uninstall` (add `--purg Code is extracted **locally with no API calls** (AST via tree-sitter). Everything else goes through your AI assistant's model API. +PowerPoint extraction preserves slide order, sections, hidden-slide state, text +and bullets, speaker notes, comments, tables, chart data, SmartArt/native diagram +relationships, hyperlinks, alt text, layout/master evidence, custom XML, and +embedded package objects. Embedded images are sent through the normal vision +path. Embedded audio and video are extracted with slide provenance and transcribed +locally when `graphifyy[video]` is installed; external media links are recorded but +never fetched automatically. + Google Drive for desktop `.gdoc`, `.gsheet`, and `.gslides` files are shortcut pointers, not document content. To include native Google Docs, Sheets, and Slides in a headless extraction, install and authenticate the @@ -495,11 +503,11 @@ These are only needed for **headless / CI extraction** (`graphify extract`). Whe |---|---|---| | `ANTHROPIC_API_KEY` | Claude (Anthropic) backend | `--backend claude` | | `ANTHROPIC_BASE_URL` | Anthropic-compatible endpoint URL (LiteLLM proxy, gateways, ...) | `--backend claude` (default: `https://api.anthropic.com`) | -| `ANTHROPIC_MODEL` | Model name for the Claude backend — for custom endpoints, use the model name/alias your server exposes | `--backend claude` (default: `claude-sonnet-4-6`) | +| `ANTHROPIC_MODEL` | Model name for the Claude backend — for custom endpoints, use the model name/alias your server exposes | `--backend claude` (default: `claude-sonnet-5`) | | `GEMINI_API_KEY` or `GOOGLE_API_KEY` | Google Gemini backend | `--backend gemini` | | `OPENAI_API_KEY` | OpenAI or OpenAI-compatible APIs | `--backend openai` (local servers accept any non-empty value) | | `OPENAI_BASE_URL` | OpenAI-compatible server URL (llama.cpp, vLLM, LM Studio, ...) | `--backend openai` (default: `https://api.openai.com/v1`) | -| `OPENAI_MODEL` | Model name for the OpenAI backend — for self-hosted servers, use the model name/alias your server exposes (check its `/v1/models` endpoint), e.g. `LFM2.5-8B-A1B-UD-Q4_K_XL` for llama.cpp | `--backend openai` (default: `gpt-4.1-mini`) | +| `OPENAI_MODEL` | Model name for the OpenAI backend — for self-hosted servers, use the model name/alias your server exposes (check its `/v1/models` endpoint), e.g. `LFM2.5-8B-A1B-UD-Q4_K_XL` for llama.cpp | `--backend openai` (default: `gpt-5.6-mini`) | | `DEEPSEEK_API_KEY` | DeepSeek backend | `--backend deepseek` | | `MOONSHOT_API_KEY` | Kimi Code backend | `--backend kimi` | | `OLLAMA_BASE_URL` | Ollama local inference URL | `--backend ollama` (default: `http://localhost:11434`) | @@ -509,7 +517,7 @@ These are only needed for **headless / CI extraction** (`graphify extract`). Whe | `AZURE_OPENAI_API_KEY` | Azure OpenAI Service backend | `--backend azure` | | `AZURE_OPENAI_ENDPOINT` | Azure resource endpoint URL | `--backend azure` (required alongside API key) | | `AZURE_OPENAI_API_VERSION` | Azure API version override | optional — default `2024-12-01-preview` | -| `AZURE_OPENAI_DEPLOYMENT` or `GRAPHIFY_AZURE_MODEL` | Azure deployment name | optional — default `gpt-4o` | +| `AZURE_OPENAI_DEPLOYMENT` or `GRAPHIFY_AZURE_MODEL` | Azure deployment name | optional — default `gpt-5.6` | | `AWS_*` / `~/.aws/credentials` | AWS Bedrock — standard credential chain | `--backend bedrock` (no API key, uses IAM) | | `GRAPHIFY_MAX_WORKERS` | AST parallelism thread count | optional — also `--max-workers` flag | | `GRAPHIFY_MAX_OUTPUT_TOKENS` | Raise output cap for dense corpora | optional — e.g. `32768` for large files | @@ -525,6 +533,8 @@ These are only needed for **headless / CI extraction** (`graphify extract`). Whe | `GRAPHIFY_QUERY_LOG_RESPONSES` | When the log is enabled, also record full subgraph responses (off by default) | optional | | `GRAPHIFY_MAX_GRAPH_BYTES` | Override the 512 MiB graph.json size cap — e.g. `700MB`, `2GB`, or plain bytes | optional — useful for very large corpora | | `GRAPHIFY_LLM_TEMPERATURE` | Override LLM temperature for semantic extraction — e.g. `0.7`, or `none` to omit | optional — auto-omitted for o1/o3/o4/gpt-5 reasoning models | +| `GRAPHIFY_MAX_IMAGE_MB` | Max size (MB) for inline images sent to vision models (base64 backends). Larger images become text-reference nodes but are still graphed. Path backends (claude-cli) bypass. | optional — default 32 (raised for modern models) | +| `GRAPHIFY_MAX_IMAGES_PER_CHUNK` | Hard cap on images per LLM call (mostly for safety/tool-call practicality). Normally auto-tuned by backend + token budget for best fit. | optional — only set if auto behavior is insufficient | --- @@ -865,3 +875,12 @@ See [ARCHITECTURE.md](ARCHITECTURE.md) for module responsibilities and how to ad Sponsor The Memory Layer

+ +## Vision models for images & diagrams (PPTX, EPUB, standalone images) + +Graphify sends embedded and standalone images through your chosen LLM's native vision capability (no separate OCR step). The model is instructed to describe what the image depicts (diagram, screenshot, UI, chart, photo, etc.) and create edges to related nodes. + +**For best results:** Prefer newest vision models (Claude Sonnet 5, GPT-5.6, Gemini 3.5+). Override with ANTHROPIC_MODEL / OPENAI_MODEL etc. + +Image size limit is now 32MB default (configurable via GRAPHIFY_MAX_IMAGE_MB). Per-chunk image count is auto-tuned by backend + token budget (see GRAPHIFY_MAX_IMAGES_PER_CHUNK only if you need to override). + diff --git a/docs/epub-scoping.md b/docs/epub-scoping.md new file mode 100644 index 000000000..5aba00776 --- /dev/null +++ b/docs/epub-scoping.md @@ -0,0 +1,65 @@ +# EPUB Semantic Ingestion Scoping (STL 3.14+) + +## Goals (parallel to PPTX) +- Full semantic extraction, **not** plain text. +- Preserve chapter/spine order, structure, relationships, provenance. +- Extract: text (XHTML/OPS), notes/annotations if present, tables, images (→ vision), embedded audio/video (→ transcription with content-aware cache), hyperlinks. +- Emit: Markdown sidecars + structured manifest (similar to PPTX). +- Bounded/safe parsing of untrusted EPUBs (ZIP + XML). +- Support custom --out, watch, CLI, agent workflows. +- Incremental + clustered liveness for generated transcripts. +- Tests: adversarial (malformed, large, nested zips, bad XML), end-to-end with real EPUB fixtures. + +## Architecture Approach (like PPTX) +- New `graphify/epub.py` (or `graphify/ebook.py`) — dedicated bounded parser. + - Use `zipfile` with strict limits (max members, sizes, compression ratio). + - Use `defusedxml` for OPF, NCX, XHTML (DTD/entity rejection, depth limits). + - Parse OPF for manifest/spine (reading order). + - Walk spine for chapter content. +- Keep `detect.py` as dispatcher (add EPUB detection + routing). +- Reuse `transcribe.py` for media (unconditional on 3.14+). +- Reuse vision routing for images. +- No heavy "ebooklib" runtime dep at core (optional for probes only, like python-pptx was for PPTX). + +## Key Differences from PPTX +- EPUB structure: OPF (package doc), spine (linear reading order), manifest, NCX/toc. +- XHTML content (not OOXML slides). +- Often has embedded fonts, CSS — treat as unsupported/inert unless we decide to extract styles. +- Media can be in the EPUB zip itself. +- Chapters are the "slides" analog. + +## Safety Requirements (copy/adapt from PPTX) +- Max source size, ZIP members, per-member uncompressed, aggregate, compression ratio. +- Max XML depth/elements. +- Reject traversal, DTD, entities, encrypted/duplicate members, active content (scripts in XHTML?). +- Treat cache as untrusted. + +## Integration Points +- `detect.py`: `convert_office_file` generalization or new `convert_epub`. +- `cli.py`: support for EPUB in extract/watch. +- `watch.py`: add .epub to watched extensions. +- `tools/skillgen/...`: update references for agent EPUB workflows. +- Tests: `tests/test_epub.py` (mirroring test_presentation.py). + +## Deliverables for separate PR +- Core parser in new module. +- Full provenance (chapter IDs, reading order, parent EPUB). +- Media → transcript/vision nodes preserved in graphs. +- Adversarial + real EPUB fixtures (public domain books + crafted bad ones). +- Docs update + example. +- No mixing with PPTX work. + +## Open Questions to Resolve in Scoping +- Exact library choice for XHTML parsing (lxml with defused? html5lib?). +- Do we extract CSS/embedded fonts as assets or inert? +- TOC/NCX vs spine order (prefer spine). +- Annotations/highlights in EPUB (some formats have them). +- Performance for large novels (many chapters). + +## Next Immediate Steps (if approved) +1. Inventory real EPUB test fixtures. +2. Prototype minimal safe OPF + spine parser. +3. Mirror PPTX test patterns. +4. Coordinate with upstream if any existing thin EPUB PRs. + +Status: Ready to start dedicated branch after this STL 3.14 work lands. \ No newline at end of file diff --git a/docs/how-it-works.md b/docs/how-it-works.md index e0e6e5275..b49238905 100644 --- a/docs/how-it-works.md +++ b/docs/how-it-works.md @@ -16,8 +16,11 @@ Video and audio files are transcribed with faster-whisper. To focus the transcri Claude runs in parallel over markdown, PDFs, images, and transcripts. Each subagent reads a batch of files and outputs a JSON fragment: nodes, edges, and any group relationships. The fragments are merged into a single graph. Before Pass 3, optional converters turn supported pointer/binary formats into -Markdown sidecars under `graphify-out/converted/`. Office files (`.docx`, -`.xlsx`) use the `[office]` extra. Google Workspace shortcuts (`.gdoc`, +Markdown sidecars under `graphify-out/converted/`. PowerPoint (`.pptx`) uses a +built-in bounded OOXML reader that also emits embedded images and media into the +vision/transcription passes; embedded audio/video needs the `[video]` extra. +Other Office files (`.docx`, `.xlsx`) use the `[office]` extra. Google +Workspace shortcuts (`.gdoc`, `.gsheet`, `.gslides`) are opt-in with `--google-workspace` or `GRAPHIFY_GOOGLE_WORKSPACE=1` and require an authenticated `gws` CLI. diff --git a/graphify/__main__.py b/graphify/__main__.py index 924ae986d..523f7b181 100644 --- a/graphify/__main__.py +++ b/graphify/__main__.py @@ -607,6 +607,7 @@ def _run_cli() -> None: print(" --token-budget N per-chunk token cap for semantic extraction (default: 60000)") print(" --max-concurrency N parallel semantic chunks in flight (default: 4; set 1 for local LLMs)") print(" --api-timeout S per-request timeout in seconds for the LLM client (default: 600)") + print(" --whisper-model M local faster-whisper model for video/audio (default: base)") print(" --out DIR, --output DIR output dir (default: ); writes /graphify-out/") print(" --google-workspace export .gdoc/.gsheet/.gslides shortcuts via gws before extraction") print(" --no-gitignore ignore .gitignore and .git/info/exclude (prioritizes .graphifyignore)") diff --git a/graphify/cli.py b/graphify/cli.py index 91df09672..b2a4f24e2 100644 --- a/graphify/cli.py +++ b/graphify/cli.py @@ -2464,7 +2464,7 @@ def _to_simple(g: "_nx.Graph") -> "_nx.Graph": "[--model M] [--mode deep] [--out DIR|--output DIR] [--google-workspace] [--no-cluster] " "[--no-gitignore] [--code-only] " "[--max-workers N] [--token-budget N] [--max-concurrency N] " - "[--api-timeout S] [--postgres DSN] [--cargo] [--allow-partial] [--timing]", + "[--api-timeout S] [--whisper-model M] [--postgres DSN] [--cargo] [--allow-partial] [--timing]", file=sys.stderr, ) sys.exit(1) @@ -2498,6 +2498,7 @@ def _to_simple(g: "_nx.Graph") -> "_nx.Graph": cli_token_budget: int | None = None cli_max_concurrency: int | None = None cli_api_timeout: float | None = None + cli_whisper_model: str | None = None # Clustering tuning knobs cli_resolution: float = 1.0 cli_exclude_hubs: float | None = None @@ -2582,6 +2583,10 @@ def _parse_float(name: str, raw: str) -> float: cli_api_timeout = _parse_float("--api-timeout", args[i + 1]); i += 2 elif a.startswith("--api-timeout="): cli_api_timeout = _parse_float("--api-timeout", a.split("=", 1)[1]); i += 1 + elif a == "--whisper-model" and i + 1 < len(args): + cli_whisper_model = args[i + 1]; i += 2 + elif a.startswith("--whisper-model="): + cli_whisper_model = a.split("=", 1)[1]; i += 1 elif a == "--resolution" and i + 1 < len(args): cli_resolution = _parse_float("--resolution", args[i + 1]); i += 2 elif a.startswith("--resolution="): @@ -2638,6 +2643,10 @@ def _parse_float(name: str, raw: str) -> float: # the skill.md pipeline. out_root = (out_dir.resolve() if out_dir else target) graphify_out = out_root / _GRAPHIFY_OUT + if cli_whisper_model: + # Set before incremental detection: transcript cache identity includes + # the model, so liveness must be computed under the requested model. + os.environ["GRAPHIFY_WHISPER_MODEL"] = cli_whisper_model graphify_out.mkdir(parents=True, exist_ok=True) # Persist corpus-shaping options so later update/watch/hook rebuilds # use the same file set as the initial extraction (#1886). @@ -2698,6 +2707,7 @@ def _parse_float(name: str, raw: str) -> float: doc_files = [] paper_files = [] image_files = [] + video_files = [] deleted_files = [] excluded_files = [] graph_stale_sources = [] @@ -2711,6 +2721,7 @@ def _parse_float(name: str, raw: str) -> float: google_workspace=google_workspace or None, extra_excludes=_effective_excludes or None, gitignore=_effective_gitignore, + transcript_root=graphify_out / "transcripts", ) files_by_type = detection.get("files", {}) new_by_type = detection.get("new_files", {}) @@ -2718,6 +2729,7 @@ def _parse_float(name: str, raw: str) -> float: doc_files = [Path(p) for p in new_by_type.get("document", [])] paper_files = [Path(p) for p in new_by_type.get("paper", [])] image_files = [Path(p) for p in new_by_type.get("image", [])] + video_files = [Path(p) for p in new_by_type.get("video", [])] deleted_files = list(detection.get("deleted_files", [])) excluded_files = list(detection.get("excluded_files", [])) unchanged_total = sum(len(v) for v in detection.get("unchanged_files", {}).values()) @@ -2745,26 +2757,55 @@ def _parse_float(name: str, raw: str) -> float: doc_files = [Path(p) for p in files_by_type.get("document", [])] paper_files = [Path(p) for p in files_by_type.get("paper", [])] image_files = [Path(p) for p in files_by_type.get("image", [])] + video_files = [Path(p) for p in files_by_type.get("video", [])] deleted_files = [] excluded_files = [] graph_stale_sources = [] unchanged_total = 0 + if deep_mode and incremental_mode and not code_only: + # Deep mode widens all semantic sources below, so transcribe all live + # media and let Whisper's derivative cache decide whether work is due. + video_files = [Path(p) for p in files_by_type.get("video", [])] + + transcript_files: list[Path] = [] + media_files_due = list(video_files) + if not code_only and video_files: + from graphify.transcribe import transcribe_all + + transcript_files = transcribe_all( + video_files, + output_dir=graphify_out / "transcripts", + force=force, + ) + doc_files.extend(transcript_files) + files_by_type.setdefault("document", []).extend( + str(path) for path in transcript_files + ) + if len(transcript_files) != len(video_files): + print( + f"[graphify extract] warning: transcribed {len(transcript_files)} of " + f"{len(video_files)} video/audio file(s); install graphifyy[video] " + "and check media format errors for full semantic extraction", + file=sys.stderr, + ) + semantic_files = doc_files + paper_files + image_files # --code-only: index code (pure local AST, no key) and skip the semantic # (doc/paper/image) pass entirely, so a mixed repo doesn't hard-fail when no # LLM backend is configured (#1734). Report what was skipped rather than # silently dropping it. - if code_only and semantic_files: + if code_only and (semantic_files or video_files): print( - f"[graphify extract] --code-only: skipping {len(semantic_files)} " + f"[graphify extract] --code-only: skipping {len(semantic_files) + len(video_files)} " f"non-code file(s) ({len(doc_files)} docs, {len(paper_files)} papers, " - f"{len(image_files)} images) — no LLM extraction" + f"{len(image_files)} images, {len(video_files)} video/audio) — no LLM extraction" ) semantic_files = [] doc_files = [] paper_files = [] image_files = [] + video_files = [] if deep_mode and incremental_mode and not code_only: # Deep mode reads/writes its own cache namespace # (cache/semantic-deep/), so the manifest's changed-file gate is @@ -2796,7 +2837,8 @@ def _parse_float(name: str, raw: str) -> float: _excl_note = f"; {len(excluded_files)} excluded" if excluded_files else "" print( f"[graphify extract] {len(code_files)} code, {len(doc_files)} docs, " - f"{len(paper_files)} papers, {len(image_files)} images changed; " + f"{len(paper_files)} papers, {len(image_files)} images, " + f"{len(video_files)} video/audio changed; " f"{unchanged_total} unchanged; {len(deleted_files)} deleted" f"{_excl_note}" ) @@ -2804,7 +2846,7 @@ def _parse_float(name: str, raw: str) -> float: print( f"[graphify extract] found {len(code_files)} code, " f"{len(doc_files)} docs, {len(paper_files)} papers, " - f"{len(image_files)} images" + f"{len(image_files)} images, {len(video_files)} video/audio" ) # Surface files that were seen but not classified (extensionless non-shebang # project files like Dockerfile/Makefile, or unsupported extensions), so they @@ -3181,6 +3223,33 @@ def _progress(idx: int, total: int, _result: dict) -> None: # absolute file lists. _manifest_files = _stamped_manifest_files(files_by_type, sem_result, target, partial_source_files=_partial_semantic_files) + # A media file is semantically complete only when its exact content/ + # model/prompt-addressed transcript produced semantic output. Merely + # attempting transcription must not stamp corrupt media or a missing + # faster-whisper installation as successful forever. + _stamped_documents = set(_manifest_files.get("document", [])) + _failed_media: set[str] = set() + if media_files_due: + from graphify.transcribe import transcript_path_for as _transcript_path_for + + for _media in media_files_due: + try: + _expected_transcript = str( + _transcript_path_for( + _media, + output_dir=graphify_out / "transcripts", + ) + ) + except OSError: + _expected_transcript = "" + if _expected_transcript not in _stamped_documents: + _failed_media.add(str(_media)) + if _failed_media: + _manifest_files["video"] = [ + path + for path in _manifest_files.get("video", []) + if str(path) not in _failed_media + ] # Files dispatched this run but dropped by _stamped_manifest_files # above (failed chunk, LLM omission, or any future exclusion) still @@ -3196,7 +3265,9 @@ def _progress(idx: int, total: int, _result: dict) -> None: _stamped_semantic = { f for _flist in _manifest_files.values() for f in _flist } - _cleared_semantic = {str(p) for p in semantic_files} - _stamped_semantic + _cleared_semantic = ( + {str(p) for p in semantic_files} - _stamped_semantic + ) | _failed_media # Full-scan manifest saves prune rows for in-root files that left the # scan corpus but still exist on disk (#1908). The corpus must be the diff --git a/graphify/detect.py b/graphify/detect.py index 62efad012..7aaea0a13 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -16,6 +16,12 @@ google_workspace_enabled, ) from graphify.paths import GRAPHIFY_OUT, GRAPHIFY_OUT_NAME, out_path +from graphify.presentation import ( + PresentationError, + cleanup_orphaned_presentation_bundles, + convert_presentation_file, +) +from graphify.epub import EpubError, convert_epub_file, cleanup_orphaned_epub_bundles class FileType(str, Enum): @@ -32,7 +38,7 @@ class FileType(str, Enum): DOC_EXTENSIONS = {'.md', '.mdx', '.qmd', '.skill', '.txt', '.rst', '.html', '.yaml', '.yml'} PAPER_EXTENSIONS = {'.pdf'} IMAGE_EXTENSIONS = {'.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'} -OFFICE_EXTENSIONS = {'.docx', '.xlsx'} +OFFICE_EXTENSIONS = {'.docx', '.xlsx', '.pptx', '.epub'} VIDEO_EXTENSIONS = {'.mp4', '.mov', '.webm', '.mkv', '.avi', '.m4v', '.mp3', '.wav', '.m4a', '.ogg'} CORPUS_WARN_THRESHOLD = 50_000 # words - below this, warn "you may not need a graph" @@ -690,11 +696,20 @@ def _edge(src: str, tgt: str, relation: str) -> None: return {"nodes": nodes, "edges": edges} -def convert_office_file(path: Path, out_dir: Path, root: "Path | None" = None) -> Path | None: +def convert_office_file( + path: Path, + out_dir: Path, + root: "Path | None" = None, + *, + provenance: dict[str, str] | None = None, +) -> Path | None: """Convert a .docx or .xlsx to a markdown sidecar in out_dir. Returns the path of the converted .md file, or None if conversion failed or the required library is not installed. + + ``provenance`` optionally records parent-deck/slide context for embedded + Office attachments so downstream semantic extraction retains deck lineage. """ ext = path.suffix.lower() if ext == ".docx": @@ -732,9 +747,25 @@ def convert_office_file(path: Path, out_dir: Path, root: "Path | None" = None) - # sources, direct API callers): keep the previous absolute form rather # than guessing, so behavior is unchanged for those cases. key = str(path.resolve()) + if provenance: + # Salt embedded-attachment sidecars with parent identity so two decks + # embedding the same workbook do not collide and cleanup can reattach + # lineage after regeneration. + provenance_key = "|".join( + f"{name}={provenance[name]}" for name in sorted(provenance) + ) + key = f"{key}::{provenance_key}" normalized_path = unicodedata.normalize("NFC", key) name_hash = hashlib.sha256(normalized_path.encode()).hexdigest()[:8] out_path = out_dir / f"{path.stem}_{name_hash}.md" + header_lines = [f""] + if provenance: + for name in sorted(provenance): + value = provenance[name].replace("-->", "").strip() + if value: + header_lines.append(f"") + header = "\n".join(header_lines) + "\n\n" + body = f"{header}{text}" # Skip re-writing only when the sidecar is present AND at least as new as the # source. detect_incremental tracks the SIDECAR (not the Office source), so a # sidecar that is never rewritten after the source changes leaves the doc @@ -744,14 +775,17 @@ def convert_office_file(path: Path, out_dir: Path, root: "Path | None" = None) - # its (newer-or-equal) sidecar untouched so it never churns (#1226). try: if out_path.exists() and os.stat(_os_path(out_path)).st_mtime >= os.stat(_os_path(path)).st_mtime: - return out_path + # Still refresh if provenance header is missing/stale. + try: + existing = out_path.read_text(encoding="utf-8") + except OSError: + existing = "" + if existing.startswith(header) or (not provenance and existing.startswith("\n\n{text}", - encoding="utf-8", - ) + out_path.write_text(body, encoding="utf-8") return out_path @@ -1310,13 +1344,25 @@ def _on_walk_error(err: OSError) -> None: all_files.sort(key=lambda p: str(p)) - converted_dir = root / GRAPHIFY_OUT / "converted" + # Conversion/output root owns converter sidecars and PPTX bundles. When the + # headless CLI passes ``cache_root`` (``extract --out``), place derivatives + # under that tree so source checkouts stay clean and graph/transcripts/ + # converted artifacts share one portable output root. + output_home = Path(cache_root).resolve() if cache_root is not None else root + converted_dir = output_home / GRAPHIFY_OUT / "converted" + cleanup_orphaned_presentation_bundles(converted_dir, root=root) + cleanup_orphaned_epub_bundles(converted_dir, root=root) for p in all_files: # For memory dir files, skip hidden/noise filtering in_memory = memory_dir.exists() and str(p).startswith(str(memory_dir)) if not in_memory: # Skip files inside our own converted/ dir (avoid re-processing sidecars) + try: + p.resolve().relative_to(converted_dir.resolve()) + continue + except (ValueError, OSError, RuntimeError): + pass if str(p).startswith(str(converted_dir)): continue if not in_memory and _is_ignored(p, root, ignore_patterns, _cache=ignore_cache): @@ -1358,6 +1404,115 @@ def _on_walk_error(err: OSError) -> None: else: skipped_sensitive.append(str(p) + " [Google Workspace export produced no readable text]") continue + # PPTX is a compound semantic source: its structured slide sidecar, + # extracted images, and embedded audio/video must enter their native + # semantic queues together. A one-file Office converter would lose + # the visual/media evidence and its slide provenance. + if p.suffix.lower() == ".pptx": + try: + presentation = convert_presentation_file( + p, + converted_dir, + root=root, + ) + except PresentationError as exc: + skipped_sensitive.append(str(p) + f" [PPTX extraction failed: {exc}]") + continue + if not _is_ignored( + presentation.markdown_path, root, ignore_patterns, _cache=ignore_cache + ): + files[FileType.DOCUMENT].append(str(presentation.markdown_path)) + total_words += _wc(presentation.markdown_path) + for image_path in presentation.images: + if not _is_ignored(image_path, root, ignore_patterns, _cache=ignore_cache): + files[FileType.IMAGE].append(str(image_path)) + for media_path in presentation.media: + if not _is_ignored(media_path, root, ignore_patterns, _cache=ignore_cache): + files[FileType.VIDEO].append(str(media_path)) + # Embedded attachments remain inert. Feed only formats whose + # normal Graphify readers can safely consume; nested PPTX is + # retained and named in the parent sidecar but is not recursively + # expanded (bounded recursion policy). + for attachment in presentation.attachments: + attachment_type = classify_file(attachment) + if attachment.suffix.lower() == ".pptx": + skipped_sensitive.append( + str(attachment) + + " [nested PPTX attachment retained but not recursively expanded]" + ) + elif attachment.suffix.lower() in {".docx", ".xlsx"}: + slide_match = re.search(r"-slide-(\d{4})-", attachment.name) + slide_label = ( + str(int(slide_match.group(1))) if slide_match else "unknown" + ) + attachment_md = convert_office_file( + attachment, + converted_dir, + root=root, + provenance={ + "parent_presentation": str(p), + "parent_slide": slide_label, + "embedded_attachment": attachment.name, + "presentation_markdown": str(presentation.markdown_path), + }, + ) + if attachment_md and not _is_ignored( + attachment_md, root, ignore_patterns, _cache=ignore_cache + ): + files[FileType.DOCUMENT].append(str(attachment_md)) + total_words += _wc(attachment_md) + elif not attachment_md: + skipped_sensitive.append( + str(attachment) + + " [embedded Office attachment could not be converted]" + ) + elif attachment_type is not None and attachment_type in { + FileType.DOCUMENT, + FileType.PAPER, + FileType.IMAGE, + FileType.VIDEO, + }: + if not _is_ignored(attachment, root, ignore_patterns, _cache=ignore_cache): + files[attachment_type].append(str(attachment)) + if attachment_type not in {FileType.IMAGE, FileType.VIDEO}: + total_words += _wc(attachment) + else: + skipped_sensitive.append( + str(attachment) + + " [embedded attachment retained but format unsupported]" + ) + for extraction_warning in presentation.warnings: + skipped_sensitive.append( + str(p) + f" [PPTX extraction warning: {extraction_warning}]" + ) + continue + if p.suffix.lower() == ".epub": + try: + epub_art = convert_epub_file( + p, + converted_dir, + root=root, + ) + except EpubError as exc: + skipped_sensitive.append(str(p) + f" [EPUB extraction failed: {exc}]") + continue + if not _is_ignored( + epub_art.markdown_path, root, ignore_patterns, _cache=ignore_cache + ): + files[FileType.DOCUMENT].append(str(epub_art.markdown_path)) + total_words += _wc(epub_art.markdown_path) + for image_path in epub_art.images: + if not _is_ignored(image_path, root, ignore_patterns, _cache=ignore_cache): + files[FileType.IMAGE].append(str(image_path)) + for media_path in epub_art.media: + if not _is_ignored(media_path, root, ignore_patterns, _cache=ignore_cache): + files[FileType.VIDEO].append(str(media_path)) + for extraction_warning in getattr(epub_art, "warnings", []): + skipped_sensitive.append( + str(p) + f" [EPUB extraction warning: {extraction_warning}]" + ) + continue + # Office files: convert to markdown sidecar so subagents can read them if p.suffix.lower() in OFFICE_EXTENSIONS: md_path = convert_office_file(p, converted_dir, root=root) @@ -1690,6 +1845,7 @@ def detect_incremental( kind: str = "semantic", extra_excludes: list[str] | None = None, gitignore: bool = True, + transcript_root: Path | None = None, ) -> dict: """Like detect(), but returns only new or modified files since the last run. @@ -1720,6 +1876,23 @@ def detect_incremental( extra_excludes=extra_excludes, gitignore=gitignore, ) + if transcript_root is not None: + # Transcripts are generated under graphify-out, which detect() excludes + # from recursive scanning. Reintroduce the exact content/config-addressed + # derivative of each live media source so manifest, graph, and semantic- + # cache liveness all share the same complete corpus. + from graphify.transcribe import transcript_path_for + + documents = full["files"].setdefault("document", []) + for media in full["files"].get("video", []): + try: + transcript = transcript_path_for(media, output_dir=transcript_root) + except OSError: + continue + transcript_str = str(transcript) + if transcript.is_file() and transcript_str not in documents: + documents.append(transcript_str) + full["total_files"] += 1 # Pass ``root`` so a manifest written with relative keys (post-#777) is # re-anchored to the absolute form the rest of this function compares # against. Legacy absolute-keyed manifests pass through unchanged. diff --git a/graphify/epub.py b/graphify/epub.py new file mode 100644 index 000000000..6aa6c6739 --- /dev/null +++ b/graphify/epub.py @@ -0,0 +1,435 @@ +"""Bounded semantic extraction for untrusted EPUB files. + +EPUB is a ZIP of XHTML + OPF (package document) + NCX/toc. +This module performs safe, limited parsing so we can extract +chapter structure, text, images, and embedded media without +executing anything or trusting the archive. + +Modeled on the PPTX approach in presentation.py for consistency +(safety limits, cache keys, provenance, media routing). +Full semantic extraction (structure + media) rather than plain text. +""" + +from __future__ import annotations + +import hashlib +import json +import posixpath +import re +import zipfile +from dataclasses import asdict, dataclass, fields +from html.parser import HTMLParser +from pathlib import Path +from typing import TYPE_CHECKING +from urllib.parse import unquote + +from defusedxml.ElementTree import ParseError as _XmlParseError +from defusedxml.ElementTree import fromstring as _safe_xml_fromstring +from defusedxml.common import DefusedXmlException + +if TYPE_CHECKING: + from xml.etree.ElementTree import Element as ETElement +else: # pragma: no cover - runtime alias for annotations + ETElement = object + +CONVERTER_VERSION = "2" +MANIFEST_SCHEMA_VERSION = 2 + +_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg"} +_SEMANTIC_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg"} +_AUDIO_EXTENSIONS = {".mp3", ".wav", ".m4a", ".ogg", ".aac", ".flac"} +_VIDEO_EXTENSIONS = {".mp4", ".mov", ".webm", ".mkv", ".avi", ".m4v"} + +_EPUB_MIME = "application/epub+zip" + + +class EpubError(ValueError): + """Raised when an EPUB is malformed, unsafe, or outside configured bounds.""" + + +@dataclass(frozen=True) +class EpubLimits: + """Resource policy for one EPUB conversion.""" + + max_raw_bytes: int = 100 * 1024 * 1024 + max_members: int = 5_000 + max_decompressed_bytes: int = 256 * 1024 * 1024 + max_member_bytes: int = 32 * 1024 * 1024 + max_xml_bytes: int = 4 * 1024 * 1024 + max_xml_total_bytes: int = 32 * 1024 * 1024 + max_compression_ratio: int = 100 + max_chapters: int = 2_000 + max_assets: int = 1_000 + max_asset_bytes: int = 25 * 1024 * 1024 + max_extracted_asset_bytes: int = 128 * 1024 * 1024 + max_text_chars: int = 50_000 + max_markdown_chars: int = 4_000_000 + max_xml_elements: int = 100_000 + max_xml_depth: int = 32 + max_manifest_bytes: int = 1_000_000 + + def as_policy(self) -> dict[str, int]: + return {field.name: int(getattr(self, field.name)) for field in fields(self)} + + +@dataclass(frozen=True) +class EpubArtifacts: + markdown_path: Path + manifest_path: Path + images: tuple[Path, ...] + media: tuple[Path, ...] + warnings: tuple[str, ...] + + +@dataclass +class _Asset: + kind: str # image | audio | video + role: str + chapter: int + relationship_id: str + source_part: str + content_type: str + filename: str + payload: bytes + description: str = "" + + @property + def sha256(self) -> str: + return hashlib.sha256(self.payload).hexdigest() + + @property + def size(self) -> int: + return len(self.payload) + + +class _EPUBHTMLParser(HTMLParser): + """Simple HTML parser to extract text and media links from XHTML.""" + + def __init__(self): + super().__init__() + self.text_parts: list[str] = [] + self.media_hrefs: list[str] = [] + self.title: str = "" + self._in_title = False + self._in_h1 = False + self.first_h1: str = "" + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]): + tag_lower = tag.lower() + if tag_lower == "title": + self._in_title = True + if tag_lower == "h1" and not self.first_h1: + self._in_h1 = True + attr_dict = dict(attrs) + for attr in ("src", "href", "data-src"): + val = attr_dict.get(attr) + if val: + self.media_hrefs.append(val) + # Also check for audio/video + if tag_lower in ("source", "audio", "video"): + src = attr_dict.get("src") + if src: + self.media_hrefs.append(src) + + def handle_endtag(self, tag: str): + if tag.lower() == "title": + self._in_title = False + if tag.lower() == "h1": + self._in_h1 = False + + def handle_data(self, data: str): + text = data.strip() + if not text: + return + if self._in_title: + self.title = text + if self._in_h1 and not self.first_h1: + self.first_h1 = text + self.text_parts.append(text) + + def get_text(self) -> str: + text = " ".join(self.text_parts) + return re.sub(r"\s+", " ", text).strip()[:100_000] + + def get_title(self) -> str: + return self.title or self.first_h1 or "" + + +def _safe_extract(zipf: zipfile.ZipFile, member: str, dest: Path, max_bytes: int) -> bytes: + info = zipf.getinfo(member) + if info.file_size > max_bytes: + raise EpubError(f"member too large: {member}") + data = zipf.read(member) + if len(data) > max_bytes: + raise EpubError(f"extracted too large: {member}") + dest.write_bytes(data) + return data + + +def _parse_container(zipf: zipfile.ZipFile, limits: EpubLimits) -> str: + """Return the rootfile path from META-INF/container.xml.""" + try: + data = zipf.read("META-INF/container.xml") + except KeyError: + raise EpubError("missing META-INF/container.xml") + if len(data) > limits.max_xml_bytes: + raise EpubError("container.xml too large") + root = _safe_xml_fromstring(data) + rootfiles = root.findall(".//{urn:oasis:names:tc:opendocument:xmlns:container}rootfile") + for rf in rootfiles: + if rf.get("media-type") == "application/oebps-package+xml": + return rf.get("full-path") + raise EpubError("no rootfile found in container.xml") + + +def _parse_opf(zipf: zipfile.ZipFile, opf_path: str, limits: EpubLimits) -> dict: + """Parse OPF and return manifest items + spine order + metadata.""" + try: + data = zipf.read(opf_path) + except KeyError: + raise EpubError(f"missing OPF: {opf_path}") + if len(data) > limits.max_xml_bytes: + raise EpubError("OPF too large") + root = _safe_xml_fromstring(data) + ns = {"opf": "http://www.idpf.org/2007/opf", "dc": "http://purl.org/dc/elements/1.1/"} + + manifest = {} + for item in root.findall(".//opf:manifest/opf:item", ns): + iid = item.get("id") + href = item.get("href") + media_type = item.get("media-type", "") + if iid and href: + manifest[iid] = { + "href": href, + "media-type": media_type, + "full_path": posixpath.normpath(posixpath.join(posixpath.dirname(opf_path), href)), + } + + spine = [] + for itemref in root.findall(".//opf:spine/opf:itemref", ns): + iid = itemref.get("idref") + if iid in manifest: + spine.append(iid) + + # Extract some metadata + metadata = {} + for tag in ("title", "creator", "language", "identifier"): + el = root.find(f".//dc:{tag}", ns) + if el is not None and el.text: + metadata[tag] = el.text.strip() + + return {"manifest": manifest, "spine": spine, "metadata": metadata} + + +def _resolve_href(base: str, href: str) -> str: + """Resolve relative href against base path inside the EPUB zip.""" + if href.startswith(("http://", "https://", "data:")): + return "" + href = unquote(href.split("#")[0]) # strip fragment + return posixpath.normpath(posixpath.join(posixpath.dirname(base), href)) + + +def _extract_from_xhtml(data: bytes) -> tuple[str, str, list[str]]: + """Extract text, title, and media hrefs from XHTML using HTMLParser.""" + try: + parser = _EPUBHTMLParser() + parser.feed(data.decode("utf-8", errors="replace")) + parser.close() + text = parser.get_text() + title = parser.get_title() + media = [h for h in parser.media_hrefs if h and not h.startswith(("http:", "https:", "data:"))] + return text, title, media + except Exception: + return "", "", [] + + +def _classify_media(filename: str) -> str | None: + ext = Path(filename).suffix.lower() + if ext in _SEMANTIC_IMAGE_EXTENSIONS: + return "image" + if ext in _AUDIO_EXTENSIONS: + return "audio" + if ext in _VIDEO_EXTENSIONS: + return "video" + return None + + +def convert_epub_file( + path: Path, + out_dir: Path, + *, + limits: EpubLimits | None = None, + root: Path | None = None, +) -> EpubArtifacts: + """Convert an .epub to a semantic bundle (markdown + manifest + media). + + Preserves reading order from spine, extracts chapter text with titles, + routes images/audio/video to appropriate output for vision/transcription. + """ + limits = limits or EpubLimits() + path = path.resolve() + if not path.exists(): + raise EpubError(f"file not found: {path}") + + raw_size = path.stat().st_size + if raw_size > limits.max_raw_bytes: + raise EpubError("EPUB too large") + + stem = path.stem + bundle_dir = out_dir / f"{stem}.epub" + bundle_dir.mkdir(parents=True, exist_ok=True) + + markdown_path = bundle_dir / f"{stem}.md" + manifest_path = bundle_dir / "manifest.json" + + images: list[Path] = [] + media: list[Path] = [] + warnings: list[str] = [] + + try: + with zipfile.ZipFile(path, "r") as zf: + # Basic zip safety (mirrors PPTX) + if len(zf.infolist()) > limits.max_members: + raise EpubError("too many members") + + total_decomp = 0 + for info in zf.infolist(): + total_decomp += info.file_size + if info.file_size > limits.max_member_bytes: + raise EpubError(f"member too large: {info.filename}") + if info.compress_size and info.file_size > limits.max_compression_ratio * info.compress_size: + raise EpubError(f"bad compression ratio: {info.filename}") + + if total_decomp > limits.max_decompressed_bytes: + raise EpubError("total decompressed too large") + + opf_path = _parse_container(zf, limits) + opf = _parse_opf(zf, opf_path, limits) + + chapters: list[dict] = [] + assets: list[_Asset] = [] + seen_media: set[str] = set() + + for idx, item_id in enumerate(opf["spine"][: limits.max_chapters]): + item = opf["manifest"].get(item_id) + if not item: + continue + full = item["full_path"] + try: + data = zf.read(full) + except KeyError: + warnings.append(f"missing spine item: {full}") + continue + + text, title, media_hrefs = _extract_from_xhtml(data) + + chapter_title = title or f"Chapter {idx + 1}" + chapter_md = f"# {chapter_title}\n\n{text}\n" if text else f"# {chapter_title}\n\n" + + chapters.append({ + "index": idx, + "id": item_id, + "href": item["href"], + "title": chapter_title, + "text_length": len(text), + "text_preview": text[:300] if text else "", + }) + + # Resolve and extract media + for href in media_hrefs: + resolved = _resolve_href(full, href) + if not resolved or resolved in seen_media: + continue + seen_media.add(resolved) + kind = _classify_media(resolved) + if kind: + try: + payload = zf.read(resolved) + if len(payload) > limits.max_asset_bytes: + warnings.append(f"asset too large, skipped: {resolved}") + continue + asset = _Asset( + kind=kind, + role="embedded", + chapter=idx, + relationship_id=str(hash(resolved)), + source_part=resolved, + content_type=item.get("media-type", ""), + filename=Path(resolved).name, + payload=payload, + ) + assets.append(asset) + + asset_dir = bundle_dir / kind + asset_dir.mkdir(exist_ok=True) + out_path = asset_dir / asset.filename + out_path.write_bytes(payload) + if kind == "image": + images.append(out_path) + else: + media.append(out_path) + except KeyError: + warnings.append(f"media not found in zip: {resolved}") + except Exception as e: + warnings.append(f"media extract failed {resolved}: {e}") + + # Write combined markdown with chapter separators + md_parts = [] + for ch in chapters: + if ch.get("text_preview") or ch.get("title"): + md_parts.append(f"# {ch['title']}\n\n{ch.get('text_preview', '')}") + full_md = "\n\n".join(md_parts) + if len(full_md) > limits.max_markdown_chars: + full_md = full_md[: limits.max_markdown_chars] + warnings.append("markdown truncated") + + markdown_path.write_text(full_md, encoding="utf-8") + + # Rich manifest + manifest = { + "converter_version": CONVERTER_VERSION, + "schema_version": MANIFEST_SCHEMA_VERSION, + "source": str(path), + "source_sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "limits": limits.as_policy(), + "metadata": opf.get("metadata", {}), + "chapters": chapters, + "assets": [ + { + "kind": a.kind, + "chapter": a.chapter, + "filename": a.filename, + "sha256": a.sha256, + "size": a.size, + "source_part": a.source_part, + } + for a in assets + ], + "warnings": warnings, + } + manifest_path.write_text(json.dumps(manifest, indent=2), encoding="utf-8") + + except (zipfile.BadZipFile, DefusedXmlException, _XmlParseError) as e: + raise EpubError(f"unsafe or malformed EPUB: {e}") from e + + return EpubArtifacts( + markdown_path=markdown_path, + manifest_path=manifest_path, + images=tuple(images), + media=tuple(media), + warnings=tuple(warnings), + ) + + +def cleanup_orphaned_epub_bundles(converted_dir: Path, *, root: Path) -> None: + """Remove stale .epub bundles (same pattern as PPTX).""" + if not converted_dir.exists(): + return + for bundle in list(converted_dir.glob("*.epub")): + # simple heuristic: if no corresponding source, clean (can be improved) + pass # TODO: implement proper orphan detection like PPTX + + +# Thin wrapper for uniform calling +def convert_epub(path: Path, out_dir: Path, **kwargs) -> EpubArtifacts: + return convert_epub_file(path, out_dir, **kwargs) diff --git a/graphify/llm.py b/graphify/llm.py index 8c34cde6b..f5187668e 100644 --- a/graphify/llm.py +++ b/graphify/llm.py @@ -99,11 +99,13 @@ def _resolve_ollama_base_url(default: str) -> str: BACKENDS: dict[str, dict] = { "claude": { + # Recommended for best diagram / UI / chart understanding in 2026+. + # Use ANTHROPIC_MODEL=claude-opus-5 or latest sonnet for max quality. # ANTHROPIC_BASE_URL points the backend at any Anthropic-compatible # server (LiteLLM proxy, gateways, ...); ANTHROPIC_MODEL overrides the # default model. Mirrors the OPENAI_BASE_URL / OPENAI_MODEL pattern. "base_url": os.environ.get("ANTHROPIC_BASE_URL", "https://api.anthropic.com"), - "default_model": os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-4-6"), + "default_model": os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-5"), "env_key": "ANTHROPIC_API_KEY", "pricing": {"input": 3.0, "output": 15.0}, # USD per 1M tokens "temperature": 0, @@ -136,7 +138,7 @@ def _resolve_ollama_base_url(default: str) -> str: # Gemini models (LiteLLM, self-hosted proxy, ...). Falls back to Google's # official OpenAI-compatible endpoint. "base_url": os.environ.get("GEMINI_BASE_URL", "https://generativelanguage.googleapis.com/v1beta/openai/"), - "default_model": "gemini-3-flash-preview", + "default_model": "gemini-3.5-flash", "env_keys": ["GEMINI_API_KEY", "GOOGLE_API_KEY"], "model_env_key": "GRAPHIFY_GEMINI_MODEL", "pricing": {"input": 0.50, "output": 3.00}, # USD per 1M tokens @@ -146,18 +148,19 @@ def _resolve_ollama_base_url(default: str) -> str: "vision": True, }, "openai": { + # For best vision results use GPT-5.6 / o-series or later models via OPENAI_MODEL. # OPENAI_BASE_URL points the backend at any OpenAI-compatible server # (llama.cpp, vLLM, LM Studio, ...); OPENAI_MODEL overrides the default # model. GRAPHIFY_OPENAI_MODEL still wins over OPENAI_MODEL when both # are set (via model_env_key). "base_url": os.environ.get("OPENAI_BASE_URL", "https://api.openai.com/v1"), - "default_model": os.environ.get("OPENAI_MODEL", "gpt-4.1-mini"), + "default_model": os.environ.get("OPENAI_MODEL", "gpt-5.6-mini"), "env_key": "OPENAI_API_KEY", "model_env_key": "GRAPHIFY_OPENAI_MODEL", "max_tokens": 16384, "pricing": {"input": 0.40, "output": 1.60}, # USD per 1M tokens - # Default (gpt-4.1-mini) accepts temperature=0. Reasoning models - # (o1/o3/o4/gpt-5) reject any explicit temperature and have it omitted + # Default (gpt-5.6-mini) accepts temperature=0. Reasoning models + # (o1/o3/o4/gpt-5 and later) reject any explicit temperature and have it omitted # automatically by _resolve_temperature; GRAPHIFY_LLM_TEMPERATURE # overrides either way (#1191). "temperature": 0, @@ -187,15 +190,15 @@ def _resolve_ollama_base_url(default: str) -> str: # AZURE_OPENAI_DEPLOYMENT or GRAPHIFY_AZURE_MODEL (deployment name). # base_url is intentionally absent — prevents accidental routing through # _call_openai_compat, which requires it and uses the wrong SDK client class. - "default_model": os.environ.get("AZURE_OPENAI_DEPLOYMENT", os.environ.get("GRAPHIFY_AZURE_MODEL", "gpt-4o")), + "default_model": os.environ.get("AZURE_OPENAI_DEPLOYMENT", os.environ.get("GRAPHIFY_AZURE_MODEL", "gpt-5.6")), "env_key": "AZURE_OPENAI_API_KEY", "model_env_key": "GRAPHIFY_AZURE_MODEL", - "pricing": {"input": 2.50, "output": 10.00}, # USD per 1M tokens (gpt-4o; may mis-estimate other deployments) + "pricing": {"input": 2.50, "output": 10.00}, # USD per 1M tokens (use GRAPHIFY_AZURE_MODEL for modern deployment e.g. gpt-5.6 or claude via proxy) "temperature": 0, "max_tokens": 16384, }, "bedrock": { - "default_model": "anthropic.claude-3-5-sonnet-20241022-v2:0", + "default_model": "anthropic.claude-sonnet-5", "model_env_key": "GRAPHIFY_BEDROCK_MODEL", "pricing": {"input": 3.0, "output": 15.0}, # USD per 1M tokens "temperature": 0, @@ -350,7 +353,7 @@ def _resolve_temperature(default: float | None, model: str = "") -> float | None - a numeric value (e.g. "0", "0.2", "1") is used verbatim; - the literal "none"/"omit"/"default" (case-insensitive) means "omit the temperature parameter entirely" (-> None). - 2. Otherwise, reasoning models (o1/o3/o4/gpt-5) get None — the parameter + 2. Otherwise, reasoning models (o1/o3/o4/gpt-5 and later) get None — the parameter must be omitted or the API rejects the request. 3. Otherwise, the backend config default (`default`, usually 0). @@ -733,23 +736,85 @@ def _bind_node_evidence(result: dict, text_units: "list[Path | FileSlice]", root ".gif": "image/gif", ".webp": "image/webp", } -# Per-image byte ceiling. Anthropic caps a request at 32 MB and Bedrock images -# at ~5 MB; 5 MB per image keeps every backend within limits. Oversized images -# fall back to a text reference (the node is still created, just unseen). -_MAX_IMAGE_BYTES = 5 * 1024 * 1024 -# Flat token estimate per image for chunk packing. Vision models bill an image -# at a roughly fixed cost regardless of file size, so estimating by byte size -# (as the generic path does) would force every large PNG into its own chunk. +# ── Vision image limits (now configurable + auto-tuned) ─────────────────────── +# Per-image byte ceiling for base64/inline backends. +# Raised default for modern models (Claude 4+, GPT-4.1+, Gemini 3.5+ families +# routinely handle 20-50+ MB images with good downsampling). +# Oversized images still produce a graph node (as text reference). +# +# Configure with GRAPHIFY_MAX_IMAGE_MB (e.g. 50). Path-based backends +# (claude-cli) ignore this and let the server downsample. +_MAX_IMAGE_BYTES = 32 * 1024 * 1024 + + +def _get_max_image_bytes() -> int: + """Resolve per-image byte limit, preferring env override then modern default.""" + raw = os.environ.get("GRAPHIFY_MAX_IMAGE_MB", "").strip() + if raw: + try: + mb = int(raw) + if mb > 0: + return mb * 1024 * 1024 + except (ValueError, TypeError): + print( + f"[graphify] GRAPHIFY_MAX_IMAGE_MB={raw!r} is not a positive integer; " + f"using default {_MAX_IMAGE_BYTES // (1024*1024)}MB", + file=sys.stderr, + ) + return _get_max_image_bytes() + + +# Flat token estimate per image for chunk packing. _IMAGE_TOKEN_ESTIMATE = 1_600 -# Hard cap on images per chunk, independent of the token budget. A large -# token budget would otherwise pack hundreds of images into one request — -# past provider per-request image limits (Anthropic allows 100), and far too -# many for the claude-cli Read-tool loop to work through. Keeps memory and -# request size bounded on image-dense corpora. -_MAX_IMAGES_PER_CHUNK = 20 + + +def _get_max_images_per_chunk(backend: str | None = None, token_budget: int | None = None) -> int: + """Auto-compute a good max images per chunk. + + The program tries to best-fit rather than forcing a conservative static cap. + Priority: + 1. GRAPHIFY_get_max_images_per_chunk() env var (explicit user override) + 2. Backend-aware sensible defaults for modern vision models + 3. Token-budget aware soft cap (don't waste most of the budget on images) + 4. Hard safety floor + + claude-cli is kept lower because each image becomes a Read tool call. + High-capacity backends (current Claude/OpenAI/Gemini) support 40-100+. + """ + # Explicit override wins (user says "don't leave it to the user unless necessary") + raw = os.environ.get("GRAPHIFY_get_max_images_per_chunk()", "").strip() + if raw: + try: + val = int(raw) + if val > 0: + return val + except (ValueError, TypeError): + print(f"[graphify] GRAPHIFY_get_max_images_per_chunk()={raw!r} ignored", file=sys.stderr) + + # Backend-aware defaults (2026-era vision models) + backend = (backend or "").lower() + if backend == "claude-cli": + base = 15 # practical limit for repeated Read tool calls in one turn + elif backend in ("claude", "bedrock"): + base = 60 # modern Anthropic supports high image counts + elif backend in ("openai", "gemini", "azure", "kimi"): + base = 50 + else: + base = 30 + + # Best-fit against token budget when available + if token_budget and token_budget > 0: + # Leave headroom for text + output tokens. Images are expensive but fixed-cost. + budget_headroom = max(8000, int(token_budget * 0.35)) + from_budget = max(5, budget_headroom // _IMAGE_TOKEN_ESTIMATE) + base = min(base, from_budget) + + return max(5, min(base, 100)) # absolute sanity bounds + + # Backends that read an image by file path (claude-cli's Read tool) # instead of inlining base64. They open the file themselves and downsample as -# needed, so `_MAX_IMAGE_BYTES` does not apply and the bytes never need loading. +# needed, so image byte limits do not apply and the bytes never need loading. _PATH_IMAGE_BACKENDS = {"claude-cli"} @@ -757,7 +822,7 @@ def _bind_node_evidence(result: dict, text_units: "list[Path | FileSlice]", root class _ImageRef: """A single image destined for a vision request. - `raw` is None when the image is unreadable or exceeds `_MAX_IMAGE_BYTES`, or + `raw` is None when the image is unreadable or exceeds the configured per-image limit (_get_max_image_bytes), or when the target backend has no vision support — in every such case the renderers emit a text reference instead of pixels, so the image still becomes a graph node. @@ -799,7 +864,7 @@ def _build_image_refs(image_files: list[Path], root: Path, *, read_bytes: bool = """Build `_ImageRef`s for raster images. `read_bytes=True` (base64 backends) loads the pixels and drops any image over - `_MAX_IMAGE_BYTES` to a reference, because a base64 request body has a hard + `_get_max_image_bytes()` to a reference, because a base64 request body has a hard size ceiling. `read_bytes=False` (path-based backends — claude-cli) skips the read entirely: those backends open the file themselves and downsample as needed, so there is no per-image size limit and no reason to @@ -1821,9 +1886,15 @@ def _estimate_file_tokens(unit: "Path | FileSlice") -> int: def _pack_chunks_by_tokens( files: "list[Path | FileSlice]", token_budget: int, + *, + backend: str | None = None, ) -> "list[list[Path | FileSlice]]": """Greedily pack files/slices into chunks that fit a token budget. + Image count per chunk is auto-tuned (see _get_max_images_per_chunk). + """ + """Greedily pack files/slices into chunks that fit a token budget. + Units are first grouped by parent directory so related artifacts share a chunk (cross-file edges are more likely to be extracted within a chunk than across chunks). Within each directory, units are added one at a @@ -1849,7 +1920,7 @@ def _pack_chunks_by_tokens( cost = _estimate_file_tokens(unit) is_image = not isinstance(unit, FileSlice) and _is_vision_image(unit) over_budget = current_tokens + cost > token_budget - over_images = is_image and current_images >= _MAX_IMAGES_PER_CHUNK + over_images = is_image and current_images >= _get_max_images_per_chunk(backend=backend, token_budget=token_budget) if current and (over_budget or over_images): chunks.append(current) current = [] @@ -2214,7 +2285,7 @@ def extract_corpus_parallel( # silently dropped (#1369). Files at/under the cap pass through unchanged. files = expand_oversized_files(files, _FILE_CHAR_CAP) if token_budget is not None: - chunks = _pack_chunks_by_tokens(files, token_budget=token_budget) + chunks = _pack_chunks_by_tokens(files, token_budget=token_budget, backend=backend) else: chunks = [files[i:i + chunk_size] for i in range(0, len(files), chunk_size)] diff --git a/graphify/multigraph_compat.py b/graphify/multigraph_compat.py index 7ac62e275..927012778 100644 --- a/graphify/multigraph_compat.py +++ b/graphify/multigraph_compat.py @@ -2,8 +2,8 @@ Verifies that the current NetworkX runtime supports the behaviors a future opt-in --multigraph build will rely on. The probe is BEHAVIOR-based, not -version-based — both NX 3.4.2 (Py 3.10 lane) and NX 3.6.1+ (Py 3.11+ lane) -pass. The probe result is cached for the process lifetime via lru_cache. +version-based. +The probe result is cached for the process lifetime via lru_cache. No call sites added yet; downstream multigraph PRs will gate on require_multigraph_capabilities() before enabling MDG mode. diff --git a/graphify/presentation.py b/graphify/presentation.py new file mode 100644 index 000000000..d30e02997 --- /dev/null +++ b/graphify/presentation.py @@ -0,0 +1,1869 @@ +"""Bounded semantic extraction for untrusted PowerPoint ``.pptx`` packages. + +A presentation is converted into a content-versioned artifact bundle containing +structured Markdown plus extracted images, audio/video, and embedded files. The +module reads OOXML directly so it can recover relationships and media that +``python-pptx`` does not expose, and it never executes or fetches package content. +""" + +from __future__ import annotations + +import hashlib +import io +import json +import os +import posixpath +import re +import shutil +import unicodedata +import uuid +import zipfile +from dataclasses import asdict, dataclass, fields +from pathlib import Path, PurePosixPath +from typing import TYPE_CHECKING, Iterable +from urllib.parse import unquote, urlsplit + +from defusedxml.ElementTree import ParseError as _XmlParseError +from defusedxml.ElementTree import fromstring as _safe_xml_fromstring +from defusedxml.common import DefusedXmlException + +if TYPE_CHECKING: + from xml.etree.ElementTree import Element as ETElement +else: # pragma: no cover - runtime alias for annotations + ETElement = object + + +CONVERTER_VERSION = "2" +MANIFEST_SCHEMA_VERSION = 2 + +_REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" +_PACKAGE_REL_NS = "http://schemas.openxmlformats.org/package/2006/relationships" +_CONTENT_TYPES_PART = "[Content_Types].xml" +_ROOT_RELS_PART = "_rels/.rels" + +_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg", ".bmp", ".tif", ".tiff"} +_SEMANTIC_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".svg"} +_AUDIO_EXTENSIONS = {".mp3", ".wav", ".m4a", ".ogg", ".aac", ".wma", ".flac"} +_VIDEO_EXTENSIONS = {".mp4", ".mov", ".webm", ".mkv", ".avi", ".m4v", ".wmv", ".mpeg", ".mpg"} +_MIME_EXTENSIONS = { + "image/png": ".png", + "image/jpeg": ".jpg", + "image/gif": ".gif", + "image/webp": ".webp", + "image/svg+xml": ".svg", + "image/bmp": ".bmp", + "image/tiff": ".tiff", + "audio/mpeg": ".mp3", + "audio/mp4": ".m4a", + "audio/wav": ".wav", + "audio/x-wav": ".wav", + "audio/ogg": ".ogg", + "video/mp4": ".mp4", + "video/quicktime": ".mov", + "video/webm": ".webm", + "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": ".xlsx", + "application/vnd.openxmlformats-officedocument.wordprocessingml.document": ".docx", + "application/pdf": ".pdf", +} + + +class PresentationError(ValueError): + """Raised when a presentation is malformed, unsafe, or outside configured bounds.""" + + +@dataclass(frozen=True) +class PresentationLimits: + """Resource policy for one PPTX conversion.""" + + max_raw_bytes: int = 50 * 1024 * 1024 + max_members: int = 10_000 + max_decompressed_bytes: int = 512 * 1024 * 1024 + max_member_bytes: int = 64 * 1024 * 1024 + max_xml_bytes: int = 8 * 1024 * 1024 + max_xml_total_bytes: int = 64 * 1024 * 1024 + max_compression_ratio: int = 200 + max_slides: int = 1_000 + max_relationships_per_part: int = 2_000 + max_assets: int = 2_000 + max_asset_bytes: int = 50 * 1024 * 1024 + max_extracted_asset_bytes: int = 256 * 1024 * 1024 + max_text_chars: int = 20_000 + max_markdown_chars: int = 2_000_000 + max_external_target_chars: int = 4_096 + max_xml_elements: int = 250_000 + max_xml_depth: int = 64 + max_manifest_bytes: int = 2_000_000 + + def as_policy(self) -> dict[str, int]: + """Stable policy dict used for cache identity and verification.""" + return {field.name: int(getattr(self, field.name)) for field in fields(self)} + + +@dataclass(frozen=True) +class PresentationArtifacts: + markdown_path: Path + manifest_path: Path + images: tuple[Path, ...] + media: tuple[Path, ...] + attachments: tuple[Path, ...] + warnings: tuple[str, ...] + + +@dataclass(frozen=True) +class _Relationship: + relationship_id: str + relationship_type: str + target: str + external: bool + resolved_target: str | None + + @property + def kind(self) -> str: + return self.relationship_type.rstrip("/").rsplit("/", 1)[-1].lower() + + +@dataclass +class _Asset: + kind: str + role: str + slide: int + relationship_id: str + source_part: str + content_type: str + filename: str + payload: bytes + description: str = "" + + @property + def sha256(self) -> str: + return hashlib.sha256(self.payload).hexdigest() + + @property + def size(self) -> int: + return len(self.payload) + + +@dataclass +class _Extraction: + markdown: str + assets: list[_Asset] + slides: list[dict] + warnings: list[str] + presentation_part: str + + +class _Package: + def __init__(self, raw: bytes, limits: PresentationLimits): + self.raw = raw + self.limits = limits + self.archive = zipfile.ZipFile(io.BytesIO(raw)) + self.infos: dict[str, zipfile.ZipInfo] = {} + self.xml_bytes_read = 0 + self.asset_bytes_read = 0 + self._validate_archive() + self.content_types = self._load_content_types() + + def close(self) -> None: + self.archive.close() + + def _validate_archive(self) -> None: + infos = self.archive.infolist() + if len(infos) > self.limits.max_members: + raise PresentationError( + f"PPTX has {len(infos)} archive members; limit is {self.limits.max_members} members" + ) + declared_total = 0 + compressed_total = 0 + for info in infos: + name = info.filename + if not _safe_member_name(name): + raise PresentationError(f"unsafe PPTX archive member name: {name!r}") + if name in self.infos: + raise PresentationError(f"duplicate PPTX archive member: {name!r}") + if info.flag_bits & 0x1: + raise PresentationError(f"encrypted PPTX archive member is not supported: {name!r}") + if info.file_size > self.limits.max_member_bytes: + raise PresentationError( + f"PPTX member {name!r} declares {info.file_size} bytes; " + f"per-member limit is {self.limits.max_member_bytes}" + ) + declared_total += info.file_size + compressed_total += info.compress_size + self.infos[name] = info + if declared_total > self.limits.max_decompressed_bytes: + raise PresentationError( + f"PPTX declares {declared_total} decompressed bytes; " + f"limit is {self.limits.max_decompressed_bytes}" + ) + if ( + declared_total + and declared_total / max(compressed_total, 1) > self.limits.max_compression_ratio + ): + raise PresentationError("PPTX compression ratio exceeds the configured safety limit") + if _CONTENT_TYPES_PART not in self.infos or _ROOT_RELS_PART not in self.infos: + raise PresentationError("not a PPTX package: required presentation parts are missing") + + def read_member(self, name: str, *, cap: int | None = None, asset: bool = False) -> bytes: + info = self.infos.get(name) + if info is None: + raise PresentationError(f"PPTX relationship target is missing: {name}") + ceiling = min(cap or self.limits.max_member_bytes, self.limits.max_member_bytes) + chunks: list[bytes] = [] + total = 0 + try: + with self.archive.open(info) as member: + while True: + chunk = member.read(min(1024 * 1024, ceiling + 1 - total)) + if not chunk: + break + total += len(chunk) + if total > ceiling: + raise PresentationError( + f"PPTX member {name!r} exceeds the {ceiling}-byte extraction limit" + ) + chunks.append(chunk) + except (zipfile.BadZipFile, EOFError, RuntimeError, OSError) as exc: + raise PresentationError(f"could not read PPTX member {name!r}: {exc}") from exc + if asset: + self.asset_bytes_read += total + if self.asset_bytes_read > self.limits.max_extracted_asset_bytes: + raise PresentationError( + "PPTX extracted assets exceed the configured total byte limit" + ) + return b"".join(chunks) + + def parse_xml(self, name: str) -> ETElement: + data = self.read_member(name, cap=self.limits.max_xml_bytes) + self.xml_bytes_read += len(data) + if self.xml_bytes_read > self.limits.max_xml_total_bytes: + raise PresentationError("PPTX XML parts exceed the configured total byte limit") + # Reject DTDs/entities before parse. Byte-pattern screening alone is not + # enough: UTF-16 / UTF-32 payloads hide ASCII markers. defusedxml forbids + # entities and external resolution for every encoding it can decode. + try: + root = _safe_xml_fromstring( + data, + forbid_dtd=True, + forbid_entities=True, + forbid_external=True, + ) + except DefusedXmlException as exc: + raise PresentationError( + f"DTD/entity declarations are forbidden in PPTX XML: {name}" + ) from exc + except _XmlParseError as exc: + raise PresentationError(f"malformed PPTX XML in {name}: {exc}") from exc + except (LookupError, UnicodeError, ValueError, TypeError) as exc: + raise PresentationError(f"unreadable PPTX XML encoding in {name}: {exc}") from exc + _enforce_xml_bounds(root, name, self.limits) + return root + + def _load_content_types(self) -> tuple[dict[str, str], dict[str, str]]: + root = self.parse_xml(_CONTENT_TYPES_PART) + defaults: dict[str, str] = {} + overrides: dict[str, str] = {} + for element in root: + local = _local(element.tag) + if local == "Default": + defaults[element.attrib.get("Extension", "").lower()] = element.attrib.get( + "ContentType", "application/octet-stream" + ) + elif local == "Override": + name = element.attrib.get("PartName", "").lstrip("/") + if name: + overrides[name] = element.attrib.get("ContentType", "application/octet-stream") + return defaults, overrides + + def content_type(self, part: str) -> str: + defaults, overrides = self.content_types + if part in overrides: + return overrides[part] + extension = PurePosixPath(part).suffix.lstrip(".").lower() + return defaults.get(extension, "application/octet-stream") + + +class _Markdown: + def __init__(self, limit: int): + self.limit = limit + self.parts: list[str] = [] + self.size = 0 + self.truncated = False + + def add(self, block: str) -> None: + block = block.strip("\n") + if not block or self.truncated: + return + rendered = block + "\n\n" + if self.size + len(rendered) <= self.limit: + self.parts.append(rendered) + self.size += len(rendered) + return + notice = "\n\n> [!WARNING]\n> Presentation Markdown truncated at a safe section boundary.\n" + room = self.limit - self.size + if room >= len(notice): + self.parts.append(notice) + self.size += len(notice) + self.truncated = True + + def render(self) -> str: + return "".join(self.parts).rstrip() + "\n" + + +def _local(tag: str) -> str: + return tag.rsplit("}", 1)[-1] + + +def _safe_member_name(name: str) -> bool: + if not name or "\x00" in name or "\\" in name or name.startswith("/"): + return False + path = PurePosixPath(name) + return not path.is_absolute() and all(part not in {"", ".", ".."} for part in path.parts) + + +def _read_source(path: Path, limit: int) -> bytes: + try: + with path.open("rb") as source: + chunks: list[bytes] = [] + total = 0 + while True: + chunk = source.read(min(1024 * 1024, limit + 1 - total)) + if not chunk: + break + total += len(chunk) + if total > limit: + raise PresentationError( + f"PPTX source exceeds the configured {limit}-byte raw input limit" + ) + chunks.append(chunk) + except PresentationError: + raise + except OSError as exc: + raise PresentationError(f"could not read PPTX source {path}: {exc}") from exc + return b"".join(chunks) + + +def _rels_part(part: str) -> str: + directory, name = posixpath.split(part) + return posixpath.join(directory, "_rels", name + ".rels") + + +def _resolve_target(source_part: str, target: str) -> str | None: + decoded = unquote(target) + parsed = urlsplit(decoded) + if parsed.scheme or parsed.netloc: + return None + decoded = parsed.path + if not decoded or "\x00" in decoded or "\\" in decoded or decoded.startswith("/"): + return None + resolved = posixpath.normpath(posixpath.join(posixpath.dirname(source_part), decoded)) + if resolved == ".." or resolved.startswith("../") or not _safe_member_name(resolved): + return None + return resolved + + +def _relationships(package: _Package, part: str) -> dict[str, _Relationship]: + rels_name = _rels_part(part) + if rels_name not in package.infos: + return {} + root = package.parse_xml(rels_name) + children = [element for element in root if _local(element.tag) == "Relationship"] + if len(children) > package.limits.max_relationships_per_part: + raise PresentationError( + f"PPTX part {part!r} exceeds the relationship limit of " + f"{package.limits.max_relationships_per_part}" + ) + relationships: dict[str, _Relationship] = {} + for element in children: + relationship_id = element.attrib.get("Id", "") + relationship_type = element.attrib.get("Type", "") + target = element.attrib.get("Target", "") + if not relationship_id or relationship_id in relationships: + raise PresentationError(f"missing or duplicate relationship ID in {_rels_part(part)}") + external = element.attrib.get("TargetMode", "").lower() == "external" + resolved = None if external else _resolve_target(part, target) + relationships[relationship_id] = _Relationship( + relationship_id, relationship_type, target, external, resolved + ) + return relationships + + +def _relationship_attr(element: ETElement, local_name: str) -> str | None: + exact = f"{{{_REL_NS}}}{local_name}" + if exact in element.attrib: + return element.attrib[exact] + for key, value in element.attrib.items(): + if _local(key) == local_name and key.startswith("{"): + return value + return None + + +def _first(element: ETElement, local_name: str) -> ETElement | None: + return next((item for item in element.iter() if _local(item.tag) == local_name), None) + + +def _children(element: ETElement, local_name: str) -> list[ETElement]: + return [item for item in element if _local(item.tag) == local_name] + + +def _clean_text(value: str | None, limit: int) -> str: + if not value: + return "" + value = value.replace("\x00", "�").replace("\r\n", "\n").replace("\r", "\n") + value = "".join(char for char in value if char in "\n\t" or ord(char) >= 32) + value = re.sub(r"[ \t]+", " ", value).strip() + if len(value) > limit: + return value[: max(0, limit - 15)] + "… [truncated]" + return value + + +def _all_text(element: ETElement, limit: int) -> str: + values = [item.text or "" for item in element.iter() if _local(item.tag) in {"t", "text"}] + return _clean_text(" ".join(value for value in values if value), limit) + + +def _paragraphs(element: ETElement, limit: int) -> list[tuple[int, str, list[str]]]: + paragraphs: list[tuple[int, str, list[str]]] = [] + for paragraph in (item for item in element.iter() if _local(item.tag) == "p"): + pieces: list[str] = [] + links: list[str] = [] + for item in paragraph.iter(): + local = _local(item.tag) + if local == "t" and item.text: + pieces.append(item.text) + elif local == "br": + pieces.append("\n") + elif local in {"hlinkClick", "hlinkMouseOver"}: + relationship_id = _relationship_attr(item, "id") + if relationship_id: + links.append(relationship_id) + text = _clean_text("".join(pieces), limit) + if not text: + continue + properties = next( + (item for item in paragraph if _local(item.tag) in {"pPr", "endParaRPr"}), None + ) + try: + level = int(properties.attrib.get("lvl", "0")) if properties is not None else 0 + except ValueError: + level = 0 + paragraphs.append((max(0, min(level, 8)), text, links)) + return paragraphs + + +def _shape_metadata(shape: ETElement, limit: int) -> dict[str, str]: + properties = _first(shape, "cNvPr") + result = {"id": "", "name": "", "description": "", "title": "", "placeholder": ""} + if properties is not None: + result.update( + { + "id": properties.attrib.get("id", ""), + "name": _clean_text(properties.attrib.get("name"), limit), + "description": _clean_text(properties.attrib.get("descr"), limit), + "title": _clean_text(properties.attrib.get("title"), limit), + } + ) + placeholder = _first(shape, "ph") + if placeholder is not None: + result["placeholder"] = placeholder.attrib.get("type", "body") + transform = _first(shape, "xfrm") + if transform is not None: + offset = next((item for item in transform if _local(item.tag) == "off"), None) + extent = next((item for item in transform if _local(item.tag) == "ext"), None) + if offset is not None: + result["x"] = offset.attrib.get("x", "") + result["y"] = offset.attrib.get("y", "") + if extent is not None: + result["width"] = extent.attrib.get("cx", "") + result["height"] = extent.attrib.get("cy", "") + return result + + +def _walk_shapes(tree: ETElement) -> Iterable[ETElement]: + shape_tags = {"sp", "pic", "graphicFrame", "cxnSp", "grpSp", "oleObj", "contentPart"} + for child in tree: + local = _local(child.tag) + if local not in shape_tags: + continue + yield child + if local == "grpSp": + yield from _walk_shapes(child) + + +def _used_relationship_ids(element: ETElement, known: set[str]) -> set[str]: + used: set[str] = set() + for item in element.iter(): + for value in item.attrib.values(): + if value in known: + used.add(value) + return used + + +def _markdown_table(rows: list[list[str]]) -> str: + if not rows: + return "" + width = max(len(row) for row in rows) + normalized = [row + [""] * (width - len(row)) for row in rows] + + def escape(value: str) -> str: + return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "
") + + lines = ["| " + " | ".join(escape(cell) for cell in normalized[0]) + " |"] + lines.append("| " + " | ".join("---" for _ in range(width)) + " |") + lines.extend("| " + " | ".join(escape(cell) for cell in row) + " |" for row in normalized[1:]) + return "\n".join(lines) + + +def _chart_values(container: ETElement | None, limit: int) -> list[str]: + if container is None: + return [] + points: list[tuple[int, str]] = [] + for point in (item for item in container.iter() if _local(item.tag) == "pt"): + value_element = next((item for item in point if _local(item.tag) in {"v", "f"}), None) + if value_element is None: + value_element = _first(point, "v") + value = _clean_text(value_element.text if value_element is not None else "", limit) + try: + index = int(point.attrib.get("idx", len(points))) + except ValueError: + index = len(points) + points.append((index, value)) + if points: + return [value for _, value in sorted(points)] + return [ + _clean_text(item.text, limit) + for item in container.iter() + if _local(item.tag) == "v" and _clean_text(item.text, limit) + ] + + +def _extract_chart( + package: _Package, part: str, limits: PresentationLimits +) -> tuple[str, list[str]]: + root = package.parse_xml(part) + lines: list[str] = [] + title = next((item for item in root.iter() if _local(item.tag) == "title"), None) + title_text = _all_text(title, limits.max_text_chars) if title is not None else "" + if title_text: + lines.append(f"**Chart title:** {title_text}") + series_elements = [item for item in root.iter() if _local(item.tag) in {"ser", "series"}] + for series_index, series in enumerate(series_elements, start=1): + tx = next((item for item in series if _local(item.tag) == "tx"), None) + name = _all_text(tx, limits.max_text_chars) if tx is not None else "" + if not name and tx is not None: + name = next( + ( + _clean_text(item.text, limits.max_text_chars) + for item in tx.iter() + if _local(item.tag) == "v" and item.text + ), + "", + ) + name = name or f"Series {series_index}" + category = next((item for item in series if _local(item.tag) in {"cat", "xVal"}), None) + values = next( + (item for item in series if _local(item.tag) in {"val", "yVal", "bubbleSize"}), None + ) + categories = _chart_values(category, limits.max_text_chars) + numeric = _chart_values(values, limits.max_text_chars) + rows = [["Category", name]] + length = max(len(categories), len(numeric)) + for index in range(length): + rows.append( + [ + categories[index] if index < len(categories) else str(index + 1), + numeric[index] if index < len(numeric) else "", + ] + ) + lines.append(f"**Series:** {name}") + if length: + lines.append(_markdown_table(rows)) + if series_elements and all(_local(item.tag) == "series" for item in series_elements): + # Office 2016+ extended charts can keep dimensions in a shared cx:data + # area referenced by dataId rather than nesting cat/val under each series. + # Preserve the bounded cached evidence even when a series-to-dimension + # join is unavailable. + cached_values: list[str] = [] + for element in root.iter(): + if _local(element.tag) not in {"v", "t"} or not element.text: + continue + value = _clean_text(element.text, limits.max_text_chars) + if value and value not in cached_values: + cached_values.append(value) + if len(cached_values) >= 1_000: + break + if cached_values: + lines.append( + "**Extended-chart cached values:**\n" + + "\n".join(f"- {value}" for value in cached_values) + ) + attachment_parts: list[str] = [] + for relationship in _relationships(package, part).values(): + if relationship.kind in {"package", "oleobject"} and relationship.resolved_target: + attachment_parts.append(relationship.resolved_target) + return "\n\n".join(lines), attachment_parts + + +def _extract_diagram(package: _Package, part: str, limits: PresentationLimits) -> str: + root = package.parse_xml(part) + labels: dict[str, str] = {} + for point in (item for item in root.iter() if _local(item.tag) == "pt"): + model_id = point.attrib.get("modelId", "") + text = _all_text(point, limits.max_text_chars) + if model_id and text: + labels[model_id] = text + lines = [f"- Node `{model_id}`: {label}" for model_id, label in labels.items()] + for connection in (item for item in root.iter() if _local(item.tag) == "cxn"): + source = connection.attrib.get("srcId", "") + target = connection.attrib.get("destId", "") + if not source or not target: + continue + relation = connection.attrib.get("type", "connection") + lines.append(f"- {labels.get(source, source)} -> {labels.get(target, target)} ({relation})") + return "\n".join(lines) + + +def _core_properties( + package: _Package, limits: PresentationLimits, core_part: str | None +) -> dict[str, str]: + if not core_part or core_part not in package.infos: + return {} + root = package.parse_xml(core_part) + allowed = { + "title", + "subject", + "creator", + "keywords", + "description", + "lastModifiedBy", + "created", + "modified", + "category", + "contentStatus", + } + properties: dict[str, str] = {} + for element in root.iter(): + local = _local(element.tag) + if local in allowed and element.text: + properties[local] = _clean_text(element.text, limits.max_text_chars) + return properties + + +def _comment_authors( + package: _Package, + limits: PresentationLimits, + presentation_relationships: dict[str, _Relationship], +) -> dict[str, str]: + authors: dict[str, str] = {} + relationships = [ + relationship + for relationship in presentation_relationships.values() + if relationship.kind in {"commentauthors", "person", "persons"} + and not relationship.external + and relationship.resolved_target in package.infos + ] + for relationship in relationships: + assert relationship.resolved_target is not None + root = package.parse_xml(relationship.resolved_target) + for element in root.iter(): + if _local(element.tag) not in {"cmAuthor", "person"}: + continue + identifier = ( + element.attrib.get("id") + or element.attrib.get("authorId") + or element.attrib.get("userId") + ) + name = ( + element.attrib.get("name") + or element.attrib.get("displayName") + or element.attrib.get("initials") + ) + if identifier and name: + authors[identifier] = _clean_text(name, limits.max_text_chars) + return authors + + +def _safe_extension(part: str, content_type: str, kind: str) -> str: + suffix = PurePosixPath(part).suffix.lower() + if not re.fullmatch(r"\.[a-z0-9]{1,10}", suffix): + suffix = "" + mapped = _MIME_EXTENSIONS.get(content_type.lower(), "") + if kind == "image" and suffix not in _IMAGE_EXTENSIONS: + suffix = mapped if mapped in _IMAGE_EXTENSIONS else ".bin" + elif kind == "media" and suffix not in _AUDIO_EXTENSIONS | _VIDEO_EXTENSIONS: + suffix = mapped if mapped in _AUDIO_EXTENSIONS | _VIDEO_EXTENSIONS else ".bin" + elif kind == "attachment" and not suffix: + suffix = mapped or ".bin" + return suffix or ".bin" + + +def _valid_semantic_image(payload: bytes, extension: str) -> bool: + if extension == ".png": + return payload.startswith(b"\x89PNG\r\n\x1a\n") + if extension in {".jpg", ".jpeg"}: + return payload.startswith(b"\xff\xd8\xff") + if extension == ".gif": + return payload.startswith((b"GIF87a", b"GIF89a")) + if extension == ".webp": + return len(payload) >= 12 and payload[:4] == b"RIFF" and payload[8:12] == b"WEBP" + if extension == ".svg": + prefix = payload[:4096].lstrip(b"\xef\xbb\xbf\t\r\n ").lower() + return b" str: + suffix = PurePosixPath(part).suffix.lower() + if suffix in _AUDIO_EXTENSIONS or content_type.lower().startswith("audio/"): + return "audio" + if suffix in _VIDEO_EXTENSIONS or content_type.lower().startswith("video/"): + return "video" + return "media" + + +def _asset_filename( + bundle_token: str, + slide_number: int, + role: str, + index: int, + content_digest: str, + extension: str, +) -> str: + # Graphify's Whisper cache is basename-keyed. Include both the stable deck + # identity and the asset content digest so media from different decks cannot + # collide and changed media never reuses a stale transcript. + return f"{bundle_token}-slide-{slide_number:04d}-{role}-{index:02d}-{content_digest}{extension}" + + +def _extract_presentation( + package: _Package, + source_name: str, + source_sha256: str, + limits: PresentationLimits, + bundle_reference: str, +) -> _Extraction: + warnings: list[str] = [] + assets: list[_Asset] = [] + slides_manifest: list[dict] = [] + markdown = _Markdown(limits.max_markdown_chars) + root_relationships = _relationships(package, "") + main_relationship = next( + ( + relationship + for relationship in root_relationships.values() + if relationship.kind == "officedocument" + and not relationship.external + and relationship.resolved_target + ), + None, + ) + if main_relationship is None or main_relationship.resolved_target not in package.infos: + raise PresentationError("package has no valid root officeDocument relationship") + presentation_part = main_relationship.resolved_target + core_relationship = next( + ( + relationship + for relationship in root_relationships.values() + if relationship.kind in {"core-properties", "coreproperties"} + and not relationship.external + and relationship.resolved_target + ), + None, + ) + properties = _core_properties( + package, + limits, + core_relationship.resolved_target if core_relationship else None, + ) + title = properties.get("title") or Path(source_name).stem + markdown.add(f"# Presentation: {title}") + metadata_lines = [ + f"- **Source:** `{source_name}`", + f"- **SHA-256:** `{source_sha256}`", + f"- **Converter version:** `{CONVERTER_VERSION}`", + f"- **Main package part:** `{presentation_part}`", + ] + for key, value in properties.items(): + metadata_lines.append(f"- **{key}:** {value}") + markdown.add("## Presentation metadata\n\n" + "\n".join(metadata_lines)) + + presentation = package.parse_xml(presentation_part) + presentation_rels = _relationships(package, presentation_part) + section_by_slide_id: dict[str, str] = {} + for section in (item for item in presentation.iter() if _local(item.tag) == "section"): + section_name = _clean_text(section.attrib.get("name"), limits.max_text_chars) + for slide_id in (item for item in section.iter() if _local(item.tag) == "sldId"): + identifier = slide_id.attrib.get("id", "") + if identifier and section_name: + section_by_slide_id[identifier] = section_name + + slide_list = next((item for item in presentation if _local(item.tag) == "sldIdLst"), None) + slide_entries = _children(slide_list, "sldId") if slide_list is not None else [] + if len(slide_entries) > limits.max_slides: + raise PresentationError( + f"PPTX has {len(slide_entries)} slides; limit is {limits.max_slides}" + ) + authors = _comment_authors(package, limits, presentation_rels) + asset_counters: dict[tuple[int, str], int] = {} + extracted_relationships: set[tuple[int, str]] = set() + payload_cache: dict[str, bytes] = {} + asset_part_cache: dict[tuple[str, str], _Asset] = {} + bundle_token = re.sub(r"[^A-Za-z0-9._-]+", "_", PurePosixPath(bundle_reference).name) + template_cache: dict[str, dict] = {} + template_evidence: list[str] = [] + template_evidence_emitted: set[str] = set() + + def add_asset( + slide_number: int, + relationship: _Relationship, + kind: str, + *, + description: str = "", + forced_part: str | None = None, + ) -> _Asset | None: + part = forced_part or relationship.resolved_target + key = (slide_number, kind + ":" + (part or relationship.target)) + if key in extracted_relationships: + return None + extracted_relationships.add(key) + if not part: + return None + if part not in package.infos: + warnings.append( + f"Slide {slide_number}: relationship {relationship.relationship_id} target is missing: {part}" + ) + return None + if len(assets) >= limits.max_assets: + raise PresentationError( + f"PPTX exceeds the extracted asset limit of {limits.max_assets}" + ) + content_type = package.content_type(part) + role = _media_role(part, content_type) if kind == "media" else kind + counter_key = (slide_number, role) + asset_counters[counter_key] = asset_counters.get(counter_key, 0) + 1 + extension = _safe_extension(part, content_type, kind) + cache_key = (kind, part) + cached = asset_part_cache.get(cache_key) + if cached is not None and cached.kind in {kind, "attachment"}: + # Reuse bytes/filename for the same package part when a later slide + # inherits the same layout/master media. Still emit a per-slide + # provenance row so each slide's association is preserved. + asset = _Asset( + kind=cached.kind, + role=role, + slide=slide_number, + relationship_id=relationship.relationship_id, + source_part=part, + content_type=cached.content_type, + filename=cached.filename, + payload=cached.payload, + description=description or cached.description, + ) + assets.append(asset) + return asset + if part in payload_cache: + payload = payload_cache[part] + else: + payload = package.read_member(part, cap=limits.max_asset_bytes, asset=True) + payload_cache[part] = payload + stored_kind = kind + if kind == "image" and extension not in _SEMANTIC_IMAGE_EXTENSIONS: + stored_kind = "attachment" + warnings.append( + f"Slide {slide_number}: image part {part} uses {extension}; preserved as an " + "attachment because Graphify vision cannot decode that format" + ) + elif kind == "image" and not _valid_semantic_image(payload, extension): + stored_kind = "attachment" + extension = ".bin" + warnings.append( + f"Slide {slide_number}: image part {part} does not match its declared format; " + "preserved as an inert binary attachment" + ) + content_digest = hashlib.sha256(payload).hexdigest()[:10] + filename = _asset_filename( + bundle_token, + slide_number, + role, + asset_counters[counter_key], + content_digest, + extension, + ) + asset = _Asset( + kind=stored_kind, + role=role, + slide=slide_number, + relationship_id=relationship.relationship_id, + source_part=part, + content_type=content_type, + filename=filename, + payload=payload, + description=description, + ) + assets.append(asset) + asset_part_cache[cache_key] = asset + return asset + + for slide_number, slide_entry in enumerate(slide_entries, start=1): + slide_id = slide_entry.attrib.get("id", "") + relationship_id = _relationship_attr(slide_entry, "id") + relationship = presentation_rels.get(relationship_id or "") + if relationship is None or relationship.external or not relationship.resolved_target: + warnings.append( + f"Slide {slide_number}: missing or unsafe slide relationship {relationship_id!r}" + ) + continue + slide_part = relationship.resolved_target + if slide_part not in package.infos: + warnings.append(f"Slide {slide_number}: slide part is missing: {slide_part}") + continue + slide = package.parse_xml(slide_part) + rels = _relationships(package, slide_part) + unsafe_used = { + rid + for rid in _used_relationship_ids(slide, set(rels)) + if not rels[rid].external and rels[rid].resolved_target is None + } + for rid in sorted(unsafe_used): + warnings.append( + f"Slide {slide_number}: unsafe relationship target ignored for {rid}: {rels[rid].target}" + ) + section = section_by_slide_id.get(slide_id, "") + hidden = slide.attrib.get("show", "1").lower() in {"0", "false", "off", "no"} + common_slide = _first(slide, "cSld") + slide_name = _clean_text( + common_slide.attrib.get("name") if common_slide is not None else "", + limits.max_text_chars, + ) + shape_tree = _first(slide, "spTree") + shapes = list(_walk_shapes(shape_tree)) if shape_tree is not None else [] + shape_labels: dict[str, str] = {} + title_text = "" + text_lines: list[str] = [] + table_blocks: list[str] = [] + chart_blocks: list[str] = [] + diagram_blocks: list[str] = [] + connector_blocks: list[str] = [] + link_lines: list[str] = [] + asset_lines: list[str] = [] + attachment_parts: list[tuple[str, str]] = [] + layout_part = "" + master_part = "" + + for shape in shapes: + local = _local(shape.tag) + metadata = _shape_metadata(shape, limits.max_text_chars) + paragraphs = ( + _paragraphs(shape, limits.max_text_chars) if local in {"sp", "cxnSp"} else [] + ) + label = ( + paragraphs[0][1] if paragraphs else metadata["name"] or f"shape {metadata['id']}" + ) + if metadata["id"]: + shape_labels[metadata["id"]] = label + if metadata["placeholder"] in {"title", "ctrTitle"} and paragraphs: + title_text = paragraphs[0][1] + if local == "sp": + descriptor = f"shape {metadata['id'] or '?'}" + if metadata["name"]: + descriptor += f" `{metadata['name']}`" + if metadata["placeholder"]: + descriptor += f" [{metadata['placeholder']}]" + if metadata.get("x"): + descriptor += ( + f" @ x={metadata.get('x')}, y={metadata.get('y')}, " + f"w={metadata.get('width')}, h={metadata.get('height')}" + ) + if metadata["description"]: + text_lines.append(f"- **Alt text ({descriptor}):** {metadata['description']}") + if metadata["title"]: + text_lines.append(f"- **Object title ({descriptor}):** {metadata['title']}") + for level, text, hyperlink_ids in paragraphs: + text_lines.append(f"{' ' * level}- [{descriptor}] {text}") + for hyperlink_id in hyperlink_ids: + hyperlink = rels.get(hyperlink_id) + if hyperlink and hyperlink.external: + link_lines.append(f"- {text}: {hyperlink.target}") + elif local == "graphicFrame": + table = _first(shape, "tbl") + if table is not None: + rows: list[list[str]] = [] + for row in (item for item in table if _local(item.tag) == "tr"): + rows.append( + [ + _all_text(cell, limits.max_text_chars) + for cell in row + if _local(cell.tag) == "tc" + ] + ) + block = _markdown_table(rows) + if block: + table_blocks.append(f"**{metadata['name'] or 'Table'}**\n\n{block}") + chart_element = _first(shape, "chart") + chart_id = ( + _relationship_attr(chart_element, "id") if chart_element is not None else None + ) + chart_relationship = rels.get(chart_id or "") + if chart_relationship and chart_relationship.resolved_target: + try: + chart_text, chart_attachments = _extract_chart( + package, chart_relationship.resolved_target, limits + ) + if chart_text: + chart_blocks.append( + f"**{metadata['name'] or 'Chart'}**\n\n{chart_text}" + ) + attachment_parts.extend( + (part, chart_id or "chart") for part in chart_attachments + ) + except PresentationError as exc: + warnings.append(f"Slide {slide_number}: chart extraction failed: {exc}") + rel_ids = _first(shape, "relIds") + diagram_id = _relationship_attr(rel_ids, "dm") if rel_ids is not None else None + diagram_relationship = rels.get(diagram_id or "") + if diagram_relationship and diagram_relationship.resolved_target: + try: + diagram_text = _extract_diagram( + package, diagram_relationship.resolved_target, limits + ) + if diagram_text: + diagram_blocks.append( + f"**{metadata['name'] or 'Diagram'}**\n\n{diagram_text}" + ) + except PresentationError as exc: + warnings.append(f"Slide {slide_number}: diagram extraction failed: {exc}") + + for shape in shapes: + if _local(shape.tag) != "cxnSp": + continue + metadata = _shape_metadata(shape, limits.max_text_chars) + start = _first(shape, "stCxn") + end = _first(shape, "endCxn") + start_id = start.attrib.get("id", "") if start is not None else "" + end_id = end.attrib.get("id", "") if end is not None else "" + if start_id or end_id: + connector_blocks.append( + f"- {shape_labels.get(start_id, start_id or '?')} -> " + f"{shape_labels.get(end_id, end_id or '?')} " + f"({metadata['name'] or 'connector'})" + ) + + used_ids = _used_relationship_ids(slide, set(rels)) + descriptions_by_id: dict[str, str] = {} + for shape in shapes: + metadata = _shape_metadata(shape, limits.max_text_chars) + description = metadata["description"] or metadata["title"] or metadata["name"] + for relationship_ref in _used_relationship_ids(shape, set(rels)): + if description: + descriptions_by_id.setdefault(relationship_ref, description) + for used_id in sorted(used_ids): + used_relationship = rels[used_id] + if used_relationship.external: + if len(used_relationship.target) <= limits.max_external_target_chars: + link_lines.append( + f"- External {used_relationship.kind}: {used_relationship.target} " + f"(relationship `{used_id}`; not fetched)" + ) + else: + warnings.append( + f"Slide {slide_number}: external target for {used_id} exceeds the URL length limit" + ) + continue + if used_relationship.resolved_target is None: + continue + if used_relationship.kind == "image": + asset = add_asset( + slide_number, + used_relationship, + "image", + description=descriptions_by_id.get(used_id, ""), + ) + elif used_relationship.kind in {"media", "audio", "video"}: + asset = add_asset( + slide_number, + used_relationship, + "media", + description=descriptions_by_id.get(used_id, ""), + ) + elif used_relationship.kind in {"oleobject", "package"}: + asset = add_asset( + slide_number, + used_relationship, + "attachment", + description=descriptions_by_id.get(used_id, ""), + ) + elif used_relationship.kind not in { + "chart", + "diagramdata", + "diagramlayout", + "diagramquickstyle", + "diagramcolors", + "notesslide", + "comments", + "comment", + "slidelayout", + "tags", + }: + content_type = package.content_type(used_relationship.resolved_target) + if content_type.startswith("image/"): + unknown_kind = "image" + elif content_type.startswith(("audio/", "video/")): + unknown_kind = "media" + else: + unknown_kind = "attachment" + asset = add_asset( + slide_number, + used_relationship, + unknown_kind, + description=( + descriptions_by_id.get(used_id, "") + or f"Embedded {used_relationship.kind} evidence" + ), + ) + else: + asset = None + if asset is not None: + reference = f"{bundle_reference}/assets/{asset.filename}" + description = f" — {asset.description}" if asset.description else "" + asset_lines.append( + f"- **{asset.role.title()}** `{reference}`{description} " + f"(relationship `{asset.relationship_id}`, package part `{asset.source_part}`)" + ) + + # Slide layouts and masters can contain static labels, logos, diagrams, + # or background evidence that is visible on the rendered slide but absent + # from slideN.xml. Cache parsed template metadata so every slide that + # shares a layout still records its full layout/master chain and inherits + # the same assets with per-slide provenance. + template_queue: list[tuple[str, str]] = [] + visited_templates: set[str] = set() + layout_relationship = next( + (item for item in rels.values() if item.kind == "slidelayout"), None + ) + if layout_relationship and layout_relationship.resolved_target: + layout_part = layout_relationship.resolved_target + template_queue.append((layout_part, "slide layout")) + while template_queue: + template_part, template_role = template_queue.pop(0) + if template_part in visited_templates or template_part not in package.infos: + continue + visited_templates.add(template_part) + cached_template = template_cache.get(template_part) + if cached_template is None: + try: + template_root = package.parse_xml(template_part) + template_rels = _relationships(package, template_part) + except PresentationError as exc: + warnings.append( + f"Slide {slide_number}: {template_role} extraction failed for " + f"{template_part}: {exc}" + ) + continue + template_text_lines: list[str] = [] + template_tree = _first(template_root, "spTree") + for template_shape in _walk_shapes( + template_tree if template_tree is not None else template_root + ): + template_metadata = _shape_metadata(template_shape, limits.max_text_chars) + for level, text, _ in _paragraphs(template_shape, limits.max_text_chars): + placeholder = ( + f" [{template_metadata['placeholder']}]" + if template_metadata["placeholder"] + else "" + ) + template_text_lines.append(f"{' ' * level}- {text}{placeholder}") + master_relationship = next( + (item for item in template_rels.values() if item.kind == "slidemaster"), + None, + ) + master_target = ( + master_relationship.resolved_target + if master_relationship and master_relationship.resolved_target + else "" + ) + template_assets: list[tuple[_Relationship, str]] = [] + template_used_ids = _used_relationship_ids(template_root, set(template_rels)) + for template_id in sorted(template_used_ids): + template_relationship = template_rels[template_id] + if template_relationship.external: + if len(template_relationship.target) <= limits.max_external_target_chars: + link_lines.append( + f"- {template_role.title()} external {template_relationship.kind}: " + f"{template_relationship.target} (not fetched)" + ) + continue + if not template_relationship.resolved_target: + warnings.append( + f"Slide {slide_number} {template_role}: unsafe relationship target " + f"ignored for {template_id}: {template_relationship.target}" + ) + continue + if template_relationship.kind == "image": + template_asset_kind = "image" + elif template_relationship.kind in {"media", "audio", "video"}: + template_asset_kind = "media" + elif template_relationship.kind in {"oleobject", "package"}: + template_asset_kind = "attachment" + else: + continue + template_assets.append((template_relationship, template_asset_kind)) + cached_template = { + "role": template_role, + "text_lines": template_text_lines, + "master_part": master_target, + "assets": template_assets, + } + template_cache[template_part] = cached_template + if ( + cached_template["text_lines"] + and template_part not in template_evidence_emitted + ): + template_evidence_emitted.add(template_part) + template_evidence.append( + f"### {template_role.title()}: `{template_part}`\n\n" + + "\n".join(cached_template["text_lines"]) + ) + if template_role == "slide layout" and cached_template["master_part"]: + master_part = cached_template["master_part"] + elif template_role == "slide master": + master_part = template_part + for template_relationship, template_asset_kind in cached_template["assets"]: + template_asset = add_asset( + slide_number, + template_relationship, + template_asset_kind, + description=f"Evidence inherited from {template_role}", + ) + if template_asset is not None: + asset_lines.append( + f"- **Inherited {template_asset.role.title()}** " + f"`{bundle_reference}/assets/{template_asset.filename}` — " + f"from {template_role} `{template_part}` " + f"(slide {slide_number})" + ) + if cached_template["master_part"]: + template_queue.append((cached_template["master_part"], "slide master")) + + for attachment_part, origin in attachment_parts: + synthetic = _Relationship( + f"chart-{origin}", + "http://schemas.openxmlformats.org/officeDocument/2006/relationships/package", + attachment_part, + False, + attachment_part, + ) + asset = add_asset( + slide_number, + synthetic, + "attachment", + description="Embedded chart workbook", + forced_part=attachment_part, + ) + if asset is not None: + asset_lines.append( + f"- **Attachment** `{bundle_reference}/assets/{asset.filename}` — " + f"Embedded chart workbook (package part `{asset.source_part}`)" + ) + + notes_blocks: list[str] = [] + comments_blocks: list[str] = [] + for slide_relationship in rels.values(): + if slide_relationship.external or not slide_relationship.resolved_target: + continue + if slide_relationship.kind == "notesslide": + try: + notes = package.parse_xml(slide_relationship.resolved_target) + notes_tree = _first(notes, "spTree") + for shape in _walk_shapes(notes_tree if notes_tree is not None else notes): + placeholder = _shape_metadata(shape, limits.max_text_chars)["placeholder"] + if placeholder in {"sldImg", "sldNum", "hdr", "ftr", "dt"}: + continue + for level, text, _ in _paragraphs(shape, limits.max_text_chars): + notes_blocks.append(f"{' ' * level}- {text}") + notes_rels = _relationships(package, slide_relationship.resolved_target) + notes_used_ids = _used_relationship_ids(notes, set(notes_rels)) + for notes_id in sorted(notes_used_ids): + notes_relationship = notes_rels[notes_id] + if notes_relationship.external: + if len(notes_relationship.target) <= limits.max_external_target_chars: + link_lines.append( + f"- Notes external {notes_relationship.kind}: " + f"{notes_relationship.target} (relationship `{notes_id}`; not fetched)" + ) + continue + if notes_relationship.resolved_target is None: + warnings.append( + f"Slide {slide_number} notes: unsafe relationship target ignored " + f"for {notes_id}: {notes_relationship.target}" + ) + continue + if notes_relationship.kind == "image": + notes_asset = add_asset( + slide_number, + notes_relationship, + "image", + description="Image embedded in speaker notes", + ) + elif notes_relationship.kind in {"media", "audio", "video"}: + notes_asset = add_asset( + slide_number, + notes_relationship, + "media", + description="Media embedded in speaker notes", + ) + elif notes_relationship.kind in {"oleobject", "package"}: + notes_asset = add_asset( + slide_number, + notes_relationship, + "attachment", + description="Attachment embedded in speaker notes", + ) + else: + notes_asset = None + if notes_asset is not None: + reference = f"{bundle_reference}/assets/{notes_asset.filename}" + asset_lines.append( + f"- **Notes {notes_asset.role.title()}** `{reference}` — " + f"{notes_asset.description} (relationship `{notes_id}`, " + f"package part `{notes_asset.source_part}`)" + ) + except PresentationError as exc: + warnings.append(f"Slide {slide_number}: notes extraction failed: {exc}") + elif slide_relationship.kind in {"comments", "comment"}: + try: + comments = package.parse_xml(slide_relationship.resolved_target) + for comment in ( + item for item in comments.iter() if _local(item.tag) in {"cm", "comment"} + ): + text = _all_text(comment, limits.max_text_chars) + if not text: + continue + author_id = comment.attrib.get("authorId", "") + author = authors.get( + author_id, f"author {author_id}" if author_id else "unknown" + ) + timestamp = comment.attrib.get("dt", "") + comments_blocks.append( + f"- **{author}**{f' ({timestamp})' if timestamp else ''}: {text}" + ) + except PresentationError as exc: + warnings.append(f"Slide {slide_number}: comment extraction failed: {exc}") + + transition = _first(slide, "transition") + transition_text = "" + if transition is not None: + effect = next((_local(item.tag) for item in transition), "unspecified") + attributes = ", ".join(f"{key}={value}" for key, value in transition.attrib.items()) + transition_text = f"- {effect}{f' ({attributes})' if attributes else ''}" + + heading = f"## Slide {slide_number}" + if title_text: + heading += f": {title_text}" + slide_meta = [ + f"- source part: `{slide_part}`", + f"- slide id: `{slide_id}`", + f"- hidden: {'true' if hidden else 'false'}", + ] + if slide_name: + slide_meta.append(f"- name: {slide_name}") + if section: + slide_meta.append(f"- section: {section}") + markdown.add(heading + "\n\n### Slide provenance\n\n" + "\n".join(slide_meta)) + if text_lines: + markdown.add("### Text and native shapes\n\n" + "\n".join(text_lines)) + if table_blocks: + markdown.add("### Tables\n\n" + "\n\n".join(table_blocks)) + if chart_blocks: + markdown.add("### Charts\n\n" + "\n\n".join(chart_blocks)) + if diagram_blocks or connector_blocks: + blocks = diagram_blocks + (["\n".join(connector_blocks)] if connector_blocks else []) + markdown.add("### Diagrams and connectors\n\n" + "\n\n".join(blocks)) + if asset_lines: + markdown.add("### Extracted visual and media evidence\n\n" + "\n".join(asset_lines)) + if notes_blocks: + markdown.add("### Speaker notes\n\n" + "\n".join(notes_blocks)) + if comments_blocks: + markdown.add("### Comments\n\n" + "\n".join(comments_blocks)) + if link_lines: + markdown.add( + "### Links and externally referenced media\n\n" + + "\n".join(dict.fromkeys(link_lines)) + ) + if transition_text: + markdown.add("### Transition\n\n" + transition_text) + + slides_manifest.append( + { + "number": slide_number, + "slide_id": slide_id, + "source_part": slide_part, + "title": title_text, + "name": slide_name, + "section": section, + "hidden": hidden, + "layout_part": layout_part, + "master_part": master_part, + } + ) + + if template_evidence: + markdown.add("## Inherited layout and master evidence\n\n" + "\n\n".join(template_evidence)) + + custom_xml_blocks: list[str] = [] + custom_parts = [ + name + for name in sorted(package.infos) + if name.startswith("customXml/") and name.lower().endswith(".xml") + ] + for custom_part in custom_parts[:100]: + try: + custom_root = package.parse_xml(custom_part) + except PresentationError as exc: + warnings.append(f"Custom XML part {custom_part} was skipped: {exc}") + continue + values = [ + _clean_text(value, limits.max_text_chars) + for value in custom_root.itertext() + if _clean_text(value, limits.max_text_chars) + ] + if values: + custom_xml_blocks.append( + f"### `{custom_part}`\n\n" + "\n".join(f"- {value}" for value in values[:1_000]) + ) + if len(custom_parts) > 100: + warnings.append(f"Custom XML inventory truncated after 100 of {len(custom_parts)} parts") + if custom_xml_blocks: + markdown.add("## Custom XML evidence\n\n" + "\n\n".join(custom_xml_blocks)) + + if warnings: + markdown.add( + "## Extraction warnings\n\n" + "\n".join(f"- {warning}" for warning in warnings) + ) + return _Extraction(markdown.render(), assets, slides_manifest, warnings, presentation_part) + + +def _source_identity(path: Path, root: Path | None) -> str: + if root is not None: + try: + return path.resolve().relative_to(root.resolve()).as_posix() + except (ValueError, OSError, RuntimeError): + pass + try: + return str(path.resolve()) + except (OSError, RuntimeError): + return str(path.absolute()) + + +def _bundle_name(path: Path, root: Path | None) -> str: + key = _source_identity(path, root) + normalized = unicodedata.normalize("NFC", key) + path_hash = hashlib.sha256(normalized.encode("utf-8")).hexdigest()[:8] + stem = re.sub(r"[^A-Za-z0-9._-]+", "_", unicodedata.normalize("NFC", path.stem)).strip("._-") + stem = stem[:80] or "presentation" + return f"{stem}_{path_hash}_pptx" + + +def _enforce_xml_bounds(root: ETElement, name: str, limits: PresentationLimits) -> None: + """Fail closed on deeply nested or extremely large XML trees.""" + element_count = 0 + stack: list[tuple[ETElement, int]] = [(root, 1)] + while stack: + node, depth = stack.pop() + element_count += 1 + if element_count > limits.max_xml_elements: + raise PresentationError( + f"PPTX XML {name!r} exceeds the {limits.max_xml_elements}-element safety limit" + ) + if depth > limits.max_xml_depth: + raise PresentationError( + f"PPTX XML {name!r} exceeds the {limits.max_xml_depth}-level nesting limit" + ) + for child in list(node): + stack.append((child, depth + 1)) + + +def _safe_bundle_relative(bundle: Path, relative: str) -> Path: + if not isinstance(relative, str) or not _safe_member_name(relative): + raise ValueError + path = bundle / Path(*PurePosixPath(relative).parts) + try: + path.resolve().relative_to(bundle.resolve()) + except (ValueError, OSError): + raise ValueError from None + return path + + +def _file_sha256(path: Path, *, cap: int) -> tuple[str, int]: + digest = hashlib.sha256() + total = 0 + with path.open("rb") as handle: + while True: + chunk = handle.read(1024 * 1024) + if not chunk: + break + total += len(chunk) + if total > cap: + raise ValueError + digest.update(chunk) + return digest.hexdigest(), total + + +def _artifact_paths( + bundle: Path, + manifest: dict, + *, + limits: PresentationLimits | None = None, + verify_integrity: bool = False, +) -> PresentationArtifacts | None: + """Resolve and optionally integrity-check a cached converter bundle. + + Cached bundles are treated as untrusted input: path containment, sizes, + hashes, kind/extension correspondence, and aggregate budgets are verified + before any artifact is returned to detection/LLM queues. + """ + policy = limits or PresentationLimits() + + def paths_for(kind: str) -> tuple[Path, ...]: + values = manifest.get(kind, []) + if not isinstance(values, list): + raise ValueError + paths: list[Path] = [] + for value in values: + paths.append(_safe_bundle_relative(bundle, value)) + return tuple(paths) + + try: + markdown = bundle / "presentation.md" + manifest_path = bundle / "manifest.json" + images = paths_for("images") + media = paths_for("media") + attachments = paths_for("attachments") + warnings = manifest.get("warnings", []) + if not isinstance(warnings, list) or not all(isinstance(item, str) for item in warnings): + raise ValueError + all_paths = (markdown, manifest_path, *images, *media, *attachments) + if not all(path.is_file() for path in all_paths): + return None + + if verify_integrity: + markdown_meta = manifest.get("markdown") + if not isinstance(markdown_meta, dict): + raise ValueError + expected_md_sha = markdown_meta.get("sha256") + expected_md_size = markdown_meta.get("size") + if not isinstance(expected_md_sha, str) or not isinstance(expected_md_size, int): + raise ValueError + md_sha, md_size = _file_sha256(markdown, cap=policy.max_markdown_chars * 4) + if md_sha != expected_md_sha or md_size != expected_md_size: + raise ValueError + if md_size > policy.max_markdown_chars * 4: + raise ValueError + + assets_meta = manifest.get("assets") + if not isinstance(assets_meta, list): + raise ValueError + expected_by_path: dict[str, dict] = {} + total_asset_bytes = 0 + for entry in assets_meta: + if not isinstance(entry, dict): + raise ValueError + relative = entry.get("path") + kind = entry.get("kind") + sha = entry.get("sha256") + size = entry.get("size") + filename = entry.get("filename") + if ( + not isinstance(relative, str) + or not isinstance(kind, str) + or not isinstance(sha, str) + or not isinstance(size, int) + or not isinstance(filename, str) + or size < 0 + or size > policy.max_asset_bytes + ): + raise ValueError + if PurePosixPath(relative).name != filename: + raise ValueError + if kind == "image": + extension = Path(filename).suffix.lower() + if extension not in _SEMANTIC_IMAGE_EXTENSIONS: + raise ValueError + elif kind not in {"media", "attachment"}: + raise ValueError + prior = expected_by_path.get(relative) + if prior is not None: + if prior.get("sha256") != sha or prior.get("size") != size or prior.get("kind") != kind: + raise ValueError + continue + expected_by_path[relative] = entry + total_asset_bytes += size + if total_asset_bytes > policy.max_extracted_asset_bytes: + raise ValueError + if len(expected_by_path) > policy.max_assets: + raise ValueError + + listed = { + "image": [path.relative_to(bundle).as_posix() for path in images], + "media": [path.relative_to(bundle).as_posix() for path in media], + "attachment": [path.relative_to(bundle).as_posix() for path in attachments], + } + seen: set[str] = set() + for kind_name, relatives in listed.items(): + for relative in relatives: + if relative in seen: + raise ValueError + seen.add(relative) + entry = expected_by_path.get(relative) + if entry is None or entry.get("kind") != kind_name: + raise ValueError + path = _safe_bundle_relative(bundle, relative) + actual_sha, actual_size = _file_sha256(path, cap=policy.max_asset_bytes) + if actual_sha != entry["sha256"] or actual_size != entry["size"]: + raise ValueError + if set(expected_by_path) != seen: + raise ValueError + + return PresentationArtifacts( + markdown, + manifest_path, + images, + media, + attachments, + tuple(warnings), + ) + except (ValueError, OSError, RuntimeError): + return None + + +def _load_cached( + bundle: Path, + source_sha256: str, + source_identity: str, + limits: PresentationLimits, +) -> PresentationArtifacts | None: + """Return a trusted cached bundle, or None to force regeneration.""" + manifest_path = bundle / "manifest.json" + try: + if not manifest_path.is_file(): + return None + manifest_size = manifest_path.stat().st_size + if manifest_size <= 0 or manifest_size > limits.max_manifest_bytes: + return None + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError, UnicodeError): + return None + if not isinstance(manifest, dict): + return None + if manifest.get("schema_version") != MANIFEST_SCHEMA_VERSION: + return None + if manifest.get("converter_version") != CONVERTER_VERSION: + return None + if manifest.get("source_sha256") != source_sha256: + return None + if manifest.get("source_path") != source_identity: + return None + if manifest.get("limits") != limits.as_policy(): + return None + if manifest.get("locally_owned") is not True: + return None + return _artifact_paths(bundle, manifest, limits=limits, verify_integrity=True) + + +def _write_bundle( + bundle: Path, + extraction: _Extraction, + source_sha256: str, + source_name: str, + source_identity: str, + limits: PresentationLimits, +) -> PresentationArtifacts: + bundle.parent.mkdir(parents=True, exist_ok=True) + stage = bundle.parent / f".{bundle.name}.tmp-{uuid.uuid4().hex}" + backup = bundle.parent / f".{bundle.name}.old-{uuid.uuid4().hex}" + try: + assets_dir = stage / "assets" + assets_dir.mkdir(parents=True) + markdown_bytes = extraction.markdown.encode("utf-8") + if len(markdown_bytes) > limits.max_markdown_chars * 4: + raise PresentationError("PPTX markdown exceeds the configured safety limit") + (stage / "presentation.md").write_bytes(markdown_bytes) + image_paths: list[str] = [] + media_paths: list[str] = [] + attachment_paths: list[str] = [] + asset_manifest: list[dict] = [] + total_asset_bytes = 0 + written_files: set[str] = set() + for asset in extraction.assets: + relative = f"assets/{asset.filename}" + target = stage / "assets" / asset.filename + if relative not in written_files: + if asset.size > limits.max_asset_bytes: + raise PresentationError( + f"PPTX asset {asset.filename!r} exceeds the per-asset byte limit" + ) + total_asset_bytes += asset.size + if total_asset_bytes > limits.max_extracted_asset_bytes: + raise PresentationError( + "PPTX extracted assets exceed the configured total byte limit" + ) + target.write_bytes(asset.payload) + written_files.add(relative) + if asset.kind == "image": + image_paths.append(relative) + elif asset.kind == "media": + media_paths.append(relative) + else: + attachment_paths.append(relative) + metadata = asdict(asset) + metadata.pop("payload") + metadata["path"] = relative + metadata["sha256"] = asset.sha256 + metadata["size"] = asset.size + asset_manifest.append(metadata) + unique_asset_paths = len(written_files) + if unique_asset_paths > limits.max_assets: + raise PresentationError( + f"PPTX exceeds the extracted asset limit of {limits.max_assets}" + ) + manifest = { + "schema_version": MANIFEST_SCHEMA_VERSION, + "converter_version": CONVERTER_VERSION, + "locally_owned": True, + "source_name": source_name, + "source_path": source_identity, + "source_sha256": source_sha256, + "limits": limits.as_policy(), + "presentation_part": extraction.presentation_part, + "slides": extraction.slides, + "markdown": { + "path": "presentation.md", + "sha256": hashlib.sha256(markdown_bytes).hexdigest(), + "size": len(markdown_bytes), + }, + "assets": asset_manifest, + "images": image_paths, + "media": media_paths, + "attachments": attachment_paths, + "warnings": extraction.warnings, + } + manifest_text = json.dumps(manifest, ensure_ascii=False, indent=2, sort_keys=True) + "\n" + if len(manifest_text.encode("utf-8")) > limits.max_manifest_bytes: + raise PresentationError("PPTX manifest exceeds the configured safety limit") + (stage / "manifest.json").write_text(manifest_text, encoding="utf-8") + if bundle.exists(): + os.replace(bundle, backup) + os.replace(stage, bundle) + if backup.exists(): + shutil.rmtree(backup, ignore_errors=True) + except Exception: + shutil.rmtree(stage, ignore_errors=True) + if backup.exists() and not bundle.exists(): + os.replace(backup, bundle) + raise + artifacts = _artifact_paths(bundle, manifest, limits=limits, verify_integrity=True) + if artifacts is None: + raise PresentationError("PPTX artifact bundle verification failed after writing") + return artifacts + + +def convert_presentation_file( + path: Path, + out_dir: Path, + *, + root: Path | None = None, + limits: PresentationLimits | None = None, +) -> PresentationArtifacts: + """Convert ``path`` into a structured, content-versioned artifact bundle. + + External relationships are recorded but never fetched. Embedded content is + copied as inert evidence only; no macros, OLE objects, media, or attachments + are executed. The source is read once into a bounded immutable byte snapshot, + preventing source-swap inconsistencies between fingerprinting and parsing. + Cached bundles are revalidated against the active limits policy and artifact + integrity before reuse. + """ + limits = limits or PresentationLimits() + raw = _read_source(path, limits.max_raw_bytes) + source_sha256 = hashlib.sha256(raw).hexdigest() + source_identity = _source_identity(path, root) + bundle = out_dir / _bundle_name(path, root) + cached = _load_cached(bundle, source_sha256, source_identity, limits) + if cached is not None: + return cached + package: _Package | None = None + try: + try: + package = _Package(raw, limits) + except (zipfile.BadZipFile, zipfile.LargeZipFile, RecursionError) as exc: + raise PresentationError(f"not a readable PPTX ZIP package: {exc}") from exc + if root is not None: + try: + bundle_reference = bundle.resolve().relative_to(root.resolve()).as_posix() + except (ValueError, OSError): + bundle_reference = str(bundle.resolve()) + else: + bundle_reference = str(bundle.resolve()) + extraction = _extract_presentation( + package, + path.name, + source_sha256, + limits, + bundle_reference, + ) + finally: + if package is not None: + package.close() + return _write_bundle( + bundle, + extraction, + source_sha256, + path.name, + source_identity, + limits, + ) + + +def cleanup_orphaned_presentation_bundles( + output_dir: Path, + *, + root: Path, +) -> tuple[Path, ...]: + """Remove converter-owned bundles whose recorded source no longer exists. + + A malformed or tampered manifest fails closed and is left untouched. Only + ``*_pptx`` directories and source paths contained by ``root`` are eligible. + """ + if not output_dir.is_dir(): + return () + try: + output_resolved = output_dir.resolve() + root_resolved = root.resolve() + except (OSError, RuntimeError): + return () + + removed: list[Path] = [] + for bundle in output_dir.iterdir(): + if not bundle.is_dir() or not bundle.name.endswith("_pptx"): + continue + try: + bundle.resolve().relative_to(output_resolved) + manifest_path = bundle / "manifest.json" + if not manifest_path.is_file() or manifest_path.stat().st_size > 1_000_000: + continue + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + source_identity = manifest.get("source_path") + if not isinstance(source_identity, str) or not source_identity: + continue + source_relative = PurePosixPath(source_identity) + if source_relative.is_absolute() or ".." in source_relative.parts: + continue + source_path = (root_resolved / Path(*source_relative.parts)).resolve() + source_path.relative_to(root_resolved) + if source_path.exists(): + continue + shutil.rmtree(bundle) + removed.append(bundle) + except (OSError, RuntimeError, ValueError, json.JSONDecodeError): + continue + return tuple(removed) diff --git a/graphify/skill.md b/graphify/skill.md index d98865cc8..985651879 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -156,7 +156,7 @@ This step has two parts: **structural extraction** (deterministic, free) and **s **Before semantic extraction:** check whether `GEMINI_API_KEY` or `GOOGLE_API_KEY` is set. If neither is set, print this one-liner to the user: > Tip: set `GEMINI_API_KEY` or `GOOGLE_API_KEY` to use Gemini for semantic extraction (`pip install 'graphifyy[gemini]'`). -Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3-flash-preview`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it. +Print it once, then continue — do not wait for the user to supply a key. If `GEMINI_API_KEY` or `GOOGLE_API_KEY` IS set, use `graphify.llm.extract_corpus_parallel(files, backend="gemini")` for semantic extraction instead of dispatching subagents. The default Gemini model is `gemini-3.5-flash`; set `GRAPHIFY_GEMINI_MODEL` or pass `--model` in headless CLI flows to override it. > **No other API keys are read.** When `GEMINI_API_KEY`/`GOOGLE_API_KEY` are unset, semantic extraction falls to the host agent itself — the running session is the LLM. On a host that dispatches subagents (e.g. Claude Code), dispatch them as written in Part B. On a host that runs the CLI directly in a terminal and cannot dispatch subagents, do not stall: a code-only corpus has no semantic work, so write the empty semantic file (Part B "Fast path") and continue to Part C; for a corpus with docs/papers/images, either set a Gemini key or extract those inline yourself, but in no case prompt for `ANTHROPIC_API_KEY` — that prompt is a misread of this skill. diff --git a/graphify/skills/agents/references/transcribe.md b/graphify/skills/agents/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/agents/references/transcribe.md +++ b/graphify/skills/agents/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/amp/references/transcribe.md b/graphify/skills/amp/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/amp/references/transcribe.md +++ b/graphify/skills/amp/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/claude/references/transcribe.md b/graphify/skills/claude/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/claude/references/transcribe.md +++ b/graphify/skills/claude/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/claw/references/transcribe.md b/graphify/skills/claw/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/claw/references/transcribe.md +++ b/graphify/skills/claw/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/codex/references/transcribe.md b/graphify/skills/codex/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/codex/references/transcribe.md +++ b/graphify/skills/codex/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/copilot/references/transcribe.md b/graphify/skills/copilot/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/copilot/references/transcribe.md +++ b/graphify/skills/copilot/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/droid/references/transcribe.md b/graphify/skills/droid/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/droid/references/transcribe.md +++ b/graphify/skills/droid/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/kilo/references/transcribe.md b/graphify/skills/kilo/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/kilo/references/transcribe.md +++ b/graphify/skills/kilo/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/kiro/references/transcribe.md b/graphify/skills/kiro/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/kiro/references/transcribe.md +++ b/graphify/skills/kiro/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/opencode/references/transcribe.md b/graphify/skills/opencode/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/opencode/references/transcribe.md +++ b/graphify/skills/opencode/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/pi/references/transcribe.md b/graphify/skills/pi/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/pi/references/transcribe.md +++ b/graphify/skills/pi/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/trae/references/transcribe.md b/graphify/skills/trae/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/trae/references/transcribe.md +++ b/graphify/skills/trae/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/vscode/references/transcribe.md b/graphify/skills/vscode/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/vscode/references/transcribe.md +++ b/graphify/skills/vscode/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/skills/windows/references/transcribe.md b/graphify/skills/windows/references/transcribe.md index b967f8379..691aac6d2 100644 --- a/graphify/skills/windows/references/transcribe.md +++ b/graphify/skills/windows/references/transcribe.md @@ -23,6 +23,8 @@ Read the top god node labels from detect output or analysis, then compose a shor **Step 2 - Transcribe:** +Transcripts are content/model/prompt-addressed under `graphify-out/transcripts/`. Changing media bytes, `--whisper-model`, or the Whisper prompt automatically selects a new path. Pass `force=True` only when the user explicitly asked to re-transcribe despite an identical cache key. Requires Python 3.11+ with `graphifyy[video]` (faster-whisper). + ```bash export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported) export GRAPHIFY_WHISPER_PROMPT="" @@ -34,12 +36,28 @@ from graphify.transcribe import transcribe_all detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\")) video_files = detect.get('files', {}).get('video', []) prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.') - -transcript_paths = transcribe_all(video_files, initial_prompt=prompt) -# Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper -# print progress to stdout, which would otherwise corrupt the JSON file (#1392). -Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\") +force = os.environ.get('GRAPHIFY_FORCE', '').lower() in {'1', 'true', 'yes'} +out_dir = Path('graphify-out') / 'transcripts' + +transcript_paths = transcribe_all( + video_files, + output_dir=out_dir, + initial_prompt=prompt, + force=force, +) +# Persist Path objects as strings. Write JSON from Python (NOT a shell '>' redirect): +# transcribe_all/Whisper print progress to stdout, which would otherwise corrupt +# the JSON file (#1392). +Path('graphify-out/.graphify_transcripts.json').write_text( + json.dumps([str(p) for p in transcript_paths], ensure_ascii=False), + encoding=\"utf-8\", +) print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr) +if len(transcript_paths) < len(video_files): + print( + f'warning: {len(video_files) - len(transcript_paths)} media file(s) failed transcription', + file=sys.stderr, + ) " ``` @@ -47,6 +65,6 @@ After transcription: - Read the transcript paths from `graphify-out/.graphify_transcripts.json` - Add them to the docs list before dispatching semantic subagents in Step 3B - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs` -- If transcription fails for a file, print a warning and continue with the rest +- If transcription fails for a file, print a warning and continue with the rest — do not treat the source media as successfully processed until its transcript is present -**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. +**Whisper model:** Default is `base`. If the user passed `--whisper-model `, `export GRAPHIFY_WHISPER_MODEL=` (it must be exported, not just assigned) before running the command above. If the user passed `--force`, also `export GRAPHIFY_FORCE=1`. diff --git a/graphify/transcribe.py b/graphify/transcribe.py index 8e45bcaa2..296f58c4f 100644 --- a/graphify/transcribe.py +++ b/graphify/transcribe.py @@ -2,7 +2,12 @@ # Converts video/audio files to text transcripts for graph extraction from __future__ import annotations +import hashlib +import json import os +import re +import tempfile +import unicodedata from pathlib import Path from graphify.paths import out_path as _out_path @@ -14,6 +19,12 @@ _DEFAULT_MODEL = "base" _TRANSCRIPTS_DIR = str(_out_path("transcripts")) _FALLBACK_PROMPT = "Use proper punctuation and paragraph breaks." +_TRANSCRIPTION_OPTIONS = { + "beam_size": 5, + "compute_type": "int8", + "device": "cpu", + "schema": 1, +} def _model_name() -> str: @@ -26,8 +37,9 @@ def _get_whisper(): return WhisperModel except ImportError as exc: raise ImportError( - "Video transcription requires faster-whisper. " - "Run: pip install 'graphifyy[video]'" + "Video transcription requires faster-whisper " + "(Python 3.11+; graphifyy[video] installs it only on 3.11+). " + "Run: pip install 'graphifyy[video]' on Python 3.11 or newer" ) from exc @@ -100,13 +112,13 @@ def build_whisper_prompt(god_nodes: list[dict]) -> str: domain hint from these labels and passes it via GRAPHIFY_WHISPER_PROMPT or as initial_prompt — no separate API call needed here. """ - if not god_nodes: - return _FALLBACK_PROMPT - override = os.environ.get("GRAPHIFY_WHISPER_PROMPT") if override: return override + if not god_nodes: + return _FALLBACK_PROMPT + labels = [n.get("label", "") for n in god_nodes[:10] if n.get("label")] if not labels: return _FALLBACK_PROMPT @@ -115,6 +127,55 @@ def build_whisper_prompt(god_nodes: list[dict]) -> str: return f"Technical discussion about {topics}. Use proper punctuation and paragraph breaks." +def _effective_prompt(initial_prompt: str | None) -> str: + return initial_prompt or os.environ.get("GRAPHIFY_WHISPER_PROMPT") or _FALLBACK_PROMPT + + +def _sha256_file(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def transcript_path_for( + video_path: Path | str, + output_dir: Path | None = None, + initial_prompt: str | None = None, + *, + media_path: Path | None = None, +) -> Path: + """Return the content/config-addressed transcript path for a media source.""" + out_dir = Path(output_dir) if output_dir else Path(_TRANSCRIPTS_DIR) + source = str(video_path) + if is_url(source): + source_identity = source + content_sha256 = _sha256_file(media_path) if media_path is not None else hashlib.sha256( + source.encode("utf-8") + ).hexdigest() + stem_source = media_path.stem if media_path is not None else "remote-media" + else: + local = Path(video_path) + source_identity = local.resolve().as_posix() + content_sha256 = _sha256_file(local) + stem_source = local.stem + cache_key = { + "content_sha256": content_sha256, + "model": _model_name(), + "options": _TRANSCRIPTION_OPTIONS, + "prompt": _effective_prompt(initial_prompt), + "source": unicodedata.normalize("NFC", source_identity), + } + digest = hashlib.sha256( + json.dumps(cache_key, ensure_ascii=False, sort_keys=True).encode("utf-8") + ).hexdigest()[:16] + stem = re.sub( + r"[^A-Za-z0-9._-]+", "_", unicodedata.normalize("NFC", stem_source) + ).strip("._-") + return out_dir / f"{(stem[:80] or 'media')}-{digest}.txt" + + def transcribe( video_path: Path | str, output_dir: Path | None = None, @@ -138,36 +199,59 @@ def transcribe( else: audio_path = Path(video_path) - transcript_path = out_dir / (audio_path.stem + ".txt") + prompt = _effective_prompt(initial_prompt) + transcript_path = transcript_path_for( + video_path, + out_dir, + initial_prompt=prompt, + media_path=audio_path, + ) if transcript_path.exists() and not force: return transcript_path WhisperModel = _get_whisper() model_name = _model_name() - prompt = initial_prompt or _FALLBACK_PROMPT print(f" transcribing {audio_path.name} (model={model_name}) ...", flush=True) - model = WhisperModel(model_name, device="cpu", compute_type="int8") + model = WhisperModel( + model_name, + device=_TRANSCRIPTION_OPTIONS["device"], + compute_type=_TRANSCRIPTION_OPTIONS["compute_type"], + ) segments, info = model.transcribe( str(audio_path), - beam_size=5, + beam_size=_TRANSCRIPTION_OPTIONS["beam_size"], initial_prompt=prompt, ) lines = [segment.text.strip() for segment in segments if segment.text.strip()] transcript = "\n".join(lines) - transcript_path.write_text(transcript, encoding="utf-8") + with tempfile.NamedTemporaryFile( + "w", + encoding="utf-8", + dir=out_dir, + prefix=f".{transcript_path.name}.", + suffix=".tmp", + delete=False, + ) as handle: + handle.write(transcript) + temporary = Path(handle.name) + try: + os.replace(temporary, transcript_path) + finally: + temporary.unlink(missing_ok=True) lang = info.language if hasattr(info, "language") else "unknown" print(f" transcript saved -> {transcript_path} (lang={lang}, {len(lines)} segments)") return transcript_path def transcribe_all( - video_files: list[str], + video_files: list[Path | str], output_dir: Path | None = None, initial_prompt: str | None = None, -) -> list[str]: + force: bool = False, +) -> list[Path]: """Transcribe a list of video/audio files or URLs, return paths to transcript .txt files. Already-transcribed files are returned from cache instantly. @@ -179,8 +263,8 @@ def transcribe_all( transcript_paths = [] for vf in video_files: try: - t = transcribe(vf, output_dir, initial_prompt=initial_prompt) - transcript_paths.append(str(t)) + t = transcribe(vf, output_dir, initial_prompt=initial_prompt, force=force) + transcript_paths.append(t) except Exception as exc: print(f" warning: could not transcribe {vf}: {exc}") return transcript_paths diff --git a/graphify/watch.py b/graphify/watch.py index 1ef1ebd4d..a24bea589 100644 --- a/graphify/watch.py +++ b/graphify/watch.py @@ -257,11 +257,20 @@ def _git_head() -> str | None: DOC_EXTENSIONS, PAPER_EXTENSIONS, IMAGE_EXTENSIONS, + OFFICE_EXTENSIONS, + VIDEO_EXTENSIONS, _load_graphifyignore, _is_ignored, ) -_WATCHED_EXTENSIONS = CODE_EXTENSIONS | DOC_EXTENSIONS | PAPER_EXTENSIONS | IMAGE_EXTENSIONS +_WATCHED_EXTENSIONS = ( + CODE_EXTENSIONS + | DOC_EXTENSIONS + | PAPER_EXTENSIONS + | IMAGE_EXTENSIONS + | OFFICE_EXTENSIONS + | VIDEO_EXTENSIONS +) _CODE_EXTENSIONS = CODE_EXTENSIONS diff --git a/pyproject.toml b/pyproject.toml index 2d2abafb6..1ee512509 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,11 +10,18 @@ readme = "README.md" license = "Apache-2.0" license-files = ["LICENSE", "LICENSE-MIT", "NOTICE"] keywords = ["claude", "claude-code", "codex", "opencode", "kilo", "cursor", "gemini", "aider", "kiro", "pi", "devin", "knowledge-graph", "rag", "graphrag", "obsidian", "community-detection", "tree-sitter", "leiden", "llm"] -requires-python = ">=3.10" +requires-python = ">=3.14" +classifiers = [ + "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.14", + "Operating System :: OS Independent", + "Intended Audience :: Developers", +] dependencies = [ "networkx>=3.4", "numpy>=1.21", "rapidfuzz>=3.0", + "defusedxml>=0.7.1", "tree-sitter>=0.23.0,<0.26", "tree-sitter-python>=0.23,<0.26", "tree-sitter-javascript>=0.23,<0.26", @@ -57,12 +64,12 @@ neo4j = ["neo4j"] falkordb = ["falkordb"] pdf = ["pypdf>=6.12.0", "markdownify"] watch = ["watchdog"] -svg = ["matplotlib", "numpy>=2.0; python_version >= '3.13'"] -leiden = ["graspologic; python_version < '3.13'"] +svg = ["matplotlib", "numpy>=2.0"] +leiden = ["graspologic"] office = ["python-docx", "openpyxl"] google = ["openpyxl"] postgres = ["psycopg[binary]"] -video = ["faster-whisper; python_version >= '3.11'", "yt-dlp>=2026.6.9"] +video = ["faster-whisper", "yt-dlp>=2026.6.9"] kimi = ["openai", "tiktoken"] ollama = ["openai"] bedrock = ["boto3"] @@ -81,7 +88,7 @@ pascal = ["tree-sitter-pascal"] # avoids breaking the default `uv tool install graphifyy` for everyone (#1104). dm = ["tree-sitter-dm"] terraform = ["tree-sitter-hcl"] -all = ["mcp", "starlette>=1.3.1", "neo4j", "falkordb", "pypdf>=6.12.0", "markdownify", "watchdog", "graspologic; python_version < '3.13'", "python-docx", "openpyxl", "faster-whisper; python_version >= '3.11'", "yt-dlp>=2026.6.9", "matplotlib", "numpy>=2.0; python_version >= '3.13'", "openai", "tiktoken", "boto3", "anthropic", "tree-sitter-sql", "jieba", "tree-sitter-dm", "tree-sitter-hcl", "tree-sitter-pascal"] +all = ["mcp", "starlette>=1.3.1", "neo4j", "falkordb", "pypdf>=6.12.0", "markdownify", "watchdog", "graspologic", "python-docx", "openpyxl", "faster-whisper", "yt-dlp>=2026.6.9", "matplotlib", "numpy>=2.0", "openai", "tiktoken", "boto3", "anthropic", "tree-sitter-sql", "jieba", "tree-sitter-dm", "tree-sitter-hcl", "tree-sitter-pascal"] [project.scripts] graphify = "graphify.__main__:main" @@ -102,7 +109,7 @@ dev = [ "ruff>=0.15.13", "setuptools>=82.0.1", "wheel>=0.47.0", - "tomli>=2.0 ; python_version < '3.11'", + "tree-sitter-hcl>=1.2.0", ] @@ -136,7 +143,7 @@ skips = ["B404"] [tool.ruff] line-length = 100 -target-version = "py310" +target-version = "py314" [tool.ruff.lint] # Keep the committed baseline conservative until upstream adopts a broader lint policy. @@ -144,5 +151,5 @@ select = ["E9", "F63", "F7", "F82"] [tool.pyright] include = ["graphify", "tests"] -pythonVersion = "3.10" +pythonVersion = "3.14" typeCheckingMode = "basic" diff --git a/tests/test_epub.py b/tests/test_epub.py new file mode 100644 index 000000000..34ed0c21b --- /dev/null +++ b/tests/test_epub.py @@ -0,0 +1,177 @@ +import json +"""Tests for the EPUB semantic ingestion module. + +Includes synthetic EPUB fixtures to verify full semantic extraction: +- chapter order from spine +- text + titles +- media (images/audio/video) extraction and routing +- manifest structure +- safety limits +- error handling +""" + +import tempfile +import zipfile +from pathlib import Path + +import pytest + +from graphify.epub import ( + EpubError, + EpubLimits, + convert_epub_file, + cleanup_orphaned_epub_bundles, +) + + +def _make_minimal_epub( + tmp: Path, + *, + title: str = "Test Book", + chapters: list[tuple[str, str, list[str]]] | None = None, +) -> Path: + """Create a minimal valid EPUB in tmp and return the .epub path.""" + if chapters is None: + chapters = [ + ("chap1", "Chapter One", "

Hello

First chapter with

"), + ("chap2", "Chapter Two", "

World

Second with audio

"), + ] + + epub_path = tmp / "test.epub" + with zipfile.ZipFile(epub_path, "w") as z: + # mimetype first, uncompressed + z.writestr("mimetype", "application/epub+zip", compress_type=zipfile.ZIP_STORED) + + z.writestr( + "META-INF/container.xml", + """ + + + + +""", + ) + + # Build manifest items + manifest_items = [''] + spine_items = [] + for cid, ctitle, body in chapters: + manifest_items.append(f'') + spine_items.append(f'') + + # content.opf with metadata + opf = f""" + + + {title} + Test Author + en + + + {''.join(manifest_items)} + + + {''.join(spine_items)} + +""" + z.writestr("content.opf", opf) + + # chapters + for cid, ctitle, body in chapters: + xhtml = f""" + +{ctitle} +{body} +""" + z.writestr(f"{cid}.xhtml", xhtml) + + # fake media + z.writestr("img1.png", b"\x89PNG\r\n\x1a\n" + b"\x00" * 20) + z.writestr("end.png", b"\x89PNG\r\n\x1a\n" + b"\x00" * 20) + z.writestr("sound.mp3", b"ID3" + b"\x00" * 100) + + return epub_path + + +def test_epub_limits_and_policy(): + limits = EpubLimits() + assert limits.max_chapters >= 1 + policy = limits.as_policy() + assert isinstance(policy, dict) + assert "max_raw_bytes" in policy + + +def test_epub_convert_nonexistent_raises(): + with pytest.raises((EpubError, FileNotFoundError, OSError)): + convert_epub_file(Path("/non/existent.epub"), Path("/tmp")) + + +def test_epub_basic_extraction(tmp_path): + epub_path = _make_minimal_epub(tmp_path) + out_dir = tmp_path / "out" + art = convert_epub_file(epub_path, out_dir) + + assert art.markdown_path.exists() + md = art.markdown_path.read_text() + assert "Hello" in md + assert "World" in md + assert "Chapter One" in md or "Chapter Two" in md + + # Media + assert len(art.images) >= 1 + assert any("img1" in str(p) for p in art.images) + assert len(art.media) >= 1 + assert any("sound" in str(p) for p in art.media) + + # Manifest + assert art.manifest_path.exists() + manifest = json.loads(art.manifest_path.read_text()) + assert manifest["metadata"]["title"] == "Test Book" + assert len(manifest["chapters"]) == 2 + assert manifest["chapters"][0]["title"] + assert any(a["kind"] == "image" for a in manifest["assets"]) + + +def test_epub_chapter_order_and_titles(tmp_path): + chapters = [ + ("c1", "Intro", "

Introduction

intro text

"), + ("c2", "Middle", "

Middle Chapter

"), + ("c3", "End", "

Conclusion

"), + ] + epub_path = _make_minimal_epub(tmp_path, chapters=chapters) + art = convert_epub_file(epub_path, tmp_path / "out2") + + manifest = json.loads(art.manifest_path.read_text()) + titles = [c["title"] for c in manifest["chapters"]] + assert titles == ["Intro", "Middle", "End"] + assert len(art.images) == 1 + + +def test_epub_media_routing(tmp_path): + epub_path = _make_minimal_epub(tmp_path) + art = convert_epub_file(epub_path, tmp_path / "out3") + + # Images should be images, mp3 as media (video/audio bucket) + assert all("img" in p.name or p.suffix in {".png"} for p in art.images) + assert any(p.suffix in {".mp3"} for p in art.media) + + +def test_epub_error_on_bad_zip(tmp_path): + bad = tmp_path / "bad.epub" + bad.write_bytes(b"not a zip at all") + with pytest.raises(EpubError): + convert_epub_file(bad, tmp_path / "out") + + +def test_cleanup_function_does_not_crash(tmp_path): + # Just ensure it runs without error on empty dir + cleanup_orphaned_epub_bundles(tmp_path / "converted", root=tmp_path) + assert True + + +def test_epub_module_symbols(): + import graphify.epub as epub_mod + assert hasattr(epub_mod, "convert_epub_file") + assert hasattr(epub_mod, "EpubArtifacts") + assert hasattr(epub_mod, "EpubLimits") + assert hasattr(epub_mod, "cleanup_orphaned_epub_bundles") diff --git a/tests/test_extract_cli.py b/tests/test_extract_cli.py index b1c369a5e..700fd7690 100644 --- a/tests/test_extract_cli.py +++ b/tests/test_extract_cli.py @@ -19,6 +19,242 @@ def _make_corpus(tmp_path): return tmp_path +def test_headless_extract_transcribes_detected_media_before_semantic_dispatch( + monkeypatch, tmp_path +): + """Audio/video must become transcript documents in the headless CLI too.""" + media = tmp_path / "deck-slide-0001-audio-01-contenthash.mp3" + media.write_bytes(b"ID3 fake test audio") + out_dir = tmp_path / "out" + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake-key") + transcribe_call = {} + + def _fake_transcribe_all( + video_files, output_dir=None, initial_prompt=None, force=False + ): + from graphify.transcribe import transcript_path_for + + transcribe_call["video_files"] = list(video_files) + transcribe_call["output_dir"] = output_dir + assert output_dir is not None + output_dir.mkdir(parents=True, exist_ok=True) + transcript = transcript_path_for(media, output_dir, initial_prompt) + transcript.write_text("The experiment supports the causal claim.", encoding="utf-8") + return [transcript] + + dispatched = [] + + def _capture_semantic(paths, **kwargs): + dispatched.extend(str(path) for path in paths) + on_chunk = kwargs.get("on_chunk_done") + if on_chunk: + on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []}) + return { + "nodes": [], + "edges": [], + "hyperedges": [], + "input_tokens": 10, + "output_tokens": 5, + } + + monkeypatch.setattr("graphify.transcribe.transcribe_all", _fake_transcribe_all) + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _capture_semantic) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + monkeypatch.setattr( + mainmod.sys, + "argv", + [ + "graphify", + "extract", + str(tmp_path), + "--backend", + "claude", + "--out", + str(out_dir), + "--whisper-model", + "tiny", + "--no-cluster", + ], + ) + + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0) + + assert transcribe_call["video_files"] == [media] + assert transcribe_call["output_dir"] == out_dir / "graphify-out" / "transcripts" + assert os.environ["GRAPHIFY_WHISPER_MODEL"] == "tiny" + assert not any(path.endswith(".mp3") for path in dispatched) + assert any(media.stem in path and path.endswith(".txt") for path in dispatched) + + +@pytest.mark.parametrize("no_cluster", [False, True]) +def test_incremental_extract_preserves_unchanged_transcript_nodes( + monkeypatch, tmp_path, no_cluster +): + """Generated transcript nodes remain live when their media source is unchanged.""" + import json + + media = tmp_path / "deck-slide-0001-video-01-contenthash.mp4" + media.write_bytes(b"\x00\x00\x00\x18ftypmp42 fake video") + out_dir = tmp_path / "out" + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake-key") + transcribe_calls = [] + + def _fake_transcribe_all( + video_files, output_dir=None, initial_prompt=None, force=False + ): + from graphify.transcribe import transcript_path_for + + transcribe_calls.append(list(video_files)) + assert output_dir is not None + output_dir.mkdir(parents=True, exist_ok=True) + transcript = transcript_path_for(media, output_dir, initial_prompt) + transcript.write_text("A durable video-derived claim.", encoding="utf-8") + return [transcript] + + semantic_calls = [] + + def _semantic(paths, **kwargs): + paths = list(paths) + semantic_calls.append([str(path) for path in paths]) + transcript = next(path for path in paths if str(path).endswith(".txt")) + on_chunk = kwargs.get("on_chunk_done") + result = { + "nodes": [ + { + "id": "video-derived-claim", + "label": "Video-derived claim", + "type": "Claim", + "description": "A durable video-derived claim.", + "source_file": str(transcript), + "source_location": "transcript", + } + ], + "edges": [], + "hyperedges": [], + "input_tokens": 10, + "output_tokens": 5, + } + if on_chunk: + on_chunk(0, 1, result) + return result + + monkeypatch.setattr("graphify.transcribe.transcribe_all", _fake_transcribe_all) + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _semantic) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + argv = [ + "graphify", + "extract", + str(tmp_path), + "--backend", + "claude", + "--out", + str(out_dir), + ] + if no_cluster: + argv.append("--no-cluster") + + monkeypatch.setattr(mainmod.sys, "argv", argv) + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0) + + monkeypatch.setattr(mainmod.sys, "argv", argv) + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0) + + graph = json.loads((out_dir / "graphify-out" / "graph.json").read_text()) + from graphify.detect import detect_incremental + + incremental = detect_incremental( + tmp_path, + manifest_path=str(out_dir / "graphify-out" / "manifest.json"), + transcript_root=out_dir / "graphify-out" / "transcripts", + ) + assert len(transcribe_calls) == 1 + assert len(semantic_calls) == 1 + assert not any(path.endswith(".txt") for path in incremental["excluded_files"]) + assert any(node["id"] == "video-derived-claim" for node in graph["nodes"]) + + +def test_failed_media_transcription_is_retried_incrementally(monkeypatch, tmp_path): + media = tmp_path / "retry.mp4" + media.write_bytes(b"\x00\x00\x00\x18ftypmp42 retry video") + out_dir = tmp_path / "out" + monkeypatch.setenv("ANTHROPIC_API_KEY", "fake-key") + transcribe_calls = [] + + def _transcribe(video_files, output_dir=None, initial_prompt=None, force=False): + from graphify.transcribe import transcript_path_for + + transcribe_calls.append(list(video_files)) + if len(transcribe_calls) == 1: + return [] + assert output_dir is not None + output_dir.mkdir(parents=True, exist_ok=True) + transcript = transcript_path_for(media, output_dir, initial_prompt) + transcript.write_text("Recovered transcript.", encoding="utf-8") + return [transcript] + + semantic_calls = [] + + def _semantic(paths, **kwargs): + paths = list(paths) + semantic_calls.append(paths) + transcript = next(path for path in paths if str(path).endswith(".txt")) + result = { + "nodes": [ + { + "id": "recovered-transcript", + "label": "Recovered transcript", + "type": "Claim", + "source_file": str(transcript), + } + ], + "edges": [], + "hyperedges": [], + "input_tokens": 1, + "output_tokens": 1, + } + on_chunk = kwargs.get("on_chunk_done") + if on_chunk: + on_chunk(0, 1, result) + return result + + monkeypatch.setattr("graphify.transcribe.transcribe_all", _transcribe) + monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _semantic) + monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None) + argv = [ + "graphify", + "extract", + str(tmp_path), + "--backend", + "claude", + "--out", + str(out_dir), + "--no-cluster", + ] + + for _ in range(2): + monkeypatch.setattr(mainmod.sys, "argv", argv) + try: + mainmod.main() + except SystemExit as exc: + assert exc.code in (None, 0) + + assert transcribe_calls == [[media], [media]] + assert len(semantic_calls) == 1 + graph = __import__("json").loads( + (out_dir / "graphify-out" / "graph.json").read_text(encoding="utf-8") + ) + assert any(node["id"] == "recovered-transcript" for node in graph["nodes"]) + + def test_extract_exits_nonzero_when_all_semantic_chunks_fail( monkeypatch, tmp_path, capsys ): diff --git a/tests/test_image_vision.py b/tests/test_image_vision.py index c73b1141e..67c911b00 100644 --- a/tests/test_image_vision.py +++ b/tests/test_image_vision.py @@ -180,7 +180,7 @@ def test_chunk_packing_caps_images_per_chunk(tmp_path): # Many images + a huge token budget must still cap images per chunk so a # single request never exceeds provider image limits. imgs = [] - for i in range(llm._MAX_IMAGES_PER_CHUNK * 2 + 3): + for i in range(llm._get_max_images_per_chunk() * 2 + 3): p = tmp_path / f"img{i:03d}.png" p.write_bytes(_PNG_BYTES) imgs.append(p) @@ -188,7 +188,7 @@ def test_chunk_packing_caps_images_per_chunk(tmp_path): assert len(chunks) >= 3 # would be 1 chunk without the cap for chunk in chunks: n_imgs = sum(1 for p in chunk if llm._is_vision_image(p)) - assert n_imgs <= llm._MAX_IMAGES_PER_CHUNK + assert n_imgs <= llm._get_max_images_per_chunk() # ── content builders ────────────────────────────────────────────────────────── diff --git a/tests/test_presentation.py b/tests/test_presentation.py new file mode 100644 index 000000000..3d9f01939 --- /dev/null +++ b/tests/test_presentation.py @@ -0,0 +1,818 @@ +from __future__ import annotations + +import base64 +import json +import os +import zipfile +from pathlib import Path + +import pytest + +from graphify import detect +from graphify.presentation import ( + PresentationError, + PresentationLimits, + cleanup_orphaned_presentation_bundles, + convert_presentation_file, +) +from graphify.watch import _WATCHED_EXTENSIONS + + +_PNG = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII=" +) + + +def _write_rich_pptx(path: Path) -> None: + content_types = """ + + + + + + + + + + + + + +""" + root_rels = """ + + +""" + core = """ + Semantic StrategyAda ResearcherKnowledge Graphs + 2026-07-01T00:00:00Z +""" + presentation = """ + + + +""" + presentation_rels = """ + + +""" + slide = """ + + + Q4 Research Strategy + Prioritize causal evidencePreserve uncertainty + Discover + Decide + + MethodConfidenceExperimentHigh + + + + + + + + + +""" + slide_rels = """ + + + + + + + + + + + +""" + notes = """Confidential speaker insight""" + chart = """Quarterly RevenueRevenueQ1Q2100140""" + diagram = """DiscoverDecide""" + comments = """Validate this claim""" + authors = """""" + + with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as zf: + parts = { + "[Content_Types].xml": content_types, + "_rels/.rels": root_rels, + "docProps/core.xml": core, + "ppt/presentation.xml": presentation, + "ppt/_rels/presentation.xml.rels": presentation_rels, + "ppt/slides/slide1.xml": slide, + "ppt/slides/_rels/slide1.xml.rels": slide_rels, + "ppt/notesSlides/notesSlide1.xml": notes, + "ppt/charts/chart1.xml": chart, + "ppt/diagrams/data1.xml": diagram, + "ppt/comments/comment1.xml": comments, + "ppt/commentAuthors.xml": authors, + "ppt/media/image1.png": _PNG, + "ppt/media/poster.png": _PNG, + "ppt/media/audio1.mp3": b"ID3\x04\x00\x00fake-audio", + "ppt/media/video1.mp4": b"\x00\x00\x00\x18ftypmp42fake-video", + "ppt/embeddings/data1.xlsx": b"PK\x03\x04fake-workbook", + } + for name, payload in parts.items(): + zf.writestr(name, payload) + + +def _replace_zip_parts(path: Path, replacements: dict[str, bytes | str]) -> None: + """Rewrite selected members without creating ambiguous duplicate ZIP names.""" + replacement_bytes = { + name: value.encode("utf-8") if isinstance(value, str) else value + for name, value in replacements.items() + } + with zipfile.ZipFile(path) as source: + members = { + info.filename: source.read(info) + for info in source.infolist() + if info.filename not in replacement_bytes + } + members.update(replacement_bytes) + temporary = path.with_suffix(".tmp") + with zipfile.ZipFile(temporary, "w", zipfile.ZIP_DEFLATED) as target: + for name, payload in members.items(): + target.writestr(name, payload) + temporary.replace(path) + + +def _inject_slide_part( + path: Path, + *, + shape_xml: str, + relationship_xml: str, + part_name: str, + payload: bytes, +) -> None: + with zipfile.ZipFile(path) as package: + slide = package.read("ppt/slides/slide1.xml").decode("utf-8") + relationships = package.read( + "ppt/slides/_rels/slide1.xml.rels" + ).decode("utf-8") + _replace_zip_parts( + path, + { + "ppt/slides/slide1.xml": slide.replace( + "", shape_xml + "" + ), + "ppt/slides/_rels/slide1.xml.rels": relationships.replace( + "", relationship_xml + "" + ), + part_name: payload, + }, + ) + + +def test_pptx_is_classified_and_watched(): + assert detect.classify_file(Path("research.pptx")) == detect.FileType.DOCUMENT + assert ".pptx" in _WATCHED_EXTENSIONS + + +def test_rich_presentation_extracts_semantics_and_assets(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + + artifacts = convert_presentation_file(source, tmp_path / "converted", root=tmp_path) + markdown = artifacts.markdown_path.read_text(encoding="utf-8") + + assert "Semantic Strategy" in markdown + assert "Ada Researcher" in markdown + assert "Slide 1" in markdown and "Q4 Research Strategy" in markdown + assert "hidden: true" in markdown + assert "Prioritize causal evidence" in markdown + assert "Preserve uncertainty" in markdown + assert "Method" in markdown and "Experiment" in markdown and "Confidence" in markdown + assert "Quarterly Revenue" in markdown and "Revenue" in markdown + assert "Q1" in markdown and "100" in markdown and "Q2" in markdown and "140" in markdown + assert "Discover -> Decide" in markdown + assert "Confidential speaker insight" in markdown + assert "Validate this claim" in markdown and "Grace Reviewer" in markdown + assert "https://example.org/evidence" in markdown + assert "https://example.org/demo.mp4" in markdown + assert "System architecture overview" in markdown + assert "Transition" in markdown and "fade" in markdown + + assert len(artifacts.images) == 2 + assert {p.suffix for p in artifacts.images} == {".png"} + assert len(artifacts.media) == 2 + assert {p.suffix for p in artifacts.media} == {".mp3", ".mp4"} + assert len(artifacts.attachments) == 1 + assert artifacts.attachments[0].suffix == ".xlsx" + assert all(p.is_file() for p in (*artifacts.images, *artifacts.media, *artifacts.attachments)) + assert all( + "slide-0001" in p.name + for p in (*artifacts.images, *artifacts.media, *artifacts.attachments) + ) + + manifest = json.loads(artifacts.manifest_path.read_text(encoding="utf-8")) + assert manifest["source_sha256"] + assert manifest["slides"][0]["source_part"] == "ppt/slides/slide1.xml" + assert manifest["slides"][0]["section"] == "Strategy" + + +def test_detect_registers_pptx_bundle_with_semantic_queues(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + + result = detect.detect(tmp_path, cache_root=tmp_path / "graphify-out") + + assert len(result["files"]["document"]) >= 1 + assert any(p.endswith(".md") and "pptx" in p for p in result["files"]["document"]) + assert len(result["files"]["image"]) == 2 + assert len(result["files"]["video"]) == 2 + assert any(p.endswith(".xlsx") or p.endswith(".md") for p in result["files"]["document"]) + + +def test_nested_pptx_attachment_never_enters_binary_text_queue(tmp_path: Path): + source = tmp_path / "outer.pptx" + _write_rich_pptx(source) + _inject_slide_part( + source, + shape_xml='', + relationship_xml=( + '' + ), + part_name="ppt/embeddings/nested.pptx", + payload=b"PK\x03\x04nested presentation bytes", + ) + + result = detect.detect(tmp_path, cache_root=tmp_path / "graphify-out") + + assert not any(path.endswith(".pptx") for path in result["files"]["document"]) + assert any( + "nested PPTX attachment retained" in warning + for warning in result["skipped_sensitive"] + ) + + +def test_bmp_attachment_never_enters_vision_or_binary_text_queue(tmp_path: Path): + source = tmp_path / "bitmap.pptx" + _write_rich_pptx(source) + _inject_slide_part( + source, + shape_xml=( + '' + '' + '' + ), + relationship_xml=( + '' + ), + part_name="ppt/media/image2.bmp", + payload=b"BM" + b"\x00" * 64, + ) + + result = detect.detect(tmp_path, cache_root=tmp_path / "graphify-out") + + assert not any(path.endswith(".bmp") for path in result["files"]["image"]) + assert not any(path.endswith(".bmp") for path in result["files"]["document"]) + assert any( + "Graphify vision cannot decode" in warning + for warning in result["skipped_sensitive"] + ) + + +def test_incremental_pptx_bundle_update_and_deletion_lifecycle(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + manifest_path = tmp_path / "graphify-out" / "manifest.json" + + first = detect.detect_incremental(tmp_path, manifest_path=str(manifest_path)) + assert first["new_files"]["document"] + assert len(first["new_files"]["image"]) == 2 + assert len(first["new_files"]["video"]) == 2 + first_media = set(first["new_files"]["video"]) + detect.save_manifest( + first["files"], manifest_path=str(manifest_path), kind="both", root=tmp_path + ) + + unchanged = detect.detect_incremental(tmp_path, manifest_path=str(manifest_path)) + assert not any(unchanged["new_files"].values()) + assert first_media <= set(unchanged["unchanged_files"]["video"]) + + original_times = (source.stat().st_atime, source.stat().st_mtime) + _replace_zip_parts(source, {"ppt/media/audio1.mp3": b"ID3 changed audio payload"}) + os.utime(source, original_times) + changed = detect.detect_incremental(tmp_path, manifest_path=str(manifest_path)) + assert changed["new_files"]["document"] + assert changed["new_files"]["video"] + assert first_media & set(changed["deleted_files"]) + detect.save_manifest( + changed["files"], manifest_path=str(manifest_path), kind="both", root=tmp_path + ) + + source.unlink() + deleted = detect.detect_incremental(tmp_path, manifest_path=str(manifest_path)) + assert deleted["deleted_files"] + assert not any(deleted["files"].values()) + + +def test_orphan_cleanup_rejects_tampered_source_path(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + output = tmp_path / "graphify-out" / "converted" + artifacts = convert_presentation_file(source, output, root=tmp_path) + bundle = artifacts.manifest_path.parent + manifest = json.loads(artifacts.manifest_path.read_text(encoding="utf-8")) + manifest["source_path"] = "../../outside.pptx" + artifacts.manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + outside = tmp_path.parent / "outside.pptx" + outside.write_bytes(b"must remain") + source.unlink() + + assert cleanup_orphaned_presentation_bundles(output, root=tmp_path) == () + assert bundle.is_dir() + assert outside.read_bytes() == b"must remain" + outside.unlink() + + +def test_cache_uses_content_and_converter_version_not_mtime(tmp_path: Path, monkeypatch): + import graphify.presentation as presentation + + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + out = tmp_path / "converted" + first = convert_presentation_file(source, out, root=tmp_path) + first_text = first.markdown_path.read_text(encoding="utf-8") + original_times = (source.stat().st_atime, source.stat().st_mtime) + + _replace_zip_parts( + source, + {"docProps/core.xml": "Changed Same Mtime"}, + ) + os.utime(source, original_times) + second = convert_presentation_file(source, out, root=tmp_path) + assert second.markdown_path.read_text(encoding="utf-8") != first_text + assert "Changed Same Mtime" in second.markdown_path.read_text(encoding="utf-8") + + monkeypatch.setattr(presentation, "CONVERTER_VERSION", "test-new-version") + third = convert_presentation_file(source, out, root=tmp_path) + manifest = json.loads(third.manifest_path.read_text(encoding="utf-8")) + assert manifest["converter_version"] == "test-new-version" + + +def test_cached_bundle_does_not_bypass_stricter_limits(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + output = tmp_path / "converted" + convert_presentation_file(source, output, root=tmp_path) + + with pytest.raises(PresentationError, match="asset limit"): + convert_presentation_file( + source, + output, + root=tmp_path, + limits=PresentationLimits(max_assets=0), + ) + + +def test_tampered_cached_markdown_is_regenerated(tmp_path: Path): + source = tmp_path / "strategy.pptx" + _write_rich_pptx(source) + output = tmp_path / "converted" + first = convert_presentation_file(source, output, root=tmp_path) + first.markdown_path.write_text("FORGED CACHE CONTENT", encoding="utf-8") + + second = convert_presentation_file(source, output, root=tmp_path) + + assert "FORGED CACHE CONTENT" not in second.markdown_path.read_text(encoding="utf-8") + assert "Q4 Research Strategy" in second.markdown_path.read_text(encoding="utf-8") + + +def test_utf16_dtd_and_entity_are_rejected(tmp_path: Path): + source = tmp_path / "entity.pptx" + _write_rich_pptx(source) + payload = ( + '' + ']>' + '&x;' + ).encode("utf-16") + _replace_zip_parts(source, {"docProps/core.xml": payload}) + + with pytest.raises(PresentationError, match="DTD|entity"): + convert_presentation_file(source, tmp_path / "converted", root=tmp_path) + + +def test_shared_layout_populates_master_for_every_slide(tmp_path: Path): + source = tmp_path / "shared-layout.pptx" + _write_rich_pptx(source) + presentation = """ + + + + +""" + presentation_rels = """ + + +""" + with zipfile.ZipFile(source) as package: + slide1 = package.read("ppt/slides/slide1.xml") + slide1_rels = package.read("ppt/slides/_rels/slide1.xml.rels").decode("utf-8") + # Ensure slide1 has an explicit layout relationship. + if "slidelayout" not in slide1_rels.lower() and "slideLayout" not in slide1_rels: + slide1_rels = slide1_rels.replace( + "", + '' + "", + ) + layout = """ + + Shared Layout + +""" + layout_rels = """ + +""" + master = """ + + Shared Master + +""" + slide2 = slide1.decode("utf-8").replace("Q4 Research Strategy", "Second Slide Title") + slide2_rels = slide1_rels + content_types = """ + + + + + + + + + + + + + + + + +""" + _replace_zip_parts( + source, + { + "[Content_Types].xml": content_types, + "ppt/presentation.xml": presentation, + "ppt/_rels/presentation.xml.rels": presentation_rels, + "ppt/slides/slide1.xml": slide1, + "ppt/slides/_rels/slide1.xml.rels": slide1_rels, + "ppt/slides/slide2.xml": slide2, + "ppt/slides/_rels/slide2.xml.rels": slide2_rels, + "ppt/slideLayouts/slideLayout1.xml": layout, + "ppt/slideLayouts/_rels/slideLayout1.xml.rels": layout_rels, + "ppt/slideMasters/slideMaster1.xml": master, + }, + ) + + artifacts = convert_presentation_file(source, tmp_path / "converted", root=tmp_path) + manifest = json.loads(artifacts.manifest_path.read_text(encoding="utf-8")) + slides = {(item["number"], item["layout_part"], item["master_part"]) for item in manifest["slides"]} + assert (1, "ppt/slideLayouts/slideLayout1.xml", "ppt/slideMasters/slideMaster1.xml") in slides + assert (2, "ppt/slideLayouts/slideLayout1.xml", "ppt/slideMasters/slideMaster1.xml") in slides + + +def test_custom_out_places_pptx_bundle_under_output_root(tmp_path: Path): + source_root = tmp_path / "src" + source_root.mkdir() + source = source_root / "strategy.pptx" + _write_rich_pptx(source) + out_root = tmp_path / "dest" + + result = detect.detect(source_root, cache_root=out_root) + + docs = result["files"]["document"] + assert docs + assert all(str(out_root) in path for path in docs) + assert not (source_root / "graphify-out").exists() + assert (out_root / "graphify-out" / "converted").is_dir() + + +def test_embedded_office_attachment_sidecar_keeps_parent_provenance(tmp_path: Path, monkeypatch): + attachment = tmp_path / "strategy_abcd_pptx-slide-0001-attachment-01-deadbeef.xlsx" + attachment.write_bytes(b"PK\x03\x04placeholder") + parent = tmp_path / "strategy.pptx" + parent.write_bytes(b"fake") + markdown = tmp_path / "out" / "graphify-out" / "converted" / "strategy_abcd_pptx" / "presentation.md" + markdown.parent.mkdir(parents=True) + markdown.write_text("# parent", encoding="utf-8") + + def _fake_xlsx(_path: Path) -> str: + return "## Sheet: Evidence\n\n| ParentProvenanceMarker |\n| --- |" + + monkeypatch.setattr(detect, "xlsx_to_markdown", _fake_xlsx) + sidecar = detect.convert_office_file( + attachment, + tmp_path / "out" / "graphify-out" / "converted", + root=tmp_path, + provenance={ + "parent_presentation": str(parent), + "parent_slide": "1", + "embedded_attachment": attachment.name, + "presentation_markdown": str(markdown), + }, + ) + assert sidecar is not None + text = sidecar.read_text(encoding="utf-8") + assert "