diff --git a/.github/workflows/markdown-export.yml b/.github/workflows/markdown-export.yml new file mode 100644 index 00000000..cfc0c742 --- /dev/null +++ b/.github/workflows/markdown-export.yml @@ -0,0 +1,221 @@ +name: Markdown Export + +on: + pull_request: + push: + branches: [develop] + tags: ['FieldWorks*'] + workflow_dispatch: + inputs: + dry_run: + description: 'Build and report, but do not publish' + type: boolean + default: false + +concurrency: + group: markdown-export + cancel-in-progress: false + +permissions: + contents: read + +env: + PANDOC_VERSION: '3.9.0.2' + PANDOC_SHA256: ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567 + EXPORT_BRANCH: markdown-export + +jobs: + validate: + name: Validate markdown export + runs-on: ubuntu-latest + permissions: + contents: read + steps: + # The runner context is not available to job-level env, so the build + # paths are resolved here from $RUNNER_TEMP for every later step. + - name: Resolve build paths + run: | + set -euo pipefail + { + echo "EXPORT_DIR=${RUNNER_TEMP}/markdown-export" + echo "WORK_DIR=${RUNNER_TEMP}/markdown-export-work" + echo "DIAGNOSTICS=${RUNNER_TEMP}/markdown-export-diagnostics.json" + } >> "$GITHUB_ENV" + + - name: Checkout source + uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 + with: + path: src + fetch-depth: 1 + + - name: Install Pandoc and CHM extractor + run: | + set -euo pipefail + curl -fsSL -o /tmp/pandoc.deb \ + "https://github.com/jgm/pandoc/releases/download/${PANDOC_VERSION}/pandoc-${PANDOC_VERSION}-1-amd64.deb" + echo "${PANDOC_SHA256} /tmp/pandoc.deb" | sha256sum --check --strict + sudo dpkg -i /tmp/pandoc.deb + # p7zip-full remains a runner-image package dependency for CHM extraction. + sudo apt-get update -qq + sudo apt-get install -y -qq p7zip-full + pandoc --version | head -2 + 7z i > /dev/null && echo "7z ok" + + - name: Install uv + uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e + with: + version: '0.9.4' + + - name: Provision locked Python and dependencies + working-directory: src + run: | + set -euo pipefail + uv python install + uv lock --check + uv sync --frozen + + - name: Lint converters + working-directory: src + run: uv run --frozen ruff check tools + + - name: Test converters + working-directory: src + run: uv run --frozen python -m unittest discover -s tools -p 'test_*.py' -v + + - name: Convert + run: | + set -euo pipefail + rm -rf -- "${EXPORT_DIR}" "${WORK_DIR}" + # --directory already puts uv in the checkout, so --repo is relative to it. + uv run --frozen --directory src python tools/convert.py \ + --repo . \ + --out "${EXPORT_DIR}" \ + --work "${WORK_DIR}" \ + --diagnostics "${DIAGNOSTICS}" \ + --source-ref "${GITHUB_SHA::7}" + + - name: Summarise quality report + if: always() + run: | + uv run --frozen --directory src python -c ' + import json, os + from pathlib import Path + diagnostics = Path(os.environ["DIAGNOSTICS"]) + if not diagnostics.exists(): + print("## Markdown Export\n\nNo diagnostics produced — validation failed early.") + raise SystemExit(0) + r = json.loads(diagnostics.read_text(encoding="utf-8")) + corpus = r.get("corpus", {}) + summary = r.get("summary", {}) + source_ref = corpus.get("source_ref", "?") + topic_count = corpus.get("topic_count", 0) + pdf_count = corpus.get("pdf_count", 0) + fatal_count = summary.get("fatal", 0) + advisory_count = summary.get("advisory", 0) + print(f"## Markdown Export — ref {source_ref}\n") + print(f"**{topic_count:,} topics and {pdf_count:,} PDFs converted**\n") + print("| Check | Count |\n| --- | ---: |") + print(f"| fatal | {fatal_count} |") + print(f"| advisory | {advisory_count} |") + for code, count in sorted(summary.get("by_code", {}).items()): + label = code.replace("_", " ") + print(f"| {label} | {count} |") + for code in ("source_missing_link", "source_missing_image", "missing_link", "missing_image", "not_in_toc"): + items = [item for item in r.get("issues", []) if item.get("code") == code] + if items: + label = items[0].get("label", code.replace("_", " ")) + print(f"\n
{label} ({len(items)})\n") + for item in items[:50]: + item_label = item.get("label", label) + item_path = item.get("path", "") + item_message = item.get("message", "") + print(f"- **{item_label}** `{item_path}`: {item_message}") + print("\n
") + ' >> "$GITHUB_STEP_SUMMARY" + + - name: Upload diagnostics + if: always() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: markdown-export-diagnostics + path: ${{ runner.temp }}/markdown-export-diagnostics.json + if-no-files-found: ignore + retention-days: 14 + + - name: Upload completed export + if: success() + uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 + with: + name: markdown-export + path: ${{ runner.temp }}/markdown-export + if-no-files-found: error + include-hidden-files: true + retention-days: 14 + + publish: + name: Publish markdown export + needs: validate + if: >- + needs.validate.result == 'success' && + (github.event_name == 'push' || + (github.event_name == 'workflow_dispatch' && !inputs.dry_run)) + runs-on: ubuntu-latest + permissions: + contents: write + steps: + - name: Download completed export + uses: actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0 + with: + name: markdown-export + path: ${{ runner.temp }}/markdown-export + + - name: Publish incremental markdown-export history + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + EXPORT_DIR: ${{ runner.temp }}/markdown-export + PUBLISH_DIR: ${{ runner.temp }}/markdown-export-publish + run: | + set -euo pipefail + rm -rf -- "${PUBLISH_DIR}" + mkdir -p "${PUBLISH_DIR}" + git -C "${PUBLISH_DIR}" init -q -b "${EXPORT_BRANCH}" + git -C "${PUBLISH_DIR}" config user.name "github-actions[bot]" + git -C "${PUBLISH_DIR}" config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git -C "${PUBLISH_DIR}" remote add origin \ + "https://x-access-token:${GH_TOKEN}@github.com/${GITHUB_REPOSITORY}.git" + if git -C "${PUBLISH_DIR}" fetch --no-tags origin \ + "refs/heads/${EXPORT_BRANCH}:refs/remotes/origin/${EXPORT_BRANCH}"; then + git -C "${PUBLISH_DIR}" checkout -q -B "${EXPORT_BRANCH}" \ + "refs/remotes/origin/${EXPORT_BRANCH}" + else + git -C "${PUBLISH_DIR}" checkout -q -B "${EXPORT_BRANCH}" + fi + git -C "${PUBLISH_DIR}" rm -r -q --ignore-unmatch -- . + git -C "${PUBLISH_DIR}" clean -fdx -q + cp -a "${EXPORT_DIR}/." "${PUBLISH_DIR}/" + git -C "${PUBLISH_DIR}" add -A + if git -C "${PUBLISH_DIR}" diff --cached --quiet; then + echo "${EXPORT_BRANCH} is already up to date" + else + git -C "${PUBLISH_DIR}" commit -q -m "Markdown export of help and PDFs ${GITHUB_SHA::7}" + git -C "${PUBLISH_DIR}" push origin "HEAD:${EXPORT_BRANCH}" + echo "published to ${EXPORT_BRANCH}" + fi + echo "PUBLISH_DIR=${PUBLISH_DIR}" >> "$GITHUB_ENV" + + - name: Tag the export to match the release + if: startsWith(github.ref, 'refs/tags/FieldWorks') && + (github.event_name != 'workflow_dispatch' || !inputs.dry_run) + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + cd "${PUBLISH_DIR}" + TAG="${EXPORT_BRANCH}/${GITHUB_REF_NAME}" + if git ls-remote --exit-code --tags origin "refs/tags/${TAG}" > /dev/null 2>&1; then + echo "tag already exists: ${TAG}" + else + git tag -a "${TAG}" -m "Markdown export for ${GITHUB_REF_NAME}" + git push origin "refs/tags/${TAG}" + echo "tagged ${TAG}" + fi diff --git a/.gitignore b/.gitignore new file mode 100644 index 00000000..ab990338 --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +.chm-work/ +*.pyc +__pycache__/ +tools/out/ +.fwhelps-export-*.lock diff --git a/.python-version b/.python-version new file mode 100644 index 00000000..86f8c02e --- /dev/null +++ b/.python-version @@ -0,0 +1 @@ +3.13.5 diff --git a/README.md b/README.md new file mode 100644 index 00000000..09dee37f --- /dev/null +++ b/README.md @@ -0,0 +1,131 @@ +# FwHelps + +Documentation for [FieldWorks Language Explorer](https://software.sil.org/fieldworks/) +(FLEx). Help content is authored in Adobe RoboHelp and committed here as a +compiled CHM, alongside training and technical-note PDFs. + +| File | Contents | +| --- | --- | +| `FieldWorks_Language_Explorer_Help.chm` | The main help system — 1,599 topics | +| `Language Explorer/Training/` | Technical notes: Send-Receive, imports, Word export | +| `Language Explorer/Utilities/` | AlloGen, PcPatr, ToneParsFLEx, VarGen documentation | +| `WW-ConceptualIntro/` | Conceptual Introduction to FLEx | + +The FieldWorks installer consumes this repo directly: `patch-installer-cd.yml` +in [sillsdev/FieldWorks](https://github.com/sillsdev/FieldWorks) checks it out +via a `helps_ref` input, and `Build/releaseTagger.py` tags it `FieldWorks` +at release time. + +## Markdown export + +The CHM is also published as markdown on the +[**`markdown-export`**](../../tree/markdown-export) branch — one file per help +topic, with images, YAML frontmatter, and a full table of contents. + +It exists for two reasons: + +- **AI retrieval.** The FieldWorks AI bot previously ingested raw RoboHelp + HTML, where roughly two thirds of every topic is markup rather than + documentation. The markdown corpus is about 65% smaller in tokens + (~2.14M → ~759K) with the prose intact. +- **Reviewable diffs.** A help change is otherwise a 5 MB opaque binary. + On the export branch it is a readable text diff, one changed file per + edited topic. + +Built automatically by +[`markdown-export.yml`](.github/workflows/markdown-export.yml) on every push to +`develop`. Nothing there is hand-edited — edit the help in RoboHelp and commit +the CHM. + +Each build also produces `author-report.md` for RoboHelp/PDF authors and +`author-report.json` for automation. Both cover broken links, topics missing +from the table of contents, and other source/export quality findings. The +Markdown report preserves the exact source path and evidence for every finding +and gives issue-specific repair guidance; JSON retains the stable +`{corpus, summary, issues}` schema. + +Exporter mutation boundaries use a non-blocking native advisory lock on a +deterministic sibling lockfile, so cooperating invocations targeting the same +destination fail clearly instead of racing. The lock is a coordination aid, +not protection against a local process that deliberately ignores file locks; +stale lockfiles are harmless because ownership is held by the OS handle. + +### Versions + +Each FieldWorks release is tagged here by `releaseTagger.py`; the matching +export is tagged `markdown-export/`. + +| FieldWorks | Released | Markdown export | +| --- | --- | --- | +| 9.3.7-beta | 2026-02-25 | [`markdown-export/FieldWorks9.3.7-beta`](../../tree/markdown-export/FieldWorks9.3.7-beta) | +| 9.3.6-beta | 2026-01-29 | [`markdown-export/FieldWorks9.3.6-beta`](../../tree/markdown-export/FieldWorks9.3.6-beta) | +| 9.3.4 | 2025-10-30 | [`markdown-export/FieldWorks9.3.4`](../../tree/markdown-export/FieldWorks9.3.4) | +| 9.3.1 | 2025-07-25 | [`markdown-export/FieldWorks9.3.1`](../../tree/markdown-export/FieldWorks9.3.1) | +| 9.3.0 | 2025-06-17 | [`markdown-export/FieldWorks9.3.0`](../../tree/markdown-export/FieldWorks9.3.0) | + +> [!NOTE] +> Export tags are created going forward, as each release is tagged. Rows above +> that have no corresponding export tag yet can be backfilled by re-running the +> workflow against that tag. + +## Tools + +| Script | Purpose | +| --- | --- | +| [`tools/convert.py`](tools/convert.py) | CHM → markdown corpus (the build) | +| [`tools/fwhelp.lua`](tools/fwhelp.lua) | Pandoc filter: RoboHelp semantics → clean GFM | +| [`tools/chm_extract.py`](tools/chm_extract.py) | Cross-platform CHM extraction, with validation | +| [`tools/pdf_convert.py`](tools/pdf_convert.py) | PDF → markdown (bookmarks or font inference) | +| [`tools/pdf_outlines.json`](tools/pdf_outlines.json) | Pinned PDF outlines; drift fails the build | +| [`tools/survey.py`](tools/survey.py) | Read-only census of the corpus | + +Local build (needs `pandoc` 3.x, and `7z` or Windows' built-in `hh.exe`): + +```sh +uv run --frozen tools/convert.py --repo . --out export +``` + +### Reproducible local setup + +The converter uses uv with the exact Python version in +[`.python-version`](.python-version). `uv.lock` is the authoritative dependency +file; `requirements.txt` and `requirements-dev.txt` are deterministic +lock-derived compatibility exports for older tooling and should not be edited +independently. The CI workflow installs only from the frozen lock: + +```sh +uv python install +uv lock --check +uv sync --frozen +uv run --frozen \ + -m unittest discover -s tools -p 'test_*.py' -v +uv run --frozen ruff check tools +uv run --frozen \ + tools/convert.py --repo . --out export +uv run --frozen \ + tools/pdf_convert.py --repo . --out export --update-outlines +``` + +To regenerate the compatibility exports after changing project dependencies: + +```sh +uv export --frozen --no-dev --no-hashes --format requirements.txt \ + --output-file requirements.txt +uv export --frozen --only-group dev --no-hashes --format requirements.txt \ + --output-file requirements-dev.txt +``` + +On PowerShell, the same `uv run --frozen` commands work unchanged. Updating +outline locks is intentional and should be reviewed with the resulting +`tools/pdf_outlines.json` change. + +The workflow still installs the runner-image package `p7zip-full` for CHM +extraction; it is intentionally outside the Python lock. Pandoc is downloaded +as the pinned 3.9.0.2 amd64 package and its SHA-256 is checked before install. + +> [!IMPORTANT] +> `hh.exe -decompile` silently truncates filenames when the output path exceeds +> Windows' 260-character limit — no error, no non-zero exit. `chm_extract.py` +> refuses to run it into a too-long path and validates every extraction against +> the CHM's own table of contents. Use a short `--work` directory on Windows, +> or install 7-Zip. diff --git a/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md b/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md new file mode 100644 index 00000000..47caf5df --- /dev/null +++ b/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md @@ -0,0 +1,349 @@ +# Portable Markdown Export Final Hardening Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Close every confirmed safety, correctness, portability, reporting, and CI gap from the final cross-cutting review without changing valid current-corpus content. + +**Architecture:** Harden source discovery and extraction at their input boundaries, centralize issue policy in one module, and split CI validation from privileged publication. Converter-owned manifests authenticate reusable or removable state; validation treats ambiguous or unsafe output as fatal. + +**Tech Stack:** Python 3.13.5, uv/`uv.lock`, `unittest`, Pandoc 3.9.0.2, Lua filters, GitHub Actions, PowerShell/Linux shell verification. + +--- + +## File responsibilities + +- `tools/source_safety.py`: repository-containment and non-symlink source discovery helpers shared by CHM and PDF tracks. +- `tools/frontmatter.py`: shared JSON-compatible YAML scalar serialization for CHM and PDF metadata. +- `tools/issue_catalog.py`: canonical issue code, label, severity, and provenance policy. +- `tools/output_fs.py`: staged promotion plus cooperative cross-process destination locking. +- `tools/chm_extract.py`: isolated extraction, parsed reference validation, and destructive-target rejection. +- `tools/chm_convert.py`: authenticated extraction reuse, destination collision checks, safe frontmatter, and unsafe-link reporting. +- `tools/pdf_convert.py`: safe PDF discovery and authenticated converter-manifest cleanup. +- `tools/corpus_validation.py`: emitted-corpus path and URI policy. +- `tools/fwhelp.lua`: AST transformations that inventory all authored classes and neutralize unsafe URI schemes. +- `tools/convert.py`: thin orchestration and diagnostic-report persistence. +- `.github/workflows/markdown-export.yml`: read-only validation job and separately privileged publication job. +- `pyproject.toml`, `uv.lock`, `.python-version`: exact Python and complete dependency lock. + +### Task 1: Source identity, symlinks, and extraction safety + +**Files:** +- Create: `tools/source_safety.py` +- Create: `tools/test_source_safety.py` +- Modify: `tools/chm_extract.py` +- Modify: `tools/chm_convert.py` +- Modify: `tools/pdf_convert.py` +- Test: `tools/test_chm_extract.py` +- Test: `tools/test_convert.py` +- Test: `tools/test_pdf_convert.py` + +- [ ] **Step 1: Write failing source-boundary tests** + +Add tests with this behavior: + +```python +def test_source_file_rejects_symlink_outside_repository(self): + outside = self.root.parent / "outside.pdf" + outside.write_bytes(b"pdf") + link = self.root / "linked.pdf" + link.symlink_to(outside) + with self.assertRaises(SourceSafetyError): + discover_source_files(self.root, suffixes={".pdf"}, recursive=True) + +def test_extract_rejects_source_directory_as_destination(self): + chm = self.root / "Help.chm" + chm.write_bytes(b"chm") + with self.assertRaises(OutputPathError): + extract(chm, self.root) +``` + +- [ ] **Step 2: Run the focused tests and verify RED** + +Run: + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +``` + +Expected: failures for missing `SourceSafetyError`, symlink acceptance, source-directory extraction, and stale reuse. + +- [ ] **Step 3: Implement shared source discovery and authenticated reuse** + +Create a helper with this contract: + +```python +class SourceSafetyError(ValueError): + pass + +def discover_source_files( + root: Path, *, suffixes: set[str], recursive: bool +) -> list[Path]: + """Return stable, regular, non-symlink files resolving beneath root.""" +``` + +Write `.chm-extraction-manifest.json` only after validated extraction promotion: + +```json +{ + "schema": 1, + "source_name": "FieldWorks_Language_Explorer_Help.chm", + "source_sha256": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" +} +``` + +Reuse only when schema, name, and hash match; otherwise perform fresh isolated extraction. Reject `outdir.resolve() == chm.parent.resolve()` and any destination that contains the CHM path. + +- [ ] **Step 4: Verify GREEN** + +Run the focused command from Step 2. Expected: all focused tests pass. + +### Task 2: CHM parsing, collisions, frontmatter, and URI safety + +**Files:** +- Modify: `tools/chm_extract.py` +- Modify: `tools/chm_convert.py` +- Modify: `tools/chm_metadata.py` +- Modify: `tools/corpus_validation.py` +- Modify: `tools/fwhelp.lua` +- Test: `tools/test_chm_extract.py` +- Test: `tools/test_chm_metadata.py` +- Test: `tools/test_convert.py` +- Test: `tools/test_corpus_validation.py` + +- [ ] **Step 1: Write failing conversion-boundary tests** + +Cover these exact cases: + +```python +def test_absolute_local_target_is_fatal(self): + (self.root / "topic.md").write_text("# Topic\n\n[bad](/outside.md)\n", encoding="utf-8") + issues = validate_corpus(self.root) + self.assertTrue(any(i.code == "path_escape" and i.fatal for i in issues)) + +def test_mixed_known_and_unknown_span_class_reports_unknown(self): + markdown, unmapped = run_pandoc('text') + self.assertIn("NewSemantic", unmapped) + +def test_casefolded_image_collision_is_fatal(self): + # Icon.png and icon.PNG must claim one normalized destination. + self.assertIn("destination_collisions", result["report"]) +``` + +Also add HHC fixtures using single quotes, reversed `value`/`name` attributes, escaped paths, `.woff`, `.webp`, `.json`, and `.map`; YAML values with newlines/control characters; and `javascript:`, `file:`, `data:`, `https:`, and `mailto:` links. + +- [ ] **Step 2: Verify RED** + +Run: + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +``` + +Expected: failures for each new boundary behavior. + +- [ ] **Step 3: Implement parsed references and allowlisted output** + +Use `html.parser.HTMLParser` to collect `param` elements whose case-folded `name` is `local`, independent of attribute ordering or quote style. Validate local targets referenced by TOC and HTML `href`/`src`; permit unrelated asset extensions. Detect truncation only when an expected target is missing and a prefix sibling exists. + +Build a shared case-folded claim table for every topic and image destination before writing. Inventory every span class before selecting the first supported transformation. Permit only `http`, `https`, and `mailto` external schemes; fragments and relative paths remain local. Neutralize source-authored unsafe targets and report them with source provenance; emitted unsafe targets remain fatal. + +Serialize frontmatter scalars through one JSON-compatible YAML quoting function: + +```python +def yaml_scalar(value: str) -> str: + return json.dumps(value, ensure_ascii=False) +``` + +JSON quoting escapes newlines and C0 controls, preventing YAML injection while preserving adversarial source metadata as data rather than rejecting an otherwise convertible help file. + +- [ ] **Step 4: Verify GREEN** + +Run the focused command from Step 2. Expected: all focused tests pass and ordinary images/links remain unchanged. + +### Task 3: PDF manifest ownership and canonical issue policy + +**Files:** +- Create: `tools/issue_catalog.py` +- Create: `tools/test_issue_catalog.py` +- Modify: `tools/pdf_convert.py` +- Modify: `tools/reporting.py` +- Modify: `tools/convert.py` +- Modify: `tools/corpus_validation.py` +- Test: `tools/test_pdf_convert.py` +- Test: `tools/test_reporting.py` +- Test: `tools/test_convert.py` + +- [ ] **Step 1: Write failing manifest and catalog tests** + +```python +def test_corrupt_manifest_cannot_delete_unrelated_output(self): + unrelated = self.out / "keep.md" + unrelated.write_text("keep", encoding="utf-8") + previous = {"source.pdf": {"markdown": "keep.md", "images": None}} + with self.assertRaises(ManifestError): + promote_pdf_outputs(self.out, self.stage, previous, {}) + self.assertEqual(unrelated.read_text(encoding="utf-8"), "keep") + +def test_every_emitted_issue_code_has_one_policy(self): + for code in emitted_issue_codes(): + self.assertIn(code, ISSUE_CATALOG) +``` + +- [ ] **Step 2: Verify RED** + +Run: + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +``` + +Expected: corrupt-manifest and missing-central-policy failures. + +- [ ] **Step 3: Authenticate removable PDF paths and centralize policy** + +Version the PDF manifest and record source plus expected derived destinations: + +```json +{ + "schema": 2, + "files": { + "Language Explorer/Training/source.pdf": { + "markdown": "Language_Explorer/Training/source.md", + "images": "Language_Explorer/Training/source_images" + } + } +} +``` + +Before backup or deletion, recompute `slug_path(source)` and require exact equality with both manifest destinations. Reject schema 1 or corrupt entries before mutation; the next successful run may replace a valid schema-2 set. + +Define `IssuePolicy(label, fatal, provenance)` once. Make report creation use `make_issue(code, ...)`; unknown codes at integration boundaries become fatal exporter issues. Mark retained PDF HTML as source provenance. + +- [ ] **Step 4: Verify GREEN** + +Run the focused command from Step 2. Expected: all tests pass, including rollback tests. + +### Task 4: Reproducible, observable, least-privilege CI + +**Files:** +- Create: `pyproject.toml` +- Create: `uv.lock` +- Modify: `.python-version` +- Modify: `requirements.txt` +- Modify: `requirements-dev.txt` +- Modify: `.github/workflows/markdown-export.yml` +- Modify: `tools/convert.py` +- Modify: `README.md` +- Test: `tools/test_workflow.py` +- Test: `tools/test_convert.py` + +- [ ] **Step 1: Write failing workflow and diagnostics tests** + +Assert the workflow contains: + +```python +self.assertIn("pull_request:", workflow) +self.assertIn("permissions:\n contents: read", workflow) +self.assertIn("uv sync --frozen", workflow) +self.assertIn("ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567", workflow) +self.assertNotRegex(workflow, r"uses:\s+[^\s]+@v\d+") +self.assertRegex(workflow, r"publish:\s*[\s\S]+permissions:\s*\n\s+contents: write") +``` + +Add a converter test proving `--diagnostics audit-diagnostics.json` writes canonical JSON even when corpus promotion is rejected for a fatal issue. + +- [ ] **Step 2: Verify RED** + +Run: + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +``` + +Expected: failures for missing PR trigger, mutable actions, missing digest/frozen lock, broad write permission, and absent failure diagnostics. + +- [ ] **Step 3: Lock dependencies and split workflow privileges** + +Set `.python-version` to `3.13.5`. Declare runtime dependencies and Ruff in `pyproject.toml`, then run: + +```powershell +uv lock --python 3.13.5 +uv sync --frozen +``` + +Keep `requirements.txt` and `requirements-dev.txt` as compatibility exports generated from the lock and document that `uv.lock` is authoritative. + +Use these immutable pins: + +```yaml +actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 +astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e +actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 +actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0 +``` + +Verify Pandoc before installation: + +```bash +echo "ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567 /tmp/pandoc.deb" | sha256sum --check --strict +``` + +The `validate` job uses `contents: read`, runs on `pull_request`, trusted pushes/tags, and dispatch, writes diagnostics, and always uploads them. A separate `publish` job has `contents: write`, downloads the successful export artifact, and runs only for trusted push/tag or non-dry-run dispatch events. Document `p7zip-full` as the remaining runner-image package dependency. + +- [ ] **Step 4: Verify GREEN** + +Run: + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +.review-venv\Scripts\ruff.exe check tools +uv lock --check +uv sync --frozen +``` + +Expected: all commands exit zero. + +### Task 5: Full verification and independent review + +**Files:** +- Modify only files required by verified review findings. + +- [ ] **Step 1: Run the complete local gates** + +```powershell +.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py' +.review-venv\Scripts\ruff.exe check tools +git diff --check +uv lock --check +``` + +Expected: all tests pass, Ruff reports `All checks passed!`, and both remaining commands exit zero. + +- [ ] **Step 2: Run a real corpus export** + +Use fresh verified temporary directories and run: + +```powershell +$sourceRef = git rev-parse --short HEAD +.review-venv\Scripts\python.exe tools\convert.py --repo . --out $auditOut --work $auditWork --source-ref $sourceRef --diagnostics $auditDiagnostics +``` + +Expected: exit zero; report has two CHMs, 1,630 topics, 13 PDFs, and zero fatal issues. + +- [ ] **Step 3: Dispatch independent Luna spec and quality reviews** + +Review `git diff 1ecb705...HEAD` against the final-hardening design. Resolve every Critical or Important finding and repeat the relevant test/review gate. + +- [ ] **Step 4: Commit and push one implementation commit** + +```powershell +git add -- .github/workflows/markdown-export.yml .python-version README.md requirements.txt requirements-dev.txt pyproject.toml uv.lock tools docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md +git commit -m "Close final Markdown export hardening gaps" +git push fork tools/markdown-export +``` + +Expected: ordinary fast-forward push; PR #3 updates without force. + +- [ ] **Step 5: Regenerate and publish Markdown incrementally** + +Generate from the committed SHA, update the existing `johnml1135/FwHelps:markdown-export` tree using tracked-file deletion plus ordinary commit/push, and verify the remote report has zero fatal issues and 2,356 files unless intentional output changes alter that count. diff --git a/docs/superpowers/plans/2026-08-21-portable-markdown-export.md b/docs/superpowers/plans/2026-08-21-portable-markdown-export.md new file mode 100644 index 00000000..01ad41a5 --- /dev/null +++ b/docs/superpowers/plans/2026-08-21-portable-markdown-export.md @@ -0,0 +1,81 @@ +# Portable Markdown Export Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Harden the CHM/PDF exporter into a safe, reproducible, portable converter with clean and validated generated Markdown. + +**Architecture:** Repository policy remains in a thin orchestration module while extraction, PDF conversion, output staging, and corpus validation expose small reusable interfaces. Every build is staged and validated before replacement, and CI publishes an ordinary incremental branch history. + +**Tech Stack:** Python 3.13 managed by uv, Pandoc 3.9.0.2 with Lua, PyMuPDF, pymupdf4llm, unittest, Ruff, GitHub Actions. + +--- + +### Task 1: Safe CHM extraction and output ownership + +**Files:** +- Modify: `tools/chm_extract.py` +- Create: `tools/output_fs.py` +- Create: `tools/test_chm_extract.py` +- Create: `tools/test_output_fs.py` + +- [ ] Write failing tests proving each extraction backend receives an empty private directory, failed/invalid attempts do not contaminate later attempts, and only a validated extraction is promoted. +- [ ] Write failing tests proving filesystem roots, repository roots, sources, and overlapping work/output paths are rejected before removal. +- [ ] Implement `extract(chm, destination)` with per-backend staging and validated promotion. +- [ ] Implement a small output-staging interface that owns its temporary directory and atomically promotes a successful tree. +- [ ] Run the new tests and the complete test suite under uv. + +### Task 2: Complete and reproducible PDF conversion + +**Files:** +- Modify: `tools/pdf_convert.py` +- Modify: `tools/pdf_outlines.json` +- Modify: `tools/test_pdf_convert.py` + +- [ ] Write failing tests for complete normalized `(level, text)` outline comparison, fatal unpinned PDFs, traceability frontmatter, destination collisions, and removal of stale PDF outputs. +- [ ] Change outline locking to compare the complete post-normalization outline and make missing locks fatal outside `--update-outlines`. +- [ ] Add source hash, source URL, available metadata, and conversion identity to PDF frontmatter. +- [ ] Stage PDF output as a complete subtree or register all destinations before writing so stale outputs and collisions cannot survive. +- [ ] Regenerate outline locks explicitly and run all PDF tests. + +### Task 3: uv-controlled CI and incremental publication + +**Files:** +- Create: `.python-version` +- Create: `requirements.txt` +- Create: `requirements-dev.txt` +- Modify: `.github/workflows/markdown-export.yml` +- Modify: `README.md` + +- [ ] Pin Python 3.13 and exact converter dependencies, including Ruff for development. +- [ ] Replace setup-python/pip installation with uv interpreter provisioning and locked requirement installation. +- [ ] Make CI tests, lint, and conversion run through the uv-managed environment. +- [ ] Replace orphan initialization and force-push with checkout/update of ordinary `markdown-export` history, including first-publication handling. +- [ ] Document the exact local uv setup, test, conversion, and outline-update commands. + +### Task 4: Portable multi-input orchestration and emitted-corpus validation + +**Files:** +- Modify: `tools/convert.py` +- Modify: `tools/fwhelp.lua` +- Create: `tools/corpus_validation.py` +- Create: `tools/reporting.py` +- Create: `tools/test_convert.py` +- Create: `tools/test_corpus_validation.py` + +- [ ] Write failing tests for discovering both CHMs, collision rejection, disambiguating display titles, related frontmatter, proper nested lists, and validation of emitted Markdown links/images. +- [ ] Replace the hard-coded single-CHM flow with deterministic input discovery and format namespaces or explicit collision rejection. +- [ ] Move report definitions and fatality into one catalog consumed by JSON, README, console, and workflow summary. +- [ ] Fix list conversion and title selection without losing source content. +- [ ] Run post-generation validation over the staged corpus, fail on exporter-caused defects, then promote output. + +### Task 5: Full-corpus verification and review + +**Files:** +- Modify only files required by confirmed review defects. + +- [ ] Run all tests and Ruff through uv from a clean process. +- [ ] Build the complete corpus from both CHMs and all PDFs into a new temporary destination. +- [ ] Audit Markdown count, H1 contract, duplicate titles, PDF hierarchy, local links, images, malformed lists, raw HTML, replacement characters, outline locks, and stale files. +- [ ] Run the same build twice with one controlled source/output change to verify deterministic and incremental behavior. +- [ ] Perform independent specification and architecture/clean-code reviews; fix every Critical or Important issue and rerun verification. + diff --git a/docs/superpowers/plans/2026-08-22-author-report-markdown.md b/docs/superpowers/plans/2026-08-22-author-report-markdown.md new file mode 100644 index 00000000..17ed1a38 --- /dev/null +++ b/docs/superpowers/plans/2026-08-22-author-report-markdown.md @@ -0,0 +1,106 @@ +# Author Report Markdown Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Publish a repair-oriented Markdown author report beside the canonical JSON report. + +**Architecture:** Extend the canonical issue catalog with repair guidance and add a pure `Report.to_markdown()` renderer. The corpus orchestrator writes and links both formats inside its existing atomic staging boundary. + +**Tech Stack:** Python 3.13, standard-library `json`, `unittest`, uv, Ruff. + +--- + +### Task 1: Canonical Markdown renderer + +**Files:** +- Modify: `tools/issue_catalog.py` +- Modify: `tools/reporting.py` +- Test: `tools/test_reporting.py` +- Test: `tools/test_issue_catalog.py` + +- [ ] **Step 1: Write failing renderer and catalog tests** + +Add representative source findings with a `.htm` path, broken target, pipe, +newline, and structured detail. Assert that `to_markdown()` includes corpus +counts, code, severity, provenance, canonical repair guidance, exact evidence, +safe table escaping, and the JSON link. Assert every issue policy has nonempty +repair guidance. + +- [ ] **Step 2: Run the focused tests and verify RED** + +Run: `uv run --frozen python -m unittest discover -s tools -p 'test_reporting.py'` + +Expected: failure because `IssuePolicy.guidance` and `Report.to_markdown()` do +not exist. + +- [ ] **Step 3: Implement the catalog guidance and Markdown renderer** + +Add `guidance: str` to `IssuePolicy`, populate it for every canonical issue, +and implement deterministic grouping plus safe cell formatting in +`Report.to_markdown()`. + +- [ ] **Step 4: Run focused tests and verify GREEN** + +Run: `uv run --frozen python -m unittest discover -s tools -p 'test_reporting.py'` + +Expected: all reporting tests pass. + +### Task 2: Emit and link the report + +**Files:** +- Modify: `tools/convert.py` +- Modify: `tools/test_convert.py` +- Modify: `README.md` + +- [ ] **Step 1: Write a failing orchestration test** + +Assert that a successful staged build contains `author-report.md` and +`author-report.json`, and that the generated README links both files. + +- [ ] **Step 2: Run the focused orchestration test and verify RED** + +Run: `uv run --frozen python -m unittest discover -s tools -p 'test_convert.py'` + +Expected: failure because `author-report.md` is absent. + +- [ ] **Step 3: Write both reports within the atomic stage** + +Seed both linked files before corpus link validation, rewrite both after final +validation, update the generated README links, and document both formats in +the repository README. + +- [ ] **Step 4: Run focused tests and verify GREEN** + +Run: `uv run --frozen python -m unittest discover -s tools -p 'test_convert.py'` + +Expected: all orchestration tests pass. + +### Task 3: Verify, publish, and hand off + +**Files:** +- Generated: `author-report.md` on branch `markdown-export` + +- [ ] **Step 1: Run all local gates** + +Run: `uv run --frozen python -m unittest discover -s tools -p 'test_*.py'` +Run: `uv run --frozen ruff check tools` +Run: `uv lock --check` +Run: `git diff --check` + +Expected: all commands exit zero. + +- [ ] **Step 2: Commit and push the tool branch** + +Commit only the feature, tests, and approved design/plan. Keep the imported +conversation transcript untracked. Push `tools/markdown-export` to the fork. + +- [ ] **Step 3: Regenerate and verify the real corpus** + +Run the exporter from the committed SHA using the authenticated short-path +work directory. Verify 0 fatal issues, expected corpus counts, both report +formats, valid README links, and no internal lock artifacts. + +- [ ] **Step 4: Publish and verify the generated branch** + +Replace the contents of the fork's `markdown-export` branch in a temporary +clone, commit, push normally, and verify the remote file URL. diff --git a/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md b/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md new file mode 100644 index 00000000..e698a73e --- /dev/null +++ b/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md @@ -0,0 +1,49 @@ +# Portable Markdown Export Hardening Design + +## Goal + +Turn the current FwHelps CHM/PDF converter into a safe, reproducible, portable tool that produces reviewable Markdown and can later move to a shared repository without redesigning its core interfaces. + +## Scope and ownership + +The implementation remains in `FwHelps/tools` for this pass. Repository-specific policy—input discovery, source URLs, workflow triggers, and publication branch—stays at the outer orchestration seam. Extraction, document conversion, output staging, and corpus validation must not depend on the FwHelps repository name or on one hard-coded CHM. + +Both CHM files and every repository PDF are inputs. Generated destinations must be registered before writing so two inputs cannot silently claim the same path. A future extraction into a standalone repository should therefore move the converter modules with minimal changes while leaving a thin FwHelps adapter behind. + +## Safety and publication + +Conversion builds into a converter-owned staging directory. It rejects repository roots, source directories, filesystem roots, and overlapping work/output paths before any recursive removal. A successful build replaces the destination; a failed build leaves the previous destination intact. CHM extraction backends likewise use isolated temporary directories and promote only a validated result. + +The publication branch retains parent history. CI checks out the existing export branch when present, replaces only its generated tree, commits the resulting diff, and pushes normally. It must not use orphan commits or force-pushes, so ordinary branch protection remains compatible. + +## Reproducibility + +`.python-version` pins Python 3.13 for uv. Exact runtime versions live in `requirements.txt`; development-only tools live in `requirements-dev.txt`. CI installs uv, provisions the pinned interpreter, installs the locked requirements, runs tests and lint, then converts. Pandoc remains pinned. + +## Conversion contracts + +CHM conversion preserves the authored hierarchy while using disambiguating page headings when titles collide. Related-topic metadata is emitted explicitly. Nested lists must render as real Markdown lists. + +PDF conversion records the complete normalized outline `(level, text)`, not only a count. Every discovered PDF must have a lock entry in normal mode; updating locks requires the explicit update command. PDF frontmatter includes source path, stable source URL, source content hash, available PDF metadata, heading strategy, and normalized outline count. Converter-owned PDF outputs are replaced as a set so deleted inputs leave no stale files. + +## Validation and reporting + +Validation runs against the emitted corpus after all conversion steps. It verifies local links and images, destination collisions, unique output paths, duplicate display titles, malformed list markers, replacement characters, raw HTML inventory, and PDF outline locks. Build-breaking checks and advisory checks are defined once and rendered consistently in the README, JSON report, console output, and workflow summary. + +Source-document defects remain visible as advisories with provenance; exporter regressions fail the build. Complex tables may remain raw HTML when GFM cannot represent them without information loss, but every retained table is counted. + +## Tests and acceptance + +Every behavior change follows red-green-refactor. Unit tests cover path rejection, backend isolation, destination collisions, complete outline locks, unpinned PDFs, stale-output removal, multi-CHM discovery, related metadata, title disambiguation, list conversion, and emitted-corpus link/image checks. + +Acceptance requires: + +- all unit tests and focused lint passing under uv; +- a complete build of both CHMs and all PDFs; +- no exporter-caused unresolved links or images; +- no silent destination overwrites or stale generated files; +- no malformed `- -` nested-list output; +- all PDFs pinned by complete normalized outlines; +- a clean incremental publication design with no force-push; +- an audited `author-report.json` whose counts match the generated tree. + diff --git a/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md b/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md new file mode 100644 index 00000000..6465ed92 --- /dev/null +++ b/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md @@ -0,0 +1,63 @@ +# Portable Markdown Export: Final Hardening Design + +**Date:** 2026-08-21 +**Status:** Approved for planning +**Scope:** Resolve every confirmed finding from the final cross-cutting review of PR #3. + +## Objective + +Make the exporter safe to reuse outside FwHelps, reproducible in CI, observable on failure, and resistant to malformed or adversarial source files without changing the validated Markdown produced by the current corpus. + +## Safety and source identity + +- CHM extraction reuse will require a converter-owned manifest whose source SHA-256 matches the current CHM. Missing or mismatched manifests force a fresh extraction. +- Direct extraction will reject a destination equal to the source directory or containing the source CHM. A separate descendant work directory remains valid because replacing it cannot remove the source CHM. +- CHM and PDF discovery will reject symlinks and any resolved input outside the repository root. +- Absolute local Markdown and image targets will be fatal validation errors rather than silently ignored. +- PDF cleanup will accept prior manifest entries only when they match paths derivable from the manifest's recorded source PDF and remain under the PDF output root. A corrupt manifest will fail before deletion. +- Export mutation boundaries will use a non-blocking native advisory lock on a deterministic sibling lockfile. This serializes cooperating exporter invocations for the same normalized destination (and allows different destinations to proceed), but is not a defense against a hostile local process that ignores OS locks. Stale lockfiles do not block because ownership is the held OS handle, not file contents. + +## Conversion correctness + +- CHM topic and image destinations will share case-insensitive collision detection before any output is written. +- Lua span conversion will inventory every class before applying the first supported semantic transformation, so mixed known/unknown classes remain build-breaking. +- CHM TOC parsing will use an HTML parser and support attribute order, quoting, case, and escaped targets. +- Extraction validation will validate TOC targets and known truncation patterns without rejecting legitimate asset extensions solely because they are new. +- Frontmatter will use one safe serializer that rejects or escapes control characters and multiline scalar injection. +- Emitted links will use an allowlist of safe schemes. Source-authored unsafe schemes such as `javascript:` and `file:` are neutralized and reported as source advisories; any unsafe target that survives into the emitted corpus remains fatal. + +## Reporting policy + +A single issue catalog will define canonical code, label, severity, and default provenance. CHM conversion, PDF conversion, corpus validation, console rendering, README rendering, and workflow summaries will consume this catalog. Source-retained PDF HTML will be reported with source provenance. Unknown issue codes remain fatal by default at integration boundaries. + +## CI and publication + +- Pull requests will run lint, unit/integration tests, and a dry-run corpus conversion with read-only permissions. PR jobs will never publish or tag. +- Publication remains limited to trusted pushes, release tags, and non-dry-run manual dispatches. +- Failure summaries and diagnostic artifacts will upload with `if: always()` when any report or staged diagnostics exist. +- GitHub Actions will be pinned by immutable commit SHA. +- Python will be pinned to an exact patch release. +- Python dependencies, including transitive dependencies, will be captured in a uv lockfile and installed frozen. +- The downloaded Pandoc package will be verified against a committed expected SHA-256. System packages that cannot be version-pinned reliably on the runner will be explicitly identified as the remaining platform dependency. +- Workflow write permission will be scoped to the publishing job; validation jobs use read-only contents permission. + +## Testing and verification + +Each behavioral fix begins with a failing regression test. Coverage will include stale reuse, source/output overlap, symlink escape, corrupt manifests, absolute targets, image collisions, mixed span classes, HHC variants, new asset types, unsafe schemes, YAML control content, issue-catalog consistency, PR non-publication, action/checksum pins, frozen dependency installation, and failed-build artifact behavior. + +Completion requires: + +1. all unit and integration tests pass; +2. repository-wide Ruff passes; +3. workflow static tests pass; +4. a complete two-CHM/13-PDF export has zero fatal issues; +5. an independent final review has no unresolved Critical or Important findings; +6. the code branch is committed and fast-forward pushed; +7. the `markdown-export` branch is regenerated from that commit and fast-forward pushed. + +## Non-goals and retained decisions + +- Existing source-quality advisories remain advisories unless they create unsafe or invalid generated output. +- The exporter remains in FwHelps for this PR, but reusable modules contain no FwHelps-specific policy. +- GitHub repository branch-protection settings are documented and verified where visible, but are not changed by this code patch without separate authorization. +- No force pushes or history replacement are allowed. diff --git a/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md b/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md new file mode 100644 index 00000000..137b93aa --- /dev/null +++ b/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md @@ -0,0 +1,46 @@ +# Author Report Markdown Design + +## Goal + +Generate a human-readable `author-report.md` beside `author-report.json`. A +RoboHelp author must be able to identify the affected source topic or PDF, +understand the evidence, and know the appropriate repair action without +reading exporter code. + +## Design + +`Report` remains the single reporting model and JSON remains the stable +machine-readable format. The issue catalog gains canonical repair guidance so +labels, severity, provenance, and advice cannot drift between renderers. +`Report.to_markdown()` renders: + +- corpus source ref and CHM/PDF/topic/image counts; +- fatal and advisory totals plus counts by issue type; +- one section per issue code with severity, provenance, and repair guidance; +- every finding with its exact source/output path, message, and structured + evidence, escaped for deterministic Markdown tables; +- a link to the JSON report for automation and complete structured data. + +Source issues use the authored `.htm` or PDF path already retained in the +canonical issue. Generated-tree validation findings retain their emitted +`chm/.../*.md` or `pdf/.../*.md` path, making exporter failures reproducible. +Repair guidance explains whether the correction belongs in RoboHelp/PDF +source or exporter code. + +The corpus README links to both report formats. Both reports are written only +inside the private output stage and are promoted atomically with the corpus. + +## Error Handling and Safety + +Markdown rendering is pure and deterministic. Table cells escape pipes and +line breaks; structured evidence is serialized as compact Unicode JSON. An +unknown issue uses the existing fatal fallback policy and guidance directing +maintainers to add it to the canonical catalog. + +## Tests + +Tests verify the Markdown report contains corpus identity, summary totals, +canonical guidance, exact RoboHelp source paths, problematic targets, +structured evidence, escaping, and the JSON link. An orchestration test +verifies successful builds emit both report files and README links. The full +unit suite, Ruff, uv lock check, and a real corpus build remain release gates. diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 00000000..eee0c1a3 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,17 @@ +[project] +name = "fwhelps" +version = "0.0.0" +description = "FieldWorks help corpus conversion tools" +requires-python = ">=3.13.5,<3.14" +dependencies = [ + "pymupdf==1.28.2", + "pymupdf4llm==1.28.2", +] + +[dependency-groups] +dev = [ + "ruff==0.16.4", +] + +[tool.uv] +package = false diff --git a/requirements-dev.txt b/requirements-dev.txt new file mode 100644 index 00000000..c6ddef7d --- /dev/null +++ b/requirements-dev.txt @@ -0,0 +1,3 @@ +# This file was autogenerated by uv via the following command: +# uv export --frozen --only-group dev --no-hashes --format requirements.txt --output-file requirements-dev.txt +ruff==0.16.4 diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 00000000..9ec79c10 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,31 @@ +# This file was autogenerated by uv via the following command: +# uv export --frozen --no-dev --no-hashes --format requirements.txt --output-file requirements.txt +flatbuffers==25.12.19 + # via onnxruntime +networkx==3.6.1 + # via pymupdf-layout +numpy==2.5.2 + # via + # onnxruntime + # pymupdf-layout +onnxruntime==1.29.0 + # via pymupdf-layout +packaging==26.3 + # via onnxruntime +protobuf==7.36.0 + # via onnxruntime +psutil==7.2.2 + # via pymupdf4llm +pymupdf==1.28.2 + # via + # fwhelps + # pymupdf-layout + # pymupdf4llm +pymupdf-layout==1.28.2 + # via pymupdf4llm +pymupdf4llm==1.28.2 + # via fwhelps +pyyaml==6.0.3 + # via pymupdf-layout +tabulate==0.10.0 + # via pymupdf4llm diff --git a/tools/chm_convert.py b/tools/chm_convert.py new file mode 100644 index 00000000..c5c07dec --- /dev/null +++ b/tools/chm_convert.py @@ -0,0 +1,633 @@ +"""Reusable conversion of one extracted CHM into a namespaced Markdown tree.""" + +from __future__ import annotations + +import hashlib +import html +import json +import re +import subprocess +import uuid +from collections import Counter, defaultdict +from pathlib import Path +from urllib.parse import quote, unquote, urldefrag + +from chm_extract import _extract_already_locked, extract, validate +from chm_metadata import TopicMeta, parse_toc, safe_stem +from frontmatter import yaml_scalar +from output_fs import export_locks +from source_safety import ( + SourceSafetyError, + first_link_in_path, + validate_source_tree, +) + +LUA = Path(__file__).with_name("fwhelp.lua") +IMAGE_EXTS = {".gif", ".png", ".jpg", ".jpeg", ".bmp", ".svg", ".ico", ".webp"} +EXTRACTION_MANIFEST = ".chm-extraction-manifest.json" + + +def _normalize_source(source: str) -> str: + """Normalize RoboHelp NBSP spacing at the HTML-to-Markdown boundary. + + RoboHelp uses CP1252 byte 0xA0 and HTML NBSP entities for both layout + spacing and empty table cells. Markdown/Pandoc can emit U+FFFD for these + values, so ordinary spaces preserve word separation without retaining an + unsafe format-specific spacing character. + """ + source = source.replace("\u00a0", " ") + return re.sub(r"(?i) | ", " ", source) + + +def _norm(path: str) -> str: + parts: list[str] = [] + for segment in path.replace("\\", "/").split("/"): + if segment in ("", "."): + continue + if segment == "..": + if parts: + parts.pop() + else: + parts.append(segment) + return "/".join(parts) + + +def _relative(source: str, target: str) -> str: + base = Path(source).parent.parts + parts = Path(target).with_suffix(".md").parts + common = 0 + while common < len(base) and common < len(parts) and base[common] == parts[common]: + common += 1 + return "/".join(("..",) * (len(base) - common) + parts[common:]) + + +def frontmatter(fields: dict) -> str: + + lines = ["---"] + for key, value in fields.items(): + if value in (None, "", [], {}): + continue + if isinstance(value, list): + lines.append(f"{key}:") + lines.extend(f" - {yaml_scalar(item)}" for item in value) + else: + lines.append(f"{key}: {yaml_scalar(value)}") + lines.append("---") + return "\n".join(lines) + + +def run_pandoc(html_text: str, tmp: Path) -> tuple[str, list[str]]: + tmp.write_text(html_text, encoding="utf-8") + proc = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={LUA}", str(tmp)], + capture_output=True, text=True, encoding="utf-8", + check=False, + ) + if proc.returncode: + raise RuntimeError(f"pandoc failed: {(proc.stderr or '').strip()[:400]}") + unmapped: list[str] = [] + for line in (proc.stderr or "").splitlines(): + if line.startswith("FWHELP_UNMAPPED_SPAN"): + unmapped.extend(part.split("=", 1)[0] for part in line.split(" ", 1)[1].split(",")) + return _normalize_nested_lists(proc.stdout or ""), unmapped + + +def _normalize_nested_lists(markdown: str) -> str: + """Repair Pandoc's literal ``- -`` spelling without dropping list items.""" + return re.sub(r"^(\s*)-\s+-\s+", lambda m: m.group(1) + " - ", markdown, flags=re.MULTILINE) + + +_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") +_DRIVE = re.compile(r"^[A-Za-z]:[\\/]") +_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]") + + +def _uri_kind(raw: str) -> tuple[str, str]: + value = unquote(urldefrag(raw.strip())[0]).replace("\\", "/") + canonical = _URI_ASCII_SEPARATORS.sub("", value) + if not canonical: + return "fragment", value + if canonical.startswith("/") or _DRIVE.match(canonical): + return "path_escape", value + scheme = _SCHEME.match(canonical) + if scheme: + return ("external", value) if scheme.group(0)[:-1].casefold() in { + "http", "https", "mailto" + } else ("unsafe_uri", value) + return "local", value + + +def _check_links(links: list[str], topic_rel: str, known: set[str]) -> list[str]: + broken = [] + for href in links: + kind, value = _uri_kind(href) + if kind in {"external", "fragment", "unsafe_uri", "path_escape"}: + continue + path = value + if not path or Path(path).suffix.lower() in IMAGE_EXTS: + continue + target = _norm((Path(topic_rel).parent / path).as_posix()) + if target.casefold() not in known: + broken.append(href) + return broken + + +def _record_unsafe_uri(report: dict[str, list], rel: str, raw: str) -> None: + kind, _ = _uri_kind(raw) + if kind in {"unsafe_uri", "path_escape"}: + code = f"source_{kind}" + if not any( + str(item[0]).casefold() == rel.casefold() + and str(item[1]).casefold() == raw.casefold() + for item in report[code] + ): + report[code].append([rel, raw]) + + +def _append_reference(report: dict[str, list], code: str, rel: str, raw: str) -> None: + if not any( + str(item[0]).casefold() == rel.casefold() + and str(item[1]).casefold() == raw.casefold() + for item in report[code] + ): + report[code].append([rel, raw]) + + +def _sanitize_markdown_targets(markdown: str, report: dict[str, list], rel: str) -> str: + """Neutralize unsafe Markdown destinations while preserving safe syntax.""" + opener = re.compile(r"(?!)?\[[^\]]*\]\(") + output: list[str] = [] + cursor = 0 + for match in opener.finditer(markdown): + output.append(markdown[cursor:match.end()]) + pos, depth = match.end(), 1 + angle = pos < len(markdown) and markdown[pos] == "<" + while pos < len(markdown): + char = markdown[pos] + if angle and char == ">": + angle = False + elif not angle and char == "(": + depth += 1 + elif not angle and char == ")": + depth -= 1 + if depth == 0: + break + pos += 1 + if depth: + continue + inner = markdown[match.end():pos] + if inner.startswith("<") and ">" in inner: + target = inner[1:inner.index(">")] + suffix = inner[inner.index(">") + 1:] + replacement_target = "" + else: + pieces = inner.split(None, 1) + target = pieces[0] if pieces else "" + suffix = (" " + pieces[1]) if len(pieces) == 2 else "" + replacement_target = "TARGET" + kind, _ = _uri_kind(target) + if kind in {"unsafe_uri", "path_escape"}: + report.setdefault(kind, []).append([rel, target]) + replacement = replacement_target.replace("TARGET", "#") + suffix + output.append(replacement + ")") + else: + output.append(inner + ")") + cursor = pos + 1 + output.append(markdown[cursor:]) + return "".join(output) + + +_ATTR_TARGET = re.compile( + r"(?P\b(?:href|src)\s*=\s*)" + r"(?:(?P[\"'])(?P.*?)(?P=quote)|(?P[^\s>]+))", + re.IGNORECASE, +) + + +def _canonical_case(target: str, rel: str, canonical: dict[str, str]) -> str | None: + """Return target restated in its file's real case, or None if it already is.""" + kind, value = _uri_kind(target) + if kind != "local" or not value: + return None + resolved = _norm((Path(rel).parent / value).as_posix()) + actual = canonical.get(resolved.casefold()) + if actual is None or actual == resolved: + return None + prefix = _norm(Path(rel).parent.as_posix()) + shared = 0 + actual_parts, prefix_parts = actual.split("/"), prefix.split("/") if prefix else [] + while (shared < len(prefix_parts) and shared < len(actual_parts) - 1 + and prefix_parts[shared] == actual_parts[shared]): + shared += 1 + hops = [".."] * (len(prefix_parts) - shared) + fragment = urldefrag(target.strip())[1] + return "/".join(hops + actual_parts[shared:]) + (f"#{fragment}" if fragment else "") + + +def _canonicalize_link_case(source: str, rel: str, canonical: dict[str, str], + report: dict[str, list]) -> str: + """Restate authored links in their target's real case before conversion. + + RoboHelp resolves hrefs case-insensitively, so authored case drifts from the + topic's real path. Such a link still opens inside a CHM and 404s once the + corpus is published to a case-sensitive host, so emit the case the file + actually has and report the drift back to the author. + """ + def replace(match: re.Match[str]) -> str: + target = match.group("quoted") if match.group("quote") else match.group("bare") + # Source attributes carry HTML entities ("&") over path characters + # that the extracted filenames spell literally, so compare decoded and + # re-encode the corrected path on the way out. + fixed = _canonical_case(html.unescape(target), rel, canonical) + if fixed is None: + return match.group(0) + fixed = quote(fixed, safe="/#") + _append_reference(report, "link_case_mismatches", rel, target) + if match.group("quote"): + return match.group("prefix") + match.group("quote") + fixed + match.group("quote") + return match.group("prefix") + fixed + + return _ATTR_TARGET.sub(replace, source) + + +def _sanitize_raw_targets(markdown: str, report: dict[str, list], rel: str) -> str: + def replace(match: re.Match[str]) -> str: + target = match.group("quoted") if match.group("quote") else match.group("bare") + kind, _ = _uri_kind(target) + if kind in {"unsafe_uri", "path_escape"}: + report.setdefault(kind, []).append([rel, target]) + if match.group("quote"): + return match.group("prefix") + match.group("quote") + "#" + match.group("quote") + return match.group("prefix") + "#" + return match.group(0) + + return _ATTR_TARGET.sub(replace, markdown) + + +def _version(extraction: Path) -> str: + for path in sorted(extraction.rglob("*"), key=lambda item: item.as_posix().casefold()): + if not path.is_file() or path.suffix.lower() != ".hhk": + continue + match = re.search(r"_(\d+\.\d+)\.hhk$", path.name) + if match: + return match.group(1) + return "" + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with Path(path).open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _authenticated_manifest(path: Path, chm: Path, source_hash: str) -> bool: + """Return whether an extraction manifest authenticates this exact CHM.""" + try: + loaded = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return False + if not isinstance(loaded, dict): + return False + schema = loaded.get("schema") + return ( + isinstance(schema, int) and not isinstance(schema, bool) and schema == 1 + and loaded.get("source_name") == chm.name + and loaded.get("source_sha256") == source_hash + and isinstance(loaded.get("source_sha256"), str) + and loaded["source_sha256"] == loaded["source_sha256"].lower() + and len(loaded["source_sha256"]) == 64 + and all(char in "0123456789abcdef" for char in loaded["source_sha256"]) + ) + + +def _write_extraction_manifest(extraction: Path, chm: Path, source_hash: str) -> None: + extraction.mkdir(parents=True, exist_ok=True) + (extraction / EXTRACTION_MANIFEST).write_text( + json.dumps({ + "schema": 1, + "source_name": chm.name, + "source_sha256": source_hash, + }, indent=2) + "\n", + encoding="utf-8", + ) + + +def _topic_files(extraction: Path) -> list[Path]: + return sorted( + (path for path in extraction.rglob("*") + if path.is_file() and path.suffix.lower() in {".htm", ".html"}), + key=lambda path: ( + path.relative_to(extraction).as_posix().casefold(), + path.relative_to(extraction).as_posix(), + ), + ) + + +def _validate_conversion_destination(chm: Path, extraction: Path, destination: Path) -> None: + """Reject linked or source-overlapping destinations before any write.""" + if (link := first_link_in_path(destination)) is not None: + raise SourceSafetyError(f"refusing symlink/junction conversion destination: {link}") + if Path(destination).exists(): + validate_source_tree(destination) + destination_abs = Path(destination).resolve(strict=False) + chm_abs = Path(chm).resolve(strict=False) + extraction_abs = Path(extraction).resolve(strict=False) + for label, protected in (("CHM", chm_abs), ("extraction", extraction_abs)): + if destination_abs == protected or destination_abs in protected.parents: + raise SourceSafetyError( + f"refusing conversion destination overlapping {label}: {destination}" + ) + + +def _convert_chm_locked(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", source_url_base: str | None = None, + extractor=extract, extract_fn=None) -> dict: + """Extract and convert one CHM, returning report facts and TOC nodes.""" + chm = Path(chm) + if extract_fn is not None: + extractor = extract_fn + extraction = Path(work_root) / safe_stem(chm.name) + destination = Path(destination) + if (link := first_link_in_path(chm)) is not None: + raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}") + if (link := first_link_in_path(extraction)) is not None: + raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}") + _validate_conversion_destination(chm, extraction, destination) + source_hash = _sha256(chm) + advisory: list[str] = [] + manifest = extraction / EXTRACTION_MANIFEST + if reuse and extraction.exists(): + validate_source_tree(extraction) + if reuse and _topic_files(extraction) and _authenticated_manifest(manifest, chm, source_hash): + fatal, advisory = validate(extraction) + if fatal: + raise RuntimeError("reused extraction failed validation: " + "; ".join(fatal)) + else: + default_extractor = extractor is extract + if extractor is extract: + # ``convert_chm`` already owns extraction's lock for its entire + # read/convert lifetime; taking it again would deadlock. + extractor = _extract_already_locked + extractor(chm, extraction) + validate_source_tree(extraction) + if default_extractor: + fatal, advisory = validate(extraction) + if fatal: + raise RuntimeError( + "fresh extraction failed validation: " + "; ".join(fatal) + ) + else: + advisory = list(getattr(extractor, "advisory", [])) + # ``extract`` promotes only after its staged extraction passes its + # checks. Record identity only after that call has returned. + _write_extraction_manifest(extraction, chm, source_hash) + _validate_conversion_destination(chm, extraction, destination) + hhc = next((path for path in sorted(extraction.rglob("*"), key=lambda item: item.as_posix().casefold()) + if path.is_file() and path.suffix.lower() == ".hhc"), None) + toc = parse_toc(hhc) if hhc else [] + crumbs = { + node["href"].casefold(): node["breadcrumb"] + for node in toc if node["href"] + } + version = _version(extraction) + source_hash = "sha256:" + source_hash + all_topics = [path.relative_to(extraction).as_posix() for path in _topic_files(extraction)] + topics = list(all_topics) + known = {topic.casefold() for topic in topics} + extraction_files = { + path.relative_to(extraction).as_posix().casefold() + for path in extraction.rglob("*") if path.is_file() + } + if limit: + topics = topics[:limit] + claimed: dict[str, list[str]] = defaultdict(list) + # Links are canonicalized against the source paths, before the .htm to .md + # rewrite, so authored case is corrected once at the conversion boundary. + canonical_sources: dict[str, str] = {} + for rel in all_topics: + claimed[Path(rel).with_suffix(".md").as_posix().casefold()].append(f"topic:{rel}") + canonical_sources[rel.casefold()] = rel + for image in extraction.rglob("*"): + if image.is_file() and image.suffix.lower() in IMAGE_EXTS: + rel = image.relative_to(extraction).as_posix() + claimed[rel.casefold()].append(f"asset:{rel}") + canonical_sources[rel.casefold()] = rel + collisions = [(dest, paths) for dest, paths in sorted(claimed.items()) if len(paths) > 1] + if collisions: + report: dict[str, list] = {"destination_collisions": collisions} + return {"chm": chm.name, "stem": safe_stem(chm.name), "version": _version(extraction), + "toc": toc, "topics": 0, "images": 0, "topics_paths": [], "report": report} + destination.mkdir(parents=True, exist_ok=True) + # Keep each invocation's scratch file unique even when sibling + # destinations share a work directory. The finally block below removes + # only this converter-owned path. + tmp = destination.parent / ( + f".{safe_stem(chm.name)}-pandoc-{uuid.uuid4().hex}.html" + ) + report: dict[str, list] = defaultdict(list) + unmapped: Counter[str] = Counter() + unmapped_topics: dict[str, set[str]] = defaultdict(set) + source_replacement_paths: list[str] = [] + titles: dict[str, list[str]] = defaultdict(list) + records: dict[str, tuple[str, TopicMeta]] = {} + for rel in topics: + meta = TopicMeta() + meta.feed((extraction / rel).read_bytes().decode("cp1252", errors="replace")) + original = html.unescape(meta.title).strip() or Path(rel).stem.replace("_", " ") + records[rel] = (original, meta) + titles[original.casefold()].append(rel) + display_titles: dict[str, str] = {} + for paths in titles.values(): + if len(paths) == 1: + display_titles[paths[0]] = records[paths[0]][0] + continue + original = records[paths[0]][0] + used: set[str] = set() + for rel in paths: + _, meta = records[rel] + heading = html.unescape(meta.page_heading).strip() + candidate = heading if heading and heading.casefold() != original.casefold() else f"{original} ({Path(rel).stem.replace('_', ' ')})" + display_titles[rel] = candidate + used.add(candidate.casefold()) + if len(used) != len(paths): + report["duplicate_titles"].append([original, paths]) + written = 0 + try: + for rel in topics: + raw = (extraction / rel).read_bytes() + source = _normalize_source(raw.decode("cp1252", errors="replace")) + source = _canonicalize_link_case(source, rel, canonical_sources, report) + if "\ufffd" in source: + source_replacement_paths.append(rel) + original_title, meta = records[rel] + title = display_titles[rel] + for href in meta.links: + _record_unsafe_uri(report, rel, href) + for href in _check_links(meta.links, rel, known): + _append_reference(report, "broken_links", rel, href) + for image_href in meta.images: + _record_unsafe_uri(report, rel, image_href) + image_kind, image_path = _uri_kind(image_href) + if image_kind in {"external", "fragment", "unsafe_uri", "path_escape"}: + continue + image_target = _norm((Path(rel).parent / image_path).as_posix()) + if image_path and image_target.casefold() not in extraction_files: + _append_reference(report, "broken_images", rel, image_href) + try: + markdown, unknown = run_pandoc(source, tmp) + except RuntimeError as exc: + report["pandoc_failures"].append([rel, str(exc)]) + continue + unmapped.update(unknown) + for class_name in unknown: + unmapped_topics[class_name].add(rel) + markdown = re.sub(r"^\s*#\s+.*?\n+", "", markdown, count=1) + markdown = re.sub(r"^# ", "## ", markdown, flags=re.MULTILINE) + markdown = _normalize_nested_lists(markdown) + markdown = _sanitize_markdown_targets(markdown, report, rel) + markdown = _sanitize_raw_targets(markdown, report, rel) + breadcrumb = crumbs.get(rel.casefold()) or [part.replace("_", " ") for part in Path(rel).parent.parts] + if any("chm::" in item or item.endswith(".hhc") for item in breadcrumb): + breadcrumb = [] + if not crumbs.get(rel.casefold()): + report["not_in_toc"].append(rel) + related = [] + for label, href in meta.related: + target = _norm((Path(rel).parent / urldefrag(unquote(href))[0]).as_posix()) + if target.casefold() in known: + related.append(f"{label} -> {_relative(rel, target)}") + page_heading = html.unescape(meta.page_heading).strip() + if re.sub(r"[^a-z0-9]", "", page_heading.lower()) in { + re.sub(r"[^a-z0-9]", "", title.lower()), + re.sub(r"[^a-z0-9]", "", original_title.lower()), + }: + page_heading = "" + fields = { + "title": title, + "source_title": original_title, + "breadcrumb": breadcrumb, + "source": rel, + "source_url": ( + f"{source_url_base.rstrip('/')}/index.htm#t={quote(rel)}" + if source_url_base else None + ), + "source_hash": source_hash, + "keywords": list(dict.fromkeys(k.strip() for k in meta.meta.get("rh-index-keywords", "").split(",") if k.strip())), + "related": related, + "fw_help_version": version, + "page_heading": page_heading, + "type": "index" if "overview" in Path(rel).stem.lower() else "topic", + "content_hash": "sha256:" + hashlib.sha256(markdown.encode()).hexdigest()[:16], + } + safe_title = yaml_scalar(title)[1:-1] + trail = " › ".join(breadcrumb[:-1]) if len(breadcrumb) > 1 else "" + safe_trail = yaml_scalar(trail)[1:-1] if trail else "" + header = f"{frontmatter(fields)}\n\n# {safe_title}\n" + if trail: + header += f"\n*{safe_trail}*\n" + markdown = re.sub(r"^#{1,6} Related Topics\s*$", "## Related topics", markdown, flags=re.MULTILINE) + markdown = re.sub(r"^#{1,6} Related Internet Sites\s*$", "## Related links", markdown, flags=re.MULTILINE) + target = destination / Path(rel).with_suffix(".md") + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(header + "\n" + markdown.strip() + "\n", encoding="utf-8") + written += 1 + images = 0 + for image in extraction.rglob("*"): + if image.is_file() and image.suffix.lower() in IMAGE_EXTS: + target = destination / image.relative_to(extraction) + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(image.read_bytes()) + images += 1 + finally: + tmp.unlink(missing_ok=True) + for item in advisory: + if item.startswith(("source_unsafe_uri:", "source_path_escape:")): + code, source, raw = item.split(": ", 2) + if not any( + str(item[0]).casefold() == source.casefold() + and str(item[1]).casefold() == raw.casefold() + for item in report[code] + ): + report[code].append([source, raw]) + elif item.startswith("HTML "): + # HTML missing-target advisories are rechecked above so the + # converter emits one source_missing_link/image report only. + continue + else: + report["stale_toc_entries"].append(item) + if unmapped: + report["unmapped_span_classes"] = [ + [name, count, sorted(unmapped_topics[name], key=str.casefold)] + for name, count in unmapped.most_common() + ] + return {"chm": chm.name, "stem": safe_stem(chm.name), "version": version, + "toc": toc, "topics": written, "images": images, "report": dict(report), + "source_replacement_paths": source_replacement_paths, + "topics_paths": topics} + + +def convert_chm(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", source_url_base: str | None = None, + extractor=extract, extract_fn=None) -> dict: + """Convert one CHM with deterministic extraction-then-destination locks. + + The extraction/work-root lock is acquired first and held through all + extraction, validation, source reads, and Pandoc conversion. The + destination lock is acquired second. This order avoids lock inversion; + the internal already-locked extraction entry point prevents recursive + acquisition. + """ + chm = Path(chm) + extraction = Path(work_root) / safe_stem(chm.name) + destination = Path(destination) + # Perform pure validation first so equal/overlapping targets report the + # intended source-safety error rather than a duplicate-lock busy error. + if (link := first_link_in_path(chm)) is not None: + raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}") + if (link := first_link_in_path(extraction)) is not None: + raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}") + _validate_conversion_destination(chm, extraction, destination) + with export_locks(extraction, destination): + # The implementation repeats source and destination validation after + # lock acquisition, immediately before its first destination write. + return _convert_chm_locked( + chm, work_root, destination, reuse=reuse, limit=limit, + source_ref=source_ref, source_url_base=source_url_base, + extractor=extractor, extract_fn=extract_fn, + ) + + +def run_in_private_stage(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", + source_url_base: str | None = None, + extractor=extract, extract_fn=None) -> dict: + """Convert into a caller-owned private stage while locking extraction. + + The caller must guarantee that ``destination`` is an unshared staging + path. Such a path needs no destination lock; creating one beside it would + turn the lockfile into generated content when the enclosing stage is + promoted. + """ + chm = Path(chm) + extraction = Path(work_root) / safe_stem(chm.name) + destination = Path(destination) + if (link := first_link_in_path(chm)) is not None: + raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}") + if (link := first_link_in_path(extraction)) is not None: + raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}") + _validate_conversion_destination(chm, extraction, destination) + with export_locks(extraction): + _validate_conversion_destination(chm, extraction, destination) + return _convert_chm_locked( + chm, work_root, destination, reuse=reuse, limit=limit, + source_ref=source_ref, source_url_base=source_url_base, + extractor=extractor, extract_fn=extract_fn, + ) + + +__all__ = [ + "convert_chm", "frontmatter", "run_in_private_stage", "run_pandoc", "yaml_scalar", +] diff --git a/tools/chm_extract.py b/tools/chm_extract.py new file mode 100644 index 00000000..8fc11e37 --- /dev/null +++ b/tools/chm_extract.py @@ -0,0 +1,328 @@ +"""Cross-platform CHM extraction. + +Tries, in order: + 1. 7z / 7za / 7zz (Linux + Windows, best choice for CI) + 2. extract_chmLib (Linux, chmlib package) + 3. hh.exe -decompile (Windows only, ships with the OS) +""" + +from __future__ import annotations + +import os +import re +import shutil +import subprocess +import sys +import time +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urldefrag + +from output_fs import ExportLock, OutputPathError, OutputStaging +from source_safety import first_link_in_path, validate_source_tree + + +class ExtractError(RuntimeError): + pass + + +def _validate_extract_paths(chm: Path, outdir: Path) -> tuple[Path, Path]: + """Reject extraction destinations that could consume their source.""" + chm = Path(os.path.abspath(os.fspath(Path(chm).expanduser()))) + outdir = Path(os.path.abspath(os.fspath(Path(outdir).expanduser()))) + if (link := first_link_in_path(chm)) is not None: + raise ExtractError(f"refusing symlink/junction CHM path component: {link}") + if (link := first_link_in_path(outdir)) is not None: + raise OutputPathError(f"refusing symlink/junction extraction destination: {link}") + try: + source = chm.resolve(strict=True) + except OSError as exc: + raise ExtractError(f"no such CHM: {chm}") from exc + destination = outdir.resolve(strict=False) + if not source.is_file(): + raise ExtractError(f"no such CHM: {chm}") + if destination == source or destination in source.parents: + raise OutputPathError( + "refusing extraction destination that contains the source CHM: " + f"source={source}, destination={destination}" + ) + # Preserve the lexical destination so OutputStaging can independently + # enforce its own path-chain ownership checks before creating staging. + return chm, outdir + + +def _sevenzip(chm: Path, outdir: Path) -> str | None: + for exe in ("7z", "7za", "7zz"): + found = shutil.which(exe) + if not found: + continue + subprocess.run( + [found, "x", "-y", f"-o{outdir}", str(chm)], + check=True, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + return exe + return None + + +def _chmlib(chm: Path, outdir: Path) -> str | None: + found = shutil.which("extract_chmLib") + if not found: + return None + subprocess.run( + [found, str(chm), str(outdir)], + check=True, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + return "extract_chmLib" + + +# The deepest path inside FieldWorks_Language_Explorer_Help.chm is ~140 chars. +# hh.exe silently TRUNCATES any output path that exceeds Windows MAX_PATH (260) +# -- no error, no non-zero exit, just a file named "Foo_field_(Extended_Note)" +# with the ".htm" chopped off. Refuse to run rather than corrupt the corpus. +MAX_PATH = 260 +ASSUMED_MAX_INTERNAL = 160 + + +def _hh(chm: Path, outdir: Path) -> str | None: + if os.name != "nt": + return None + found = shutil.which("hh") or r"C:\Windows\hh.exe" + if not Path(found).exists(): + return None + + budget = MAX_PATH - ASSUMED_MAX_INTERNAL + if len(str(outdir.resolve())) > budget: + raise ExtractError( + "hh.exe would silently truncate filenames: output path is " + f"{len(str(outdir.resolve()))} chars, must be <= {budget}.\n" + f" {outdir.resolve()}\n" + " Use a shorter --work directory (e.g. C:\\fwhelps-work), or " + "install 7-Zip, which has no such limit." + ) + # hh.exe detaches immediately and wants native separators + absolute paths. + subprocess.run( + [str(found), "-decompile", str(outdir.resolve()), str(chm.resolve())], + check=True, + ) + # It returns before the write finishes; wait for the file count to settle. + stable, last = 0, -1 + for _ in range(60): + time.sleep(0.5) + count = sum(1 for _ in outdir.rglob("*") if _.is_file()) + stable = stable + 1 if count == last and count > 0 else 0 + if stable >= 4: + break + last = count + return "hh.exe" + + +EXTRACTION_MANIFEST_NAME = ".chm-extraction-manifest.json" + + +class _ReferenceParser(HTMLParser): + """Collect CHM sitemap locals and HTML href/src references.""" + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.local: list[str] = [] + self.html: list[tuple[str, str]] = [] + + def handle_starttag(self, tag: str, attrs) -> None: + values = {key.casefold(): value or "" for key, value in attrs} + if (tag.casefold() == "param" + and values.get("name", "").casefold() == "local" + and values.get("value")): + self.local.append(values["value"]) + for attr in ("href", "src"): + if values.get(attr): + self.html.append((attr, values[attr])) + + +_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") +_DRIVE = re.compile(r"^[A-Za-z]:[\\/]") +_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]") + + +def _reference_kind(raw: str) -> tuple[str, str]: + """Return (kind, path), classifying URI safety before filesystem access.""" + value = unquote(urldefrag(raw.strip())[0]).replace("\\", "/") + canonical = _URI_ASCII_SEPARATORS.sub("", value) + if not canonical: + return "fragment", value + # A leading double slash is a UNC path in CHM content. It is never a + # permitted external URL because only explicitly allowed schemes are + # accepted by the exporter. + if canonical.startswith(("/", "//")) or _DRIVE.match(canonical): + return "path_escape", value + scheme = _SCHEME.match(canonical) + if scheme: + return ("external", value) if scheme.group(0)[:-1].casefold() in { + "http", "https", "mailto" + } else ("unsafe_uri", value) + return "local", value + + +def _prefix_sibling(target: Path, expected: str) -> bool: + if not target.parent.is_dir(): + return False + expected_folded = expected.casefold() + return any( + sibling.is_file() + and sibling.name.casefold() != expected_folded + and expected_folded.startswith(sibling.name.casefold()) + for sibling in target.parent.iterdir() + ) + + +def validate(outdir: Path) -> tuple[list[str], list[str]]: + """Check an extraction, separating tool failure from content bugs. + + Returns (fatal, advisory). + + `fatal` means the extractor lost data -- a truncated corpus yields a subtly + wrong RAG index, which is worse than no index. `advisory` means the CHM + itself is inconsistent, e.g. the author renamed a topic and left a stale + TOC entry behind. That is real, but it is the doc author's to fix and must + not block a build. + + Distinguishing them: hh.exe truncation leaves a sibling whose name is a + prefix of the expected one ("Discussion_field_(Extended_Note)" for + "...(Extended_Note).htm"). A genuinely absent topic leaves no such trace. + """ + fatal, advisory = [], [] + + parsers: list[tuple[Path, _ReferenceParser]] = [] + for path in sorted(outdir.rglob("*"), key=lambda p: p.relative_to(outdir).as_posix().casefold()): + if not path.is_file() or path.name == EXTRACTION_MANIFEST_NAME: + continue + if path.suffix.casefold() not in {".hhc", ".htm", ".html"}: + continue + parser = _ReferenceParser() + parser.feed(path.read_text(encoding="cp1252", errors="replace")) + parsers.append((path, parser)) + + if not any(path.suffix.casefold() == ".hhc" for path, _ in parsers): + fatal.append("no .hhc table of contents found in extraction") + return fatal, advisory + + seen: set[tuple[str, str, str]] = set() + for source, parser in parsers: + references = [("toc", value) for value in parser.local] + parser.html + for kind, raw in references: + uri_kind, value = _reference_kind(raw) + if uri_kind == "fragment" or uri_kind == "external": + continue + if uri_kind in {"unsafe_uri", "path_escape"}: + key = (uri_kind, source.as_posix(), raw) + if key not in seen: + advisory.append( + f"source_{uri_kind}: {source.relative_to(outdir).as_posix()}: {raw}" + ) + seen.add(key) + continue + target = (source.parent / value).resolve() + if target != outdir.resolve() and outdir.resolve() not in target.parents: + advisory.append( + f"source_path_escape: {source.relative_to(outdir).as_posix()}: {raw}" + ) + continue + if target.exists(): + continue + if _prefix_sibling(target, target.name): + message = f"{('TOC target' if kind == 'toc' else 'HTML target')} lost to filename truncation: {value}" + fatal.append(message) + elif kind == "toc": + advisory.append(f"TOC points at a topic that does not exist: {value}") + else: + advisory.append(f"HTML {kind} points at a target that does not exist: {value}") + + return fatal, advisory + + +def _extract_locked(chm: Path, outdir: Path, clean: bool = True, check: bool = True) -> str: + """Extract `chm` into `outdir`. Returns the name of the tool that worked.""" + chm = Path(chm) + outdir = Path(outdir) + chm, outdir = _validate_extract_paths(chm, outdir) + # Keep failed attempts in private directories beside the destination. The + # staging owner guarantees that an existing destination is untouched until + # a validated extraction is promoted, and same-directory staging makes the + # final rename safe even when the system temporary directory is another + # volume. + errors: list[str] = [] + extract.advisory = [] + for backend in (_sevenzip, _chmlib, _hh): + try: + with OutputStaging(outdir) as staging: + tool = backend(chm, staging.path) + if not tool: + continue + validate_source_tree(staging.path) + if not any( + path.is_file() and path.suffix.lower() in {".htm", ".html"} + for path in staging.rglob("*") + ): + errors.append(f"{tool}: produced no .htm files") + continue + advisory: list[str] = [] + if check: + fatal, advisory = validate(staging.path) + if fatal: + errors.append( + f"{tool} produced a corrupt/incomplete extraction " + f"({len(fatal)} problems): " + "; ".join(fatal) + ) + continue + # Content-level inconsistencies ride along for the author + # report rather than failing the build. + if clean or not outdir.exists(): + staging.promote() + else: + # Retain the historical clean=False merge semantics, + # but construct the merged tree privately so a copy + # failure cannot partially modify the destination. + validate_source_tree(outdir) + with OutputStaging(outdir) as merged: + shutil.copytree(outdir, merged.path, dirs_exist_ok=True) + shutil.copytree(staging.path, merged.path, dirs_exist_ok=True) + merged.promote() + extract.advisory = advisory + return tool + except subprocess.CalledProcessError as exc: + errors.append(f"{backend.__name__}: exit {exc.returncode}") + continue + + raise ExtractError( + "no working CHM extractor found.\n" + " Linux: apt-get install p7zip-full (or libchm-bin)\n" + r" Windows: 7-Zip, or the built-in C:\Windows\hh.exe" "\n" + + (" tried: " + "; ".join(errors) if errors else "") + ) + + +def _extract_already_locked( + chm: Path, outdir: Path, clean: bool = True, check: bool = True +) -> str: + """Internal extraction entry point for callers holding ``outdir``'s lock.""" + return _extract_locked(chm, outdir, clean=clean, check=check) + + +def extract(chm: Path, outdir: Path, clean: bool = True, check: bool = True) -> str: + """Extract into ``outdir`` while serializing overlapping exporters.""" + chm, outdir = _validate_extract_paths(Path(chm), Path(outdir)) + with ExportLock(outdir): + # Revalidate after acquiring the lock to close the preflight-to-write + # path/link window for cooperating invocations. + chm, outdir = _validate_extract_paths(chm, outdir) + return _extract_locked(chm, outdir, clean=clean, check=check) + + +if __name__ == "__main__": + tool = extract(Path(sys.argv[1]), Path(sys.argv[2])) + print(f"extracted with {tool}") + for note in getattr(extract, "advisory", []): + print(f" advisory: {note}") diff --git a/tools/chm_metadata.py b/tools/chm_metadata.py new file mode 100644 index 00000000..c3d5e839 --- /dev/null +++ b/tools/chm_metadata.py @@ -0,0 +1,127 @@ +"""Small, repository-independent CHM metadata helpers.""" + +from __future__ import annotations + +import html +import re +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urldefrag + + +def safe_stem(name: str) -> str: + """Return a deterministic namespace name for a CHM filename.""" + stem = Path(name).stem + value = re.sub(r"[^A-Za-z0-9._-]+", "_", stem).strip("._-") + return value or "chm" + + +class TopicMeta(HTMLParser): + """Facts needed by conversion and corpus provenance checks.""" + + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.title = "" + self.meta: dict[str, str] = {} + self.links: list[str] = [] + self.images: list[str] = [] + self.related: list[tuple[str, str]] = [] + self.page_heading = "" + self._in_title = False + self._heading: list[str] | None = None + self._section = "" + self._anchor: tuple[str, list[str]] | None = None + + def handle_starttag(self, tag: str, attrs) -> None: + a = {k.lower(): (v or "") for k, v in attrs} + if tag == "title": + self._in_title = True + elif tag == "meta": + key = a.get("name") or a.get("http-equiv") + if key: + self.meta[key.lower()] = a.get("content", "") + elif re.fullmatch(r"h[1-6]", tag): + self._heading = [] + elif tag == "img" and a.get("src"): + self.images.append(a["src"]) + elif tag == "a" and a.get("href"): + href = a["href"] + self.links.append(href) + self._anchor = (href, []) + + def handle_endtag(self, tag: str) -> None: + if tag == "title": + self._in_title = False + elif re.fullmatch(r"h[1-6]", tag) and self._heading is not None: + text = "".join(self._heading).strip() + if text and not self.page_heading: + self.page_heading = text + if text in {"Related Topics", "Related Internet Sites"}: + self._section = text + elif text: + self._section = "" + self._heading = None + elif tag == "a" and self._anchor is not None: + href, label = self._anchor + if self._section == "Related Topics": + self.related.append(("".join(label).strip(), href)) + self._anchor = None + + def handle_data(self, data: str) -> None: + if self._in_title: + self.title += data + if self._heading is not None: + self._heading.append(data) + if self._anchor is not None: + self._anchor[1].append(data) + + +class _SitemapParser(HTMLParser): + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.depth = 0 + self.entries: list[dict] = [] + self._current: list[tuple[str, str]] | None = None + + def handle_starttag(self, tag: str, attrs) -> None: + a = {k.lower(): (v or "") for k, v in attrs} + if tag == "ul": + self.depth += 1 + elif tag == "object": + self._current = [] + elif tag == "param" and self._current is not None: + self._current.append((a.get("name", "").lower(), a.get("value", ""))) + + def handle_endtag(self, tag: str) -> None: + if tag == "ul": + self.depth = max(0, self.depth - 1) + elif tag == "object" and self._current: + self.entries.append({"depth": self.depth, "params": self._current}) + self._current = None + + +def parse_toc(path: Path) -> list[dict]: + parser = _SitemapParser() + parser.feed(path.read_text(encoding="cp1252", errors="replace")) + nodes, trail = [], {} + for entry in parser.entries: + params = dict(entry["params"]) + title = html.unescape(params.get("name", "")).strip() + local = unquote(html.unescape(params.get("local", ""))).replace("\\", "/").strip() + href = urldefrag(local)[0] if local else "" + depth = entry["depth"] + trail[depth] = title + for d in list(trail): + if d > depth: + del trail[d] + nodes.append({ + "title": title, + "href": href, + "depth": depth, + "breadcrumb": [trail[d] for d in sorted(trail) if trail[d]], + "is_container": not bool(href), + }) + return nodes + + +__all__ = ["TopicMeta", "parse_toc", "safe_stem"] diff --git a/tools/convert.py b/tools/convert.py new file mode 100644 index 00000000..8764d600 --- /dev/null +++ b/tools/convert.py @@ -0,0 +1,395 @@ +"""Thin orchestration seam for the CHM/PDF Markdown export.""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import tempfile +from pathlib import Path +from urllib.parse import quote + +import pdf_convert +from chm_convert import run_in_private_stage +from corpus_validation import validate_corpus +from output_fs import ExportLock, OutputPathError, OutputStaging, validate_output_paths +from reporting import Issue, Report, make_issue +from source_safety import discover_source_files + +DEFAULT_SOURCE_REPO = "https://github.com/sillsdev/FwHelps" +DEFAULT_CHM_SOURCE_URL_BASE = "https://downloads.languagetechnology.org/fieldworks/Documentation/en" + + +def discover_chms(repo: Path) -> list[Path]: + """Discover all CHMs at the repository root in stable case-folded order.""" + return discover_source_files(Path(repo), suffixes={".chm"}, recursive=False) + + +def _report_issue(code: str, item: object, default_path: str = ""): + if code == "unmapped_span_classes" and isinstance(item, (list, tuple)): + class_name = str(item[0]) if item else "" + count = item[1] if len(item) > 1 else 0 + topics = [str(path) for path in item[2]] if len(item) > 2 else [] + path = topics[0] if topics else default_path + detail = {"class": class_name, "count": count, "topics": topics} + message = f"span class '{class_name}' reported {count} time(s)" + return make_issue(code, message, path, detail) + if code == "duplicate_titles" and isinstance(item, (list, tuple)): + title = str(item[0]) if item else "" + raw_topics = item[1] if len(item) > 1 else [] + topics = [str(path) for path in raw_topics] if isinstance(raw_topics, list) else [str(raw_topics)] + path = topics[0] if topics else default_path + detail = {"title": title, "topics": topics} + message = f"duplicate title '{title}' appears in {len(topics)} topics" + return make_issue(code, message, path, detail) + path = item[0] if isinstance(item, (list, tuple)) and item else str(item) + message = item[1] if isinstance(item, (list, tuple)) and len(item) > 1 else str(item) + return make_issue(code, str(message), str(path or default_path), item) + + +def _write_readme(stage: Path, chms: list[dict], pdf_count: int, + source_ref: str, report: Report) -> None: + inventory = ( + "- **Root CHMs (auto-discovered):** " + + ", ".join(f"`{item['chm']}`" for item in chms) + if chms else "- **Root CHMs (auto-discovered):** none found" + ) + lines = [ + "# FieldWorks Help — Portable Markdown", "", + "Generated documentation corpus. **Do not edit these files**; the tree is replaced as a set.", "", + f"- **Source ref:** `{source_ref}`", f"- **CHMs:** {len(chms)} **PDFs:** {pdf_count}", "", + inventory, + "", + report.to_readme(), "", + ( + "Full detail: [author-report.md](author-report.md) for authors; " + "[author-report.json](author-report.json) for automation." + ), "", + "## CHM navigation", "", + ] + for result in chms: + lines.append(f"### {result['chm']}") + known_topics = { + str(item).replace("\\", "/").casefold() + for item in result.get("topics_paths", []) + } + for node in result.get("toc", []): + if not node.get("title"): + continue + indent = " " * max(0, int(node.get("depth", 1)) - 1) + href = node.get("href", "") + topic_path = Path(href.split("#", 1)[0]).as_posix() + if href and topic_path.casefold() in known_topics: + target = quote(f"chm/{result['stem']}/{Path(href).with_suffix('.md').as_posix()}") + lines.append(f"{indent}- [{node['title']}]({target})") + else: + lines.append(f"{indent}- **{node['title']}**") + lines.extend(["", "## PDF navigation", ""]) + pdf_root = stage / "pdf" + for path in sorted(pdf_root.rglob("*.md")) if pdf_root.exists() else []: + rel = path.relative_to(stage).as_posix() + lines.append(f"- [{path.stem.replace('_', ' ')}]({quote(rel)})") + (stage / "README.md").write_text("\n".join(lines) + "\n", encoding="utf-8") + (stage / ".nojekyll").write_text("", encoding="utf-8") + + +def _build_locked(repo: Path, out: Path, work: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", + source_repo: str = DEFAULT_SOURCE_REPO, + chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE) -> dict: + """Build a complete corpus in private staging and promote only if valid.""" + repo, out, work = Path(repo).resolve(), Path(out).resolve(), Path(work).resolve() + report = Report() + advisory_links: set[tuple[str, str]] = set() + advisory_images: set[tuple[str, str]] = set() + source_replacement_paths: set[str] = set() + chm_results: list[dict] = [] + chms = discover_chms(repo) + if not chms: + report.add(make_issue("chm_discovery", "no repository-root CHM files found")) + + # Validate the extraction workspace independently. The publication stage + # must live beside the destination so promotion is one same-volume rename + # even when --work is on another drive. + validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo) + with OutputStaging(out, repo_root=repo, source_root=repo) as staging: + seen_names: set[str] = set() + for chm in chms: + from chm_metadata import safe_stem + stem = safe_stem(chm.name) + if stem.casefold() in seen_names: + report.add(make_issue("destination_collision", f"CHM namespace collision: {stem}", chm.name)) + continue + seen_names.add(stem.casefold()) + try: + result = run_in_private_stage(chm, work, staging.path / "chm" / stem, + reuse=reuse, limit=limit, source_ref=source_ref, + source_url_base=chm_source_url_base) + chm_results.append(result) + stem = result.get("stem", "") + for source_rel, raw_target in result.get("report", {}).get("broken_links", []): + output_rel = f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}" + advisory_links.add((output_rel, raw_target)) + if "#" in raw_target: + source_target, fragment = raw_target.split("#", 1) + else: + source_target, fragment = raw_target, "" + if re.search(r"\.html?$", source_target, re.IGNORECASE): + source_target = re.sub(r"\.html?$", ".md", source_target, flags=re.IGNORECASE) + emitted_target = source_target + (f"#{fragment}" if fragment else "") + advisory_links.add((output_rel, emitted_target)) + advisory_links.add((output_rel, quote(emitted_target, safe="/#()%"))) + for source_rel, raw_target in result.get("report", {}).get("broken_images", []): + output_rel = f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}" + advisory_images.add((output_rel, raw_target)) + advisory_images.add((output_rel, quote(raw_target, safe="/#()%"))) + for source_rel in result.get("source_replacement_paths", []): + source_replacement_paths.add( + f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}" + ) + for code, items in result.get("report", {}).items(): + for item in items: + report.add(_report_issue(code, item, chm.name)) + except Exception as exc: # noqa: BLE001 - isolate one corrupt CHM + report.add(make_issue("chm_failure", f"{type(exc).__name__}: {exc}", chm.name)) + + pdf_url = f"{source_repo.rstrip('/')}/blob/{source_ref}/{{path}}" + try: + pdf_result, _ = pdf_convert.run_in_private_stage( + repo, staging.path / "pdf", update=False, source_url=pdf_url, + ) + except Exception as exc: # noqa: BLE001 - isolate PDF backend failure + pdf_result = {"converted": 0, "report": { + "pdf_failures": [["", f"{type(exc).__name__}: {exc}"]] + }} + pdf_report = pdf_result.get("report", {}) + pdf_export_paths = { + str(item[0]) for item in pdf_report.get("pdf_export_replacements", []) + if isinstance(item, (list, tuple)) and item + } + for item in pdf_report.get("pdf_source_replacements", []): + if not isinstance(item, (list, tuple)) or len(item) < 2: + continue + source_rel, details = item[0], item[1] + emitted = Path(pdf_convert.slug_path(str(source_rel))).with_suffix(".md").as_posix() + emitted_rel = f"pdf/{emitted}" + # A page reporting both source and exporter replacements must not + # be allowlisted: validator evidence must keep the exporter part + # fatal even though the source risk is also reported. + if str(source_rel) not in pdf_export_paths: + source_replacement_paths.add(emitted_rel) + report.add(make_issue( + "source_replacement_character", + f"source PDF replacement characters: {details}", + str(source_rel), { + "source_pdf": str(source_rel), + "generated_markdown": emitted_rel, + "pages": details, + }, + )) + for code, items in pdf_report.items(): + if code == "pdf_source_replacements": + continue + for item in items: + report.add(_report_issue(code, item)) + report.metadata = { + "source_ref": source_ref, + "source_repo": source_repo.rstrip("/"), + "chms": [ + {"name": item.get("chm", ""), "stem": item.get("stem", ""), + "version": item.get("version", ""), "topics": item.get("topics", 0), + "images": item.get("images", 0)} + for item in chm_results + ], + "chm_count": len(chm_results), + "topic_count": sum(item.get("topics", 0) for item in chm_results), + "image_count": sum(item.get("images", 0) for item in chm_results), + "pdf_count": pdf_result.get("converted", 0), + } + # README navigation is part of the emitted corpus. Seed the linked + # report target, render navigation, then validate all links before the + # final report/count rewrite. + (staging.path / "author-report.json").write_text("{}\n", encoding="utf-8") + (staging.path / "author-report.md").write_text( + "# Author quality report\n\nBuild validation is in progress.\n", + encoding="utf-8", + ) + _write_readme(staging.path, chm_results, pdf_result.get("converted", 0), source_ref, report) + report.extend(validate_corpus( + staging.path, + advisory_links=advisory_links, + advisory_images=advisory_images, + source_replacement_paths=source_replacement_paths, + )) + _write_readme(staging.path, chm_results, pdf_result.get("converted", 0), source_ref, report) + (staging.path / "author-report.json").write_text( + report.to_json(), encoding="utf-8" + ) + (staging.path / "author-report.md").write_text( + report.to_markdown(), encoding="utf-8" + ) + if report.fatal: + return {"report": report.as_dict(), "chms": chm_results, + "pdfs": pdf_result.get("converted", 0), "promoted": False} + staging.promote() + return {"report": report.as_dict(), "chms": chm_results, + "pdfs": pdf_result.get("converted", 0), "promoted": True} + + +def _build(repo: Path, out: Path, work: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", + source_repo: str = DEFAULT_SOURCE_REPO, + chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE) -> dict: + """Serialize one complete corpus mutation for cooperating exporters.""" + repo, out, work = Path(repo).resolve(), Path(out).resolve(), Path(work).resolve() + validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo) + with ExportLock(out): + # The destination/link chain is checked again after lock acquisition, + # immediately before staging and eventual promotion begin. + validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo) + return _build_locked( + repo, out, work, reuse=reuse, limit=limit, source_ref=source_ref, + source_repo=source_repo, chm_source_url_base=chm_source_url_base, + ) + + +def _path_overlaps(left: Path, right: Path) -> bool: + return left == right or left in right.parents or right in left.parents + + +def _validate_diagnostics_path( + diagnostics: Path | str, + *, + repo: Path, + out: Path, + work: Path, +) -> Path: + """Validate an external diagnostics destination before creating anything.""" + lexical = Path(os.path.abspath(os.fspath(Path(diagnostics).expanduser()))) + resolved = lexical.resolve(strict=False) + if resolved == Path(resolved.anchor): + raise OutputPathError(f"refusing filesystem root as diagnostics: {resolved}") + if lexical.is_symlink() or (lexical.exists() and not lexical.is_file()): + raise OutputPathError(f"refusing non-regular diagnostics path: {lexical}") + # Existing symlinked parents can redirect an apparently external path into + # a protected tree. Missing parents are created only after this check. + current = lexical.parent + while current != Path(current.anchor): + if current.is_symlink() or ( + current.exists() and getattr(current, "is_junction", lambda: False)() + ): + raise OutputPathError(f"refusing symlink/junction in diagnostics parent: {current}") + current = current.parent + protected = { + "repository": Path(repo).resolve(strict=False), + "output": Path(out).resolve(strict=False), + "work": Path(work).resolve(strict=False), + } + for label, root in protected.items(): + if _path_overlaps(resolved, root): + raise OutputPathError(f"diagnostics must not overlap {label}: {resolved}") + return resolved + + +def _sanitize_diagnostic_value(value: object, roots: dict[str, Path]) -> object: + """Remove run-specific absolute paths while preserving report structure.""" + if isinstance(value, dict): + return {key: _sanitize_diagnostic_value(item, roots) for key, item in value.items()} + if isinstance(value, list): + return [_sanitize_diagnostic_value(item, roots) for item in value] + if isinstance(value, tuple): + return [_sanitize_diagnostic_value(item, roots) for item in value] + if not isinstance(value, str): + return value + sanitized = value + for label, root in sorted(roots.items(), key=lambda item: len(str(item[1])), reverse=True): + for spelling in (str(root), str(root).replace("\\", "/")): + sanitized = sanitized.replace(spelling, f"<{label}>") + sanitized = re.sub(r"\.output-(?:stage|backup)-[0-9A-Za-z-]+", ".output-", sanitized) + return sanitized + + +def _write_diagnostics(path: Path, report: dict, *, roots: dict[str, Path]) -> None: + """Atomically write a stable report without exposing staging filenames.""" + path.parent.mkdir(parents=True, exist_ok=True) + stable_report = _sanitize_diagnostic_value(report, roots) + payload = json.dumps(stable_report, indent=2, ensure_ascii=False, sort_keys=True) + "\n" + fd, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent) + temporary_path = Path(temporary) + try: + with os.fdopen(fd, "w", encoding="utf-8", newline="\n") as handle: + handle.write(payload) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temporary_path, path) + finally: + if temporary_path.exists(): + temporary_path.unlink() + + +def build(repo: Path, out: Path, work: Path, *, reuse: bool = False, + limit: int = 0, source_ref: str = "develop", + source_repo: str = DEFAULT_SOURCE_REPO, + chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE, + diagnostics: Path | str | None = None) -> dict: + """Build a corpus and optionally persist diagnostics outside generated trees.""" + repo_path, out_path, work_path = ( + Path(repo).resolve(), Path(out).resolve(), Path(work).resolve() + ) + diagnostics_path = ( + _validate_diagnostics_path( + diagnostics, repo=repo_path, out=out_path, work=work_path + ) + if diagnostics is not None else None + ) + try: + result = _build( + repo_path, out_path, work_path, reuse=reuse, limit=limit, + source_ref=source_ref, source_repo=source_repo, + chm_source_url_base=chm_source_url_base, + ) + except Exception as exc: + if diagnostics_path is not None: + _write_diagnostics( + diagnostics_path, + Report([make_issue("unknown_issue", f"{type(exc).__name__}: {exc}")]).as_dict(), + roots={"repo": repo_path, "out": out_path, "work": work_path}, + ) + raise + if diagnostics_path is not None: + _write_diagnostics( + diagnostics_path, + result["report"], + roots={"repo": repo_path, "out": out_path, "work": work_path}, + ) + return result + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--repo", default=".", type=Path) + parser.add_argument("--out", default="out", type=Path) + parser.add_argument("--work", default=None, type=Path) + parser.add_argument("--reuse", action="store_true") + parser.add_argument("--limit", type=int, default=0) + parser.add_argument("--source-ref", default="develop") + parser.add_argument("--source-repo", default=DEFAULT_SOURCE_REPO) + parser.add_argument("--chm-source-url-base", default=DEFAULT_CHM_SOURCE_URL_BASE) + parser.add_argument("--diagnostics", default=None, type=Path) + args = parser.parse_args() + repo = args.repo.resolve() + work = (args.work or repo / ".chm-work").resolve() + result = build(repo, args.out.resolve(), work, reuse=args.reuse, + limit=args.limit, source_ref=args.source_ref, + source_repo=args.source_repo, + chm_source_url_base=args.chm_source_url_base, + diagnostics=args.diagnostics) + print(Report(Issue(**{key: value for key, value in item.items() + if key in {"code", "message", "path", "fatal", "provenance", "detail"}}) + for item in result["report"].get("issues", [])).to_console()) + return 1 if result["report"]["summary"]["fatal"] else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/corpus_validation.py b/tools/corpus_validation.py new file mode 100644 index 00000000..b6117eee --- /dev/null +++ b/tools/corpus_validation.py @@ -0,0 +1,260 @@ +"""Validate the files actually emitted by the portable Markdown exporter.""" + +from __future__ import annotations + +import os +import re +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urldefrag + +from reporting import Issue, make_issue + +_EXTERNAL = re.compile(r"^(?:[a-z][a-z0-9+.-]*:|//)", re.IGNORECASE) +_SCHEME = re.compile(r"^[a-z][a-z0-9+.-]*:", re.IGNORECASE) +_DRIVE = re.compile(r"^[a-z]:[\\/]") +_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]") +_FENCE = re.compile(r"^\s*(```|~~~)") + + +class _RawHTML(HTMLParser): + def __init__(self) -> None: + super().__init__(convert_charrefs=True) + self.targets: list[tuple[str, str]] = [] + self.tags: list[str] = [] + + def handle_starttag(self, tag: str, attrs) -> None: + self.tags.append(tag.lower()) + values = {key.lower(): value or "" for key, value in attrs} + if tag.lower() == "a" and values.get("href"): + self.targets.append(("href", values["href"])) + if tag.lower() == "img" and values.get("src"): + self.targets.append(("src", values["src"])) + + +def _frontmatter(text: str) -> dict[str, str]: + if not text.startswith("---"): + return {} + end = text.find("\n---", 3) + if end < 0: + return {} + fields = {} + for line in text[4:end].splitlines(): + if ":" in line and not line.startswith(" "): + key, value = line.split(":", 1) + fields[key.strip()] = value.strip().strip('"') + return fields + + +def _target_path(raw: str) -> str: + return urldefrag(unquote(raw.replace("\\", "/")))[0] + + +def _is_external(raw: str) -> bool: + value = _target_path(raw) + if not value or value.startswith("#"): + return True + scheme = _SCHEME.match(value) + return bool(scheme and scheme.group(0)[:-1].casefold() in {"http", "https", "mailto"}) + + +def _uri_kind(raw: str) -> str: + value = _target_path(raw).replace("\\", "/") + canonical = _URI_ASCII_SEPARATORS.sub("", value) + if not canonical or canonical.startswith("#"): + return "fragment" + if canonical.startswith("/") or _DRIVE.match(canonical): + return "path_escape" + scheme = _SCHEME.match(canonical) + if scheme: + return "external" if scheme.group(0)[:-1].casefold() in { + "http", "https", "mailto" + } else "unsafe_uri" + return "local" + + +def _resolve(root: Path, source: Path, raw: str) -> Path | None: + path = _target_path(raw) + if _uri_kind(raw) != "local" or not path: + return None + # Absolute paths are never local corpus references. + if path.startswith(("/", "\\")): + return None + # Normalize ".." lexically rather than with Path.resolve(): on Windows + # resolve() rewrites the path to the real on-disk case, so a link whose + # case does not match its target would look valid here and 404 on the + # case-sensitive host the corpus is published to. + return Path(os.path.normpath(source.parent / path)) + + +def _inside_root(root: Path, candidate: Path) -> bool: + return candidate == root or root in candidate.parents + + +def _corpus_entries(root: Path) -> set[str]: + """Return every corpus path, in the exact case it was emitted with.""" + return {item.relative_to(root).as_posix() for item in root.rglob("*")} + + +def _exists(root: Path, entries: set[str], candidate: Path) -> bool: + """Return whether candidate exists, matching case on every platform.""" + if candidate == root: + return True + try: + relative = candidate.relative_to(root).as_posix() + except ValueError: + return False + return relative in entries + + +def _markdown_targets(text: str): + """Yield (is_image, target) while balancing parentheses in destinations.""" + start = re.compile(r"(?!)?\[[^\]]*\]\(") + for match in start.finditer(text): + cursor, depth = match.end(), 1 + angle = cursor < len(text) and text[cursor] == "<" + while cursor < len(text): + char = text[cursor] + if angle and char == ">": + angle = False + elif not angle and char == "(": + depth += 1 + elif not angle and char == ")": + depth -= 1 + if depth == 0: + break + cursor += 1 + if depth: + continue + value = text[match.end():cursor].strip() + if value.startswith("<") and ">" in value: + value = value[1:value.find(">")] + else: + value = value.split(None, 1)[0] if value else "" + yield bool(match.group("image")), value + + +def _without_fenced(text: str) -> str: + lines, fenced = [], False + for line in text.splitlines(): + if _FENCE.match(line): + fenced = not fenced + continue + if not fenced: + lines.append(line) + return "\n".join(lines) + + +def validate_corpus(root: Path, *, advisory_links: set[tuple[str, str]] | None = None, + source_replacement_paths: set[str] | None = None, + source_links: set[tuple[str, str]] | None = None, + advisory_images: set[tuple[str, str]] | None = None) -> list[Issue]: + """Return issues for a corpus tree; never mutate the emitted files.""" + root = Path(root).resolve() + advisory_links = advisory_links or source_links or set() + advisory_images = advisory_images or set() + source_replacement_paths = source_replacement_paths or set() + issues: list[Issue] = [] + entries = _corpus_entries(root) + markdown = sorted(root.rglob("*.md")) + titles: dict[str, list[str]] = {} + for path in markdown: + rel = path.relative_to(root).as_posix() + text = path.read_text(encoding="utf-8", errors="replace") + body = text + if text.startswith("---") and "\n---" in text[3:]: + body = text[text.find("\n---", 3) + 4:] + link_body = _without_fenced(body) + h1s = [] + fenced = False + for line in body.splitlines(): + if _FENCE.match(line): + fenced = not fenced + continue + if fenced: + continue + heading = re.match(r"^\s*(#{1,6})\s+(.+?)\s*#*\s*$", line) + if heading and len(heading.group(1)) == 1: + h1s.append(heading.group(2).strip()) + if len(h1s) != 1: + issues.append(make_issue("one_h1", f"expected one H1, found {len(h1s)}", rel)) + if "\ufffd" in text: + source = rel in source_replacement_paths + issues.append(make_issue( + "source_replacement_character" if source else "replacement_character", + "contains U+FFFD", rel, + )) + if re.search(r"(?m)^\s*-\s+-\s+", link_body): + issues.append(make_issue("malformed_list", "literal '- -' list marker", rel)) + if h1s: + titles.setdefault(h1s[0].casefold(), []).append(rel) + + # Parse Markdown links separately so image links do not produce a + # second ordinary-link finding. + for is_image, raw in _markdown_targets(link_body): + uri_kind = _uri_kind(raw) + if uri_kind in {"unsafe_uri", "path_escape"}: + issues.append(make_issue( + uri_kind, f"unsafe target: {raw}", rel + )) + continue + target = _resolve(root, path, raw) + if target is not None: + inside = _inside_root(root, target) + if not inside or not _exists(root, entries, target): + # Source-authored missing targets are visible but advisory; + # an escape is always an exporter safety error. + advisory = inside and ( + (not is_image and (rel, raw) in advisory_links) + or (is_image and (rel, raw) in advisory_images) + ) + code = (("source_missing_image" if is_image else "source_missing_link") + if advisory else ("missing_image" if is_image else "missing_link")) + issues.append(make_issue( + code, + f"target {'escapes corpus root' if not inside else 'does not exist'}: {raw}", + rel, + )) + + raw_html = _RawHTML() + raw_html.feed(link_body) + if any(tag not in {"a", "img"} for tag in raw_html.tags): + issues.append(make_issue( + "raw_html", f"raw HTML tags: {', '.join(sorted(set(raw_html.tags)))}", + rel, + )) + for kind, raw in raw_html.targets: + uri_kind = _uri_kind(raw) + if uri_kind in {"unsafe_uri", "path_escape"}: + issues.append(make_issue( + uri_kind, f"unsafe target: {raw}", rel + )) + continue + target = _resolve(root, path, raw) + if target is not None and (not _exists(root, entries, target) + or not _inside_root(root, target)): + inside = _inside_root(root, target) + advisory = inside and ( + (kind == "href" and (rel, raw) in advisory_links) + or (kind == "src" and (rel, raw) in advisory_images) + ) + code = (("source_missing_image" if kind == "src" else "source_missing_link") + if advisory else ("missing_image" if kind == "src" else "missing_link")) + issues.append(make_issue( + code, + f"raw HTML target {'escapes corpus root' if not inside else 'does not exist'}: {raw}", + rel, + )) + + for title, paths in sorted(titles.items()): + if len(paths) > 1: + issues.append(make_issue( + "duplicate_title", f"display title {title!r}: {', '.join(paths)}", + paths[0], paths, + )) + return issues + + +validate_emitted_corpus = validate_corpus + +__all__ = ["validate_corpus", "validate_emitted_corpus"] diff --git a/tools/frontmatter.py b/tools/frontmatter.py new file mode 100644 index 00000000..d9d0e157 --- /dev/null +++ b/tools/frontmatter.py @@ -0,0 +1,13 @@ +"""Shared helpers for safely rendering YAML front matter values.""" + +from __future__ import annotations + +import json + + +def yaml_scalar(value: object) -> str: + """Render a scalar as a JSON-quoted YAML-compatible string.""" + return json.dumps(str(value), ensure_ascii=False) + + +__all__ = ["yaml_scalar"] diff --git a/tools/fwhelp.lua b/tools/fwhelp.lua new file mode 100644 index 00000000..8dc7ee2b --- /dev/null +++ b/tools/fwhelp.lua @@ -0,0 +1,285 @@ +-- Pandoc Lua filter: RoboHelp XHTML -> clean GFM for humans and RAG. +-- +-- Runs inside pandoc (Lua 5.4 is bundled; no separate install). Everything here +-- needs the document tree, which is why it lives in Lua rather than Python. +-- +-- Unmapped span classes are collected and reported so the build can fail loudly +-- rather than silently dropping a semantic distinction we did not know about. + +local unmapped = {} + +-- RoboHelp's authored character styles. Anything not listed is a build error. +local SPAN_MAP = { + UserInterface = "strong", -- 18708x: menu/button/field names + Strong = "strong", + Emphasis = "emph", + DefinedWord = "emph", + BookTitle = "emph", + VernacularWord = "emph", -- example-language data, NOT English prose + TypedText = "code", -- literal text the user types + Keyboard = "code", -- key names + FileName = "code", + Filename = "code", -- authoring typo, same intent + Placeholder = "emph", + -- RoboHelp table presentation classes carry no document semantics. + hcp1 = "plain", + hcp2 = "plain", + hcp3 = "plain", + hcp4 = "plain", + Superscript = "superscript", + nobr = "plain", + expandtext = "plain", + ["Strong\""] = "strong", -- malformed class attribute in source +} + +-- h4 callout classes -> GitHub alert blocks +local ALERT_MAP = { + Note = "NOTE", Tip = "TIP", Important = "IMPORTANT", + Warning = "WARNING", Caution = "CAUTION", +} + +local function uri_kind(target) + local path = tostring(target or "") + path = path:gsub("%%([0-9A-Fa-f][0-9A-Fa-f])", function(hex) + return string.char(tonumber(hex, 16)) + end) + path = path:gsub("\\", "/") + path = path:gsub("[%z\001-\032]", "") + path = path:gsub("#.*$", "") + if path == "" then return "fragment" end + if path:sub(1, 1) == "/" or path:match("^[A-Za-z]:/") then return "path_escape" end + local scheme = path:match("^([A-Za-z][A-Za-z0-9+%%.-]*):") + if scheme then + scheme = scheme:lower() + if scheme == "http" or scheme == "https" or scheme == "mailto" then return "external" end + return "unsafe_uri" + end + return "local" +end + +local function neutralize_uri(target) + local kind = uri_kind(target) + if kind == "unsafe_uri" or kind == "path_escape" then + io.stderr:write("FWHELP_" .. kind:upper() .. " " .. tostring(target) .. "\n") + return "#" + end + return target +end + +local function text_of(inlines) + -- gsub returns (string, count); returning it directly would pass the count + -- as pandoc.Code's second argument, which it reads as an Attr. + local s = pandoc.utils.stringify(inlines) + s = s:gsub("^%s+", "") + s = s:gsub("%s+$", "") + return s +end + +--- Collapse authored character styles into real markdown emphasis. +function Span(el) + if #el.classes == 0 then return el.content end + -- Inventory every authored class before applying a supported transform. + -- A span may carry both a supported class and an unknown semantic. + for _, cls in ipairs(el.classes) do + if SPAN_MAP[cls] == nil then + unmapped[cls] = (unmapped[cls] or 0) + 1 + end + end + for _, cls in ipairs(el.classes) do + local kind = SPAN_MAP[cls] + if kind == "strong" then return pandoc.Strong(el.content) + elseif kind == "emph" then return pandoc.Emph(el.content) + elseif kind == "code" then return pandoc.Code(text_of(el.content)) + elseif kind == "superscript" then return pandoc.Superscript(el.content) + elseif kind == "plain" then return el.content + end + end + return el.content +end + +--- Repoint internal topic links at their .md counterparts. +--- Done here rather than with a regex over the rendered markdown because +--- filenames like "Export_full_lexicon_(LIFT).htm" contain parentheses, which +--- no sane link regex survives. The AST has the target as a plain string. +function Link(el) + local t = el.target + local kind = uri_kind(t) + if kind == "unsafe_uri" or kind == "path_escape" then + el.target = neutralize_uri(t) + return el + end + if kind == "fragment" or kind == "external" then return el end + local path, frag = t:match("^([^#]*)(.*)$") + local rewritten, n = path:gsub("%.html?$", ".md") + if n > 0 then el.target = rewritten .. frag end + return el +end + +--- Drop the presentational attributes RoboHelp puts on every image. +--- GFM cannot express width/height/style, so pandoc falls back to a raw +--- tag and 2,109 of them were surviving into the markdown across 842 files. +--- The sizes are RoboHelp's inline-icon dimensions, not information. +function Image(el) + -- Decorative RoboHelp marker on Tip/Note headings. Empty-alt images + -- stringify as U+FFFD in alert detection and GFM output. + local src = tostring(el.src or "") + if src == "" and el.target then src = tostring(el.target) end + if src:lower():match("note[_-]?icon%.gif") then + return pandoc.List() + end + el.src = neutralize_uri(src) + el.attr = pandoc.Attr() + return el +end + +--- Collect every row across head/bodies/foot as a flat list. +local function all_rows(el) + local rows = pandoc.List() + for _, r in ipairs(el.head.rows) do rows:insert(r) end + for _, b in ipairs(el.bodies) do + for _, r in ipairs(b.head) do rows:insert(r) end + for _, r in ipairs(b.body) do rows:insert(r) end + end + for _, r in ipairs(el.foot.rows) do rows:insert(r) end + return rows +end + +--- Is this a 2-column "label:/value" layout rather than real tabular data? +--- 560 of 769 tables in the corpus are these -- "Full name:", "Location:", +--- "Description:", "Field type:" -- i.e. definition lists that RoboHelp +--- happened to render with . 79% of tables carry block content in a +--- cell, so pandoc must fall back to raw HTML for them; turning them into +--- prose removes almost all remaining HTML from the output. +local function is_definition_table(rows) + if #rows == 0 then return false end + local labelish = 0 + for _, row in ipairs(rows) do + if #row.cells ~= 2 then return false end + local label = text_of(row.cells[1].contents) + if label:sub(-1) == ":" and #label < 40 then labelish = labelish + 1 end + end + return labelish / #rows >= 0.8 +end + +function Table(el) + local rows = all_rows(el) + + if is_definition_table(rows) then + local out = pandoc.List() + for _, row in ipairs(rows) do + local label = text_of(row.cells[1].contents):gsub(":%s*$", "") + local value = row.cells[2].contents + local head = pandoc.List({ pandoc.Strong(pandoc.Str(label .. ":")) }) + -- Fold a single-paragraph value onto the label line; otherwise keep the + -- label on its own line and let lists/multiple paragraphs follow. + if #value == 1 and value[1].t == "Para" then + head:insert(pandoc.Space()) + head:extend(value[1].content) + out:insert(pandoc.Para(head)) + else + out:insert(pandoc.Para(head)) + out:extend(value) + end + end + return out + end + + -- Genuine data table: drop RoboHelp's fixed column widths so it emits as a + -- GFM pipe table instead of a grid table (11,670 inline width: styles). + for i, spec in ipairs(el.colspecs) do + el.colspecs[i] = { spec[1], nil } + end + return el +end + +--- Strip presentational wrappers pandoc lifts from RoboHelp's
/. +function Div(el) + -- el.attributes is an AttributeList, not a plain table, so `next` fails on it. + if #el.classes == 0 and el.identifier == "" then return el.content end + return el +end + +--- Every topic opens with an

title, which Python promotes to the page's +--- single h1. Shift body headings up to match, leaving the 4 stray h1s alone. +function Header(el) + if el.level > 1 then el.level = el.level - 1 end + return el +end + +local function alert_kind(block) + if block.t ~= "Header" then return nil end + for _, cls in ipairs(block.classes) do + if ALERT_MAP[cls] then return ALERT_MAP[cls] end + end + -- Some callouts carry the word as heading text with no class. + local t = text_of(block.content):gsub("^%s*[^%w]*%s*", "") + return ALERT_MAP[t] +end + +--- Wrap Note/Tip/Important/Warning/Caution callouts in GitHub alert +--- blockquotes. This has to happen here rather than in Header() because the +--- callout body is the *following* blocks, not the heading's children. +function Blocks(blocks) + local out = pandoc.List() + local i = 1 + while i <= #blocks do + local b = blocks[i] + + -- The "Related Topics" / "Related Internet Sites" trailers stay where the + -- author put them. Reconstructing them from link labels alone lost the + -- prose between the links -- "Lists overview (task helps)" became "Lists + -- overview", and "Choose a translation type (in an Example Lexicon Edit)" + -- lost both its qualifier and its second link. Python only normalises the + -- heading level afterwards. + local kind = alert_kind(b) + if kind then + -- Raw, not Str: pandoc would escape the brackets to "\[!NOTE\]", which + -- GitHub no longer recognises as an alert. + local marker = pandoc.RawInline("gfm", "[!" .. kind .. "]") + local body = pandoc.List({ pandoc.Para({ marker }) }) + local level = b.level + i = i + 1 + while i <= #blocks and not (blocks[i].t == "Header" and blocks[i].level <= level) do + body:insert(blocks[i]) + i = i + 1 + end + out:insert(pandoc.BlockQuote(body)) + goto continue + end + + -- A handful of RoboHelp list fragments arrive as literal text beginning + -- "- -" instead of a nested list node. Reparse only that malformed + -- marker through Pandoc's Markdown reader so the item remains content but + -- is emitted as a real nested list (never a literal marker). + if b.t == "Para" and text_of(b.content):match("^%s*%-%s+%-%s+") then + local repaired = pandoc.read(text_of(b.content), "markdown") + for _, item in ipairs(repaired.blocks) do out:insert(item) end + else + out:insert(b) + end + i = i + 1 + ::continue:: + end + return out +end + +--- Emit unmapped classes on stderr for the build to pick up. +function Pandoc(doc) + local names = {} + for cls, n in pairs(unmapped) do names[#names + 1] = cls .. "=" .. n end + if #names > 0 then + io.stderr:write("FWHELP_UNMAPPED_SPAN " .. table.concat(names, ",") .. "\n") + end + return doc +end + +-- Only the functions named here run: returning an explicit filter list opts out +-- of pandoc's pick-up-every-global behaviour, so a handler left off this table +-- is silently dead code. +-- Span/Table/Div/Header run before Blocks so trailers are detected on clean text. +return { + { Span = Span, Table = Table, Div = Div, Header = Header, Link = Link, + Image = Image }, + { Blocks = Blocks }, + { Pandoc = Pandoc }, +} diff --git a/tools/issue_catalog.py b/tools/issue_catalog.py new file mode 100644 index 00000000..620234ce --- /dev/null +++ b/tools/issue_catalog.py @@ -0,0 +1,110 @@ +"""The single issue vocabulary shared by every exporter stage.""" + +from __future__ import annotations + +from dataclasses import dataclass +from types import MappingProxyType + + +@dataclass(frozen=True) +class IssuePolicy: + label: str + fatal: bool + provenance: str + guidance: str = "" + + +def _policy(label: str, fatal: bool, provenance: str, guidance: str) -> IssuePolicy: + return IssuePolicy(label, fatal, provenance, guidance) + + +ISSUE_CATALOG = MappingProxyType({ + "missing_link": _policy("Missing local link", True, "exporter", + "Do not edit RoboHelp yet. Inspect the generated source path and target, then fix the exporter so a valid authored link remains valid."), + "source_missing_link": _policy("Missing local link", False, "source", + "Open the source topic in RoboHelp, find the hyperlink named in Evidence, and retarget or remove it; then rebuild the CHM."), + "missing_image": _policy("Missing local image", True, "exporter", + "Do not edit RoboHelp yet. Verify the source image exists, then fix exporter copying or link rewriting for the reported generated path."), + "source_missing_image": _policy("Missing local image", False, "source", + "Open the source topic in RoboHelp, find the image reference named in Evidence, and restore, retarget, or remove it; then rebuild the CHM."), + "source_link_case": _policy("Link case mismatch", False, "source", + "RoboHelp resolves links case-insensitively, so this one opens in the CHM but would 404 on a case-sensitive host. The export publishes the corrected case; open the source topic in RoboHelp and retype the hyperlink to match the target topic's real path so the two stay in step."), + "duplicate_title": _policy("Duplicate display title", False, "source", + "Open the listed topics in RoboHelp and give each page a distinct, descriptive title or heading so search results identify the correct page."), + "malformed_list": _policy("Malformed nested list", True, "exporter", + "Inspect the reported generated page and its source topic, then correct list conversion while preserving the authored nesting."), + "replacement_character": _policy("Replacement character", True, "exporter", + "Compare the generated page with its CHM/PDF source and fix decoding or conversion where the replacement character was introduced."), + "source_replacement_character": _policy("Replacement character", False, "source", + "Open the reported source topic or PDF at the page in Evidence and replace the invalid or unsupported source character."), + "raw_html": _policy("Raw HTML retained", False, "source", + "Inspect the reported RoboHelp topic or PDF content and the tag names in Problem. Simplify unsupported source markup when practical; otherwise confirm the retained HTML is intentional."), + "one_h1": _policy("Invalid H1 count", True, "exporter", + "Compare the generated page headings with the source topic and fix title normalization so the Markdown has exactly one H1."), + "destination_collision": _policy("Destination collision", True, "exporter", + "Rename colliding source files/topics or adjust deterministic destination naming so every source maps to one unique output path."), + "unsafe_uri": _policy("Unsafe URI", True, "exporter", + "Inspect the generated path and URI in Evidence, then fix sanitization so unsafe source targets cannot be emitted as active links."), + "path_escape": _policy("Local path escape", True, "exporter", + "Inspect the generated link or image target and fix path normalization so it cannot resolve outside the published corpus."), + "source_unsafe_uri": _policy("Source unsafe URI", False, "source", + "Open the source topic in RoboHelp, find the URI in Evidence, and replace malformed, file:, or script-like targets with a valid safe link or remove them."), + "source_path_escape": _policy("Source local path escape", False, "source", + "Open the source topic in RoboHelp and retarget the local link or image so it stays inside the help project."), + "pandoc_failure": _policy("Pandoc conversion failure", True, "exporter", + "Reproduce conversion for the reported source topic, inspect the Pandoc error in Problem, and fix the exporter or unsupported source markup."), + "unmapped_span": _policy("Unmapped span class", True, "exporter", + "Find the reported CSS class in RoboHelp source, decide its intended semantics, and add an explicit exporter mapping before publishing."), + "pdf_failure": _policy("PDF conversion failure", True, "exporter", + "Open the reported PDF to confirm it is readable, then reproduce the error in Problem and fix the PDF conversion path."), + "outline_drift": _policy("PDF outline drift", True, "exporter", + "Review the reported PDF headings against the document, then either fix heading inference or deliberately repin the approved outline."), + "outline_unpinned": _policy("PDF outline unpinned", True, "exporter", + "Review the generated headings for the reported PDF and deliberately add its approved outline to pdf_outlines.json."), + "stale_toc_entries": _policy("Stale TOC entry", False, "source", + "Open the RoboHelp table of contents, locate the target in Evidence, and retarget or remove the entry before rebuilding the CHM."), + "not_in_toc": _policy("Topic missing from TOC", False, "source", + "Find the source topic path in RoboHelp and either add it to the appropriate table-of-contents location or remove the orphan topic."), + "chm_failure": _policy("CHM conversion failure", True, "exporter", + "Verify the named CHM opens normally, then reproduce the extraction/conversion error in Problem and correct the failing backend or source package."), + "chm_discovery": _policy("CHM discovery failure", True, "exporter", + "Place the intended CHM files at the repository root or correct discovery configuration, then rerun the exporter."), + "unknown_issue": _policy("Unknown issue", True, "exporter", + "Add the producer code shown in Problem to the canonical issue catalog with explicit severity, provenance, and repair guidance."), +}) + + +# Producer report keys are deliberately aliases, so their severity and +# provenance are still selected by ISSUE_CATALOG rather than local mappings. +ISSUE_ALIASES = MappingProxyType({ + "broken_links": "source_missing_link", + "broken_images": "source_missing_image", + "duplicate_titles": "duplicate_title", + "link_case_mismatches": "source_link_case", + "pandoc_failures": "pandoc_failure", + "unmapped_span_classes": "unmapped_span", + "destination_collisions": "destination_collision", + "source_unsafe_uris": "source_unsafe_uri", + "source_path_escapes": "source_path_escape", + "pdf_failures": "pdf_failure", + "html_tables_kept": "raw_html", + "pdf_export_replacements": "replacement_character", + "pdf_source_replacements": "source_replacement_character", +}) + + +def canonical_code(code: str) -> str: + """Return the catalog code for a producer report or issue code.""" + return ISSUE_ALIASES.get(code, code) + + +def policy_for(code: str) -> tuple[str, IssuePolicy]: + """Return a safe catalog code and policy, including unknown integration codes.""" + canonical = canonical_code(str(code)) + policy = ISSUE_CATALOG.get(canonical) + if policy is None: + return "unknown_issue", ISSUE_CATALOG["unknown_issue"] + return canonical, policy + + +__all__ = ["ISSUE_ALIASES", "ISSUE_CATALOG", "IssuePolicy", "canonical_code", "policy_for"] diff --git a/tools/output_fs.py b/tools/output_fs.py new file mode 100644 index 00000000..6c4ffbd4 --- /dev/null +++ b/tools/output_fs.py @@ -0,0 +1,349 @@ +"""Owned staging and conservative filesystem promotion for generated trees. + +The public surface deliberately has no general-purpose delete operation. A +``OutputStaging`` instance may remove only its own temporary tree and the +destination tree that it validated before promotion. +""" + +from __future__ import annotations + +import errno +import hashlib +import os +import shutil +import tempfile +import uuid +from collections.abc import Iterator +from contextlib import contextmanager +from pathlib import Path +from typing import Self + +if os.name == "nt": + import msvcrt +else: + import fcntl + + +class OutputPathError(ValueError): + """A requested generated path is unsafe or internally inconsistent.""" + + +class ExportBusyError(OutputPathError): + """Another cooperating exporter currently owns the destination lock.""" + + +def _lexical_absolute(path: Path | str) -> Path: + """Make an absolute, normalized path without following links.""" + return Path(os.path.abspath(os.fspath(Path(path).expanduser()))) + + +def _absolute(path: Path | str) -> Path: + return _lexical_absolute(path).resolve(strict=False) + + +def _first_link(path: Path) -> Path | None: + """Return the first symlink/junction in a lexical path, if any.""" + current = Path(path.anchor) + for component in path.parts[1:]: + current /= component + is_junction = getattr(current, "is_junction", lambda: False) + if current.is_symlink() or is_junction(): + return current + return None + + +def _is_root(path: Path) -> bool: + return path == Path(path.anchor) + + +def _overlaps(left: Path, right: Path) -> bool: + return left == right or left in right.parents or right in left.parents + + +def validate_output_paths( + destination: Path | str, + *, + work_dir: Path | str | None = None, + repo_root: Path | str | None = None, + source_root: Path | str | None = None, +) -> tuple[Path, Path]: + """Validate generated paths before staging or recursive removal. + + Repository and source roots are protected exact paths. Their children are + valid outputs (the normal CLI writes ``repo/out``), while trying to replace + either root itself is rejected. ``work_dir`` and ``destination`` may not + overlap because either relationship could make a promotion consume its + own input tree. + """ + + destination_lexical = _lexical_absolute(destination) + destination_path = destination_lexical.resolve(strict=False) + explicit_work = work_dir is not None + work_lexical = _lexical_absolute(work_dir) if explicit_work else destination_lexical.parent + work_path = work_lexical.resolve(strict=False) + protected = { + label: (_lexical_absolute(value), _absolute(value)) + for label, value in (("repository", repo_root), ("source", source_root)) + if value is not None + } + + for label, lexical, path in ( + ("destination", destination_lexical, destination_path), + ("work", work_lexical, work_path), + ): + if _is_root(path): + raise OutputPathError(f"refusing filesystem root as {label}: {path}") + if (link := _first_link(lexical)) is not None: + raise OutputPathError(f"refusing symlink/junction in {label}: {link}") + + for label, (protected_lexical, protected_path) in protected.items(): + if _is_root(protected_path): + raise OutputPathError(f"refusing filesystem root as {label} root: {protected_path}") + if (link := _first_link(protected_lexical)) is not None: + raise OutputPathError(f"refusing symlink/junction in {label} root: {link}") + if destination_path == protected_path: + raise OutputPathError(f"destination must not replace {label} root: {destination_path}") + if explicit_work and work_path == protected_path: + raise OutputPathError(f"work directory must not be {label} root: {work_path}") + + if explicit_work and _overlaps(destination_path, work_path): + raise OutputPathError( + "work and output paths must not overlap: " + f"work={work_path}, output={destination_path}" + ) + return destination_path, work_path + + +class ExportLock: + """A non-blocking, cooperative process lock for one output destination. + + The lock is advisory: it serializes exporter invocations that use this + class, but cannot stop a hostile process that ignores OS file locks. The + lock file is a deterministic sibling derived from the normalized + destination path. It is intentionally never deleted, so an abandoned + (stale) lock file does not block a later acquisition. + + ``ExportLock`` is non-reentrant. Callers should acquire it once at their + mutation boundary and pass through to lower-level helpers without taking + another lock for the same destination. + """ + + _LOCK_PREFIX = ".fwhelps-export-" + _LOCK_SUFFIX = ".lock" + + def __init__(self, destination: Path | str) -> None: + self.destination = _lexical_absolute(destination) + self.lock_path = self._lock_path(self.destination) + self._fd: int | None = None + + @classmethod + def _lock_path(cls, destination: Path) -> Path: + normalized = os.path.normcase(os.path.normpath(os.fspath(destination))) + digest = hashlib.sha256(os.fsencode(normalized)).hexdigest() + return destination.parent / f"{cls._LOCK_PREFIX}{digest}{cls._LOCK_SUFFIX}" + + @staticmethod + def _validate_chain(destination: Path, lock_path: Path) -> None: + if _is_root(destination): + raise OutputPathError(f"refusing filesystem root as export destination: {destination}") + if (link := _first_link(destination)) is not None: + raise OutputPathError( + f"refusing symlink/junction in export destination: {link}" + ) + + # Missing destination parents are safe to create only when every + # existing ancestor is a real directory and has no link component. + current = destination.parent + while current != Path(current.anchor): + if (link := _first_link(current)) is not None: + raise OutputPathError(f"refusing symlink/junction in export lock parent: {link}") + if current.exists() and not current.is_dir(): + raise OutputPathError(f"export lock parent is not a directory: {current}") + current = current.parent + + if (link := _first_link(lock_path)) is not None: + raise OutputPathError(f"refusing symlink/junction in export lock path: {link}") + if lock_path.exists() and not lock_path.is_file(): + raise OutputPathError(f"export lock path is not a regular file: {lock_path}") + + def acquire(self) -> Self: + """Acquire this destination lock without waiting for another owner.""" + if self._fd is not None: + raise RuntimeError("export lock is already held") + self._validate_chain(self.destination, self.lock_path) + self.lock_path.parent.mkdir(parents=True, exist_ok=True) + # Revalidate after creating missing parents, before opening the lock. + self._validate_chain(self.destination, self.lock_path) + flags = os.O_CREAT | os.O_RDWR + if hasattr(os, "O_CLOEXEC"): + flags |= os.O_CLOEXEC + if hasattr(os, "O_NOFOLLOW"): + flags |= os.O_NOFOLLOW + try: + fd = os.open(self.lock_path, flags, 0o600) + except OSError as exc: + raise OutputPathError(f"cannot open export lock {self.lock_path}: {exc}") from exc + + try: + if os.name == "nt": + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_NBLCK, 1) + else: + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError as exc: + os.close(fd) + if exc.errno in {errno.EACCES, errno.EAGAIN, errno.EDEADLK}: + raise ExportBusyError( + f"export destination is busy: {self.destination} " + f"(lock: {self.lock_path})" + ) from exc + raise OutputPathError( + f"cannot acquire export lock {self.lock_path}: {exc}" + ) from exc + self._fd = fd + return self + + def release(self) -> None: + """Release the OS lock and close this instance's handle.""" + if self._fd is None: + return + fd, self._fd = self._fd, None + try: + try: + if os.name == "nt": + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_UNLCK, 1) + else: + fcntl.flock(fd, fcntl.LOCK_UN) + finally: + os.close(fd) + except OSError: + # The handle is closed and no longer owned even if the platform + # reports an unlock error; do not mask the caller's exception. + pass + + def __enter__(self) -> Self: + return self.acquire() + + def __exit__(self, exc_type, exc_value, traceback) -> bool: + self.release() + return False + + +@contextmanager +def export_locks(*destinations: Path | str) -> Iterator[list[ExportLock]]: + """Acquire distinct destination locks and always unwind partial success. + + Targets are deduplicated by normalized lexical path and acquired in a + stable platform-aware canonical absolute-path order, independent of caller + order. If a later lock is busy, all earlier locks are released before the + error escapes. + """ + unique: dict[str, ExportLock] = {} + for destination in destinations: + lock = ExportLock(destination) + key = os.path.normcase(os.path.normpath(os.fspath(lock.destination))) + unique.setdefault(key, lock) + locks = [unique[key] for key in sorted(unique)] + acquired: list[ExportLock] = [] + try: + for lock in locks: + lock.acquire() + acquired.append(lock) + yield locks + finally: + for lock in reversed(acquired): + lock.release() + + +class OutputStaging: + """Own a temporary generated tree and promote it as one destination. + + ``with OutputStaging(...) as stage`` yields this object; use ``stage.path`` + or path-like operations to populate it, then call ``stage.promote()`` only + after the caller's validation succeeds. Exiting without promotion removes + only the temporary tree and leaves an existing destination untouched. + ``OutputStaging`` does not acquire an ``ExportLock`` itself; callers own one + non-reentrant lock for the complete mutation boundary. + """ + + _STAGE_PREFIX = ".output-stage-" + _BACKUP_PREFIX = ".output-backup-" + + def __init__( + self, + destination: Path | str, + *, + work_dir: Path | str | None = None, + repo_root: Path | str | None = None, + source_root: Path | str | None = None, + ) -> None: + self.destination, self.work_dir = validate_output_paths( + destination, + work_dir=work_dir, + repo_root=repo_root, + source_root=source_root, + ) + self.work_dir.mkdir(parents=True, exist_ok=True) + self.path = Path(tempfile.mkdtemp(prefix=self._STAGE_PREFIX, dir=self.work_dir)) + self._promoted = False + + def __enter__(self) -> Self: + return self + + def __exit__(self, exc_type, exc_value, traceback) -> bool: + if not self._promoted: + self._remove_owned(self.path, self._STAGE_PREFIX) + return False + + def __fspath__(self) -> str: + return os.fspath(self.path) + + def __truediv__(self, child: str) -> Path: + return self.path / child + + def iterdir(self): + return self.path.iterdir() + + def rglob(self, pattern: str): + return self.path.rglob(pattern) + + def promote(self) -> Path: + """Replace the validated destination with the staged tree.""" + if self._promoted: + raise RuntimeError("staging tree has already been promoted") + if not self.path.is_dir() or self.path.parent != self.work_dir: + raise OutputPathError("staging tree is no longer owned by this instance") + + self.destination.parent.mkdir(parents=True, exist_ok=True) + backup: Path | None = None + if self.destination.exists() or self.destination.is_symlink(): + if self.destination.is_symlink(): + raise OutputPathError(f"refusing symlink destination: {self.destination}") + backup = self.destination.parent / f"{self._BACKUP_PREFIX}{uuid.uuid4().hex}" + os.replace(self.destination, backup) + try: + os.replace(self.path, self.destination) + except Exception: + if backup is not None and not self.destination.exists(): + os.replace(backup, self.destination) + raise + if backup is not None: + self._remove_owned(backup, self._BACKUP_PREFIX) + self._promoted = True + return self.destination + + @staticmethod + def _remove_owned(path: Path, prefix: str) -> None: + """Remove a path only when it is a child with our ownership prefix.""" + if path.name.startswith(prefix) and path.parent != path: + if path.is_dir() and not path.is_symlink(): + shutil.rmtree(path) + elif path.exists() or path.is_symlink(): + path.unlink() + + +__all__ = [ + "ExportBusyError", "ExportLock", "OutputPathError", "OutputStaging", + "export_locks", "validate_output_paths", +] diff --git a/tools/pdf_convert.py b/tools/pdf_convert.py new file mode 100644 index 00000000..337d7463 --- /dev/null +++ b/tools/pdf_convert.py @@ -0,0 +1,1318 @@ +"""Convert the FwHelps PDFs into markdown. + +Thirteen PDFs, 349 pages: technical notes, utility documentation, and the +Conceptual Introduction. All have real text layers, so no OCR is involved. + +Heading structure comes from one of two places: + + * 7 PDFs carry bookmarks, which give exact headings and levels. + * 6 -- the Word-produced "Technical Notes" family -- carry none, so headings + are inferred from font size and weight. + +Inference is pinned. The expected outline of every PDF is recorded in +pdf_outlines.json and re-checked on each build: these files change roughly once +a decade (9 of 13 have exactly one commit in the repo's history), so drift +almost always means the inference broke rather than the document changing. +A mismatch fails the build instead of quietly shipping a mis-structured corpus. + +Usage: + python tools/pdf_convert.py --repo . --out export [--update-outlines] +""" + +from __future__ import annotations + +import argparse +import collections +import hashlib +import json +import os +import re +import shutil +import subprocess +import tempfile +import unicodedata +from collections.abc import Mapping +from pathlib import Path +from urllib.parse import quote, urlsplit, urlunsplit + +import pymupdf as fitz # "fitz" name is deprecated; alias keeps call sites short +import pymupdf4llm +from frontmatter import yaml_scalar +from output_fs import export_locks +from pymupdf4llm.helpers.pymupdf_rag import TocHeaders +from reporting import Report, make_issue +from source_safety import discover_source_files, first_link_in_path + +OUTLINES = Path(__file__).parent / "pdf_outlines.json" +_WINDOWS_RESERVED_BASENAMES = { + "CON", "PRN", "AUX", "NUL", + *(f"COM{i}" for i in range(1, 10)), + *(f"LPT{i}" for i in range(1, 10)), +} + + +def _slug_component(part: str) -> str: + value = part.replace(" ", "_") + value = re.sub(r'[<>:"|?*\\]', "_", value) + while value.endswith((".", " ")): + value = value[:-1] + "_" + if value.split(".", 1)[0].rstrip(" .").upper() in _WINDOWS_RESERVED_BASENAMES: + value = "_" + value + return value or "_" + + +def slug_path(rel: str) -> str: + """Space-free output path. + + pymupdf4llm rewrites spaces to underscores when it derives image filenames + from image_path, so a directory created as "Technical Notes_images" is not + the one it writes into. Underscores throughout also keep the URLs clean and + match the CHM side of the export, which RoboHelp already names that way. + """ + return "/".join(_slug_component(part) for part in rel.split("/")) + + +def clean(text: str) -> str: + """Normalise the non-breaking spaces Word leaves in headings and bookmarks.""" + text = unicodedata.normalize("NFKC", text) + return re.sub(r"\s+", " ", text).strip() + + +class FontHeaders: + """Infer heading levels from font size and weight. + + The corpus's bookmark-less PDFs are Word documents with a consistent style: + 12pt regular body, 18pt bold H1, 14pt bold H2. Crucially, 12pt *bold* is + used for inline emphasis ("Note:", "Tip:") and must not become a heading -- + so size alone is not enough, and bold alone is not enough either. A span has + to be both bold and meaningfully larger than the body text. + """ + + BOLD = 1 << 4 + + def __init__(self, doc: fitz.Document, min_lines: int = 2): + sizes: collections.Counter = collections.Counter() + bold_sizes: collections.Counter = collections.Counter() + + for page in doc: + for block in page.get_text("dict")["blocks"]: + for line in block.get("lines", []): + spans = line.get("spans", []) + if not spans: + continue + text = "".join(s["text"] for s in spans).strip() + if not text: + continue + span = spans[0] + size = round(span["size"], 1) + sizes[size] += 1 + if span["flags"] & self.BOLD: + bold_sizes[size] += 1 + + self.body = sizes.most_common(1)[0][0] if sizes else 12.0 + # Candidate heading sizes: bold, bigger than the body, and used often + # enough to be a real style rather than a one-off. + candidates = sorted( + (s for s, n in bold_sizes.items() if s > self.body + 0.4 and n >= min_lines), + reverse=True, + ) + self.levels = {size: i + 1 for i, size in enumerate(candidates[:6])} + + def get_header_id(self, span: dict, page=None) -> str: + if not (span["flags"] & self.BOLD): + return "" + level = self.levels.get(round(span["size"], 1)) + return "#" * level + " " if level else "" + + +def running_margins(doc: fitz.Document, threshold: float = 0.6) -> tuple[float, float]: + """Find the top/bottom bands occupied by running headers and footers. + + Every Word-produced PDF here repeats a header ("2 Getting started ... 4") + and footer ("Technical Notes on ...doc Edited on 8/13/2026") on each page; + left in, they appear in the markdown once per page. + + Measured per document, not globally: silewp2007_002.pdf carries real body + text where the others put a footer, so a blanket margin would silently eat + content. A band only counts as furniture if it recurs on most pages. + """ + if not doc.page_count: + return 0.0, 0.0 + height = doc[0].rect.height + top_ys: list[float] = [] + bottom_ys: list[float] = [] + top_pages: set[int] = set() + bottom_pages: set[int] = set() + + for i, page in enumerate(doc): + for block in page.get_text("dict")["blocks"]: + y0, y1 = block["bbox"][1], block["bbox"][3] + text = "".join( + s["text"] for line in block.get("lines", []) for s in line["spans"] + ).strip() + if not text: + continue + if y1 < height * 0.12: + top_pages.add(i) + top_ys.append(y1) + elif y0 > height * 0.90: + bottom_pages.add(i) + bottom_ys.append(y0) + + def pct(values: list[float], q: float) -> float: + vs = sorted(values) + return vs[min(len(vs) - 1, int(q * len(vs)))] + + need = threshold * doc.page_count + top = pct(top_ys, 0.9) + 2 if len(top_pages) >= need else 0.0 + bottom = height - pct(bottom_ys, 0.1) + 2 if len(bottom_pages) >= need else 0.0 + return round(top, 1), round(bottom, 1) + + +# Word runs the dots together ("Introduction ....... 4"); XLingPaper/LaTeX +# spaces them out (". . . . . . . 2"). Both end in a page number. +LEADER = re.compile(r"(\.\s*){4,}\s*\d+\s*$") +CONTENTS_HEAD = re.compile( + r"^#{1,6}\s*\**\s*(table of contents|contents|list of figures|list of tables)" + r"\s*:?[ ]*\**\s*$", re.IGNORECASE) +INDEX_HEAD = re.compile( + r"^#{1,6}\s*\**\s*(language|subject|topic)?\s*index\s*\**\s*$", + re.IGNORECASE, +) +HEADING = re.compile(r"^(#{1,6})\s+(.*\S)\s*$") +EQUATION_LABEL = re.compile(r"^\(\d+\)$") + + +def strip_toc(md: str) -> str: + """Drop the document's own table of contents. + + Every Word-produced PDF opens with dotted-leader contents lines, which + pymupdf4llm renders as a bogus one-column table -- 16 to 37 lines of + "2.1 Starting up a Project .......... 4" per document. The markdown file + already has real headings, and this branch has a README index, so the + inline copy is pure noise for a reader and for retrieval alike. + """ + out = [] + lines = md.splitlines() + front_limit = max(1, int(len(lines) * 0.25)) + for i, line in enumerate(lines): + bare = line.strip().strip("|").strip() + in_front_matter = i < front_limit + if in_front_matter and LEADER.search(bare): + continue + # The table skeleton left behind once its rows are gone. Tested with a + # character-set check, not a regex: a nested-quantifier pattern like + # (:?-+:?\s*\|?)+ backtracks exponentially on a long separator row. + # A separator is only real if an actual table row precedes it. + if in_front_matter and bare and set(bare) <= set("|-: "): + prev = next((x for x in reversed(out) if x.strip()), "") + if not prev.strip().startswith("|"): + continue + if in_front_matter and re.fullmatch( + r"\|\s*Contents\s*\|", line.strip(), re.IGNORECASE + ): + continue + out.append(line) + return "\n".join(out) + + +def strip_contents_sections(md: str) -> str: + """Drop front-matter contents sections, whatever their rendered shape.""" + lines = md.splitlines() + out, i = [], 0 + while i < len(lines): + match = CONTENTS_HEAD.match(lines[i].strip()) + if match and i < len(lines) * 0.4: + level = len(HEADING.match(lines[i]).group(1)) + end = i + 1 + while end < len(lines): + # The TonePars paper places edition notes immediately after a + # compact contents list. They are prose, not TOC furniture. + if re.match( + r"^Editor['’]s note\b", lines[end].strip(), re.IGNORECASE + ): + break + heading = HEADING.match(lines[end]) + if heading and len(heading.group(1)) <= level: + break + end += 1 + # Without a verified closing boundary, preserve the section. A + # lower-level chapter may be the real body, and deleting to EOF is + # worse than retaining a redundant contents list. + if end < len(lines): + i = end + continue + out.append(lines[i]) + i += 1 + return "\n".join(out) + + +def strip_back_index(md: str) -> str: + """Drop a back-of-book index. + + ConceptualIntroFLEx ends with "Language index" and "Subject index": several + thousand words of headword-plus-page-number that carry no sentences, cannot + be followed without the printed pagination, and would otherwise be the + single largest retrievable block in the document. + + Only honoured near the end of a document, so a section legitimately called + "Index" mid-text is left alone. + """ + lines = md.splitlines() + for i, line in enumerate(lines): + if INDEX_HEAD.match(line.strip()) and i > len(lines) * 0.75: + return "\n".join(lines[:i]).rstrip() + "\n" + return md + + +def demote_headings(md: str) -> str: + """Push every heading down one level and strip emphasis markers. + + The page's own H1 is the document title, added by the caller, so the PDF's + top-level sections belong at H2. Word also bolds its headings, which pandoc + faithfully reproduces as "# **1 Introduction**". + """ + out = [] + for line in md.splitlines(): + m = HEADING.match(line) + if not m: + out.append(line) + continue + level = min(6, len(m.group(1)) + 1) + text = re.sub(r"^[*_]{1,2}|[*_]{1,2}$", "", m.group(2).strip()).strip() + out.append(f"{'#' * level} {text}" if text else "") + return "\n".join(out) + + +def markdown_label(line: str) -> str: + """Plain text from one simple Markdown heading or emphasized line.""" + text = re.sub(r"^#{1,6}\s+", "", line.strip()) + text = clean(re.sub(r"[*_`]", "", text)) + return re.sub(r"\s+([:;,])", r"\1", text) + + +def pick_title(meta_title: str, md: str, stem: str) -> str: + """Choose useful metadata, a document heading, or the curated filename.""" + stem = stem.replace("_", " ").strip() + bad = re.compile( + r"\.(doc|pdf|rtf)x?\b|^microsoft word\b|\breadme\b", re.IGNORECASE + ) + + candidate = clean(meta_title) + if candidate and not bad.search(candidate) and 4 < len(candidate) < 120: + if candidate.lower() in stem.lower() and len(candidate) < len(stem): + return stem + return candidate + + generic = {"contents", "table of contents", "list of figures", "list of tables"} + for line in md.splitlines()[:24]: + text = markdown_label(line) + if not text or text.rstrip(":").lower() in generic: + continue + if re.match(r"^\d+(?:\.\d+)*\s+(?=\S)", text) or text.isdigit(): + continue + heading = HEADING.match(line) + if heading: + if 4 < len(text) < 120 and not bad.search(text): + return text + # Some title pages have no bookmark or heading style. Accept a title + # line before falling through to the generic Contents bookmark. + elif 12 < len(text) < 120 and not bad.search(text): + return text + return stem + + +def drop_repeated_title(md: str, title: str) -> str: + """Remove the document's own title heading when it restates the page title. + + Otherwise every PDF opens with the title twice -- once as the H1 this tool + adds, once as the heading from the PDF's title page ("Technical Notes on + FieldWorks Send-Receive" then "Technical Notes on Fieldworks Send/Receive"). + Compared on letters and digits alone, so punctuation and casing differences + like Send-Receive vs Send/Receive still count as the same title. + """ + key = lambda s: re.sub(r"[^a-z0-9]", "", markdown_label(s).lower()) + want = key(title) + lines = md.splitlines() + + # A title may be a plain emphasized line, a heading, or a heading followed + # by a separately styled continuation. Remove every opening copy: the + # Conceptual Introduction PDF contains the same split title twice. + while True: + found = False + for start in range(min(len(lines), 24)): + if not lines[start].strip(): + continue + joined = "" + used = 0 + for end in range(start, min(len(lines), start + 8)): + if not lines[end].strip(): + continue + joined += key(lines[end]) + used += 1 + if joined == want: + del lines[start:end + 1] + while start < len(lines) and not lines[start].strip(): + del lines[start] + found = True + break + if used >= 3 or not want.startswith(joined): + break + if found: + break + if not found: + break + return "\n".join(lines).strip() + "\n" + + +def normalize_pdf_headings(md: str) -> str: + """Remove audited false headings and make the first real level H2. + + Some bookmark trees promote author bylines and equation numbers to + headings. The former only occurs as a short title-page heading at level + four or deeper; the latter is unambiguously a standalone parenthesized + number. After those are removed, shift an otherwise valid tree so its + shallowest heading is the PDF body's H2. + """ + lines = md.splitlines() + headings: list[tuple[int, int, str]] = [] + fenced = False + for i, line in enumerate(lines): + if line.lstrip().startswith("```"): + fenced = not fenced + continue + if fenced: + continue + match = HEADING.match(line) + if match: + headings.append((i, len(match.group(1)), clean(match.group(2)))) + + remove: set[int] = set() + for i, level, text in headings: + if EQUATION_LABEL.fullmatch(text): + remove.add(i) + + remaining = [(i, level, text) for i, level, text in headings if i not in remove] + if remaining: + first_i, first_level, first_text = remaining[0] + words = first_text.split() + looks_like_name = ( + first_i < 8 and first_level >= 4 and 2 <= len(words) <= 5 + and all(word[:1].isupper() for word in words if word) + and not any(char.isdigit() for char in first_text) + ) + if looks_like_name: + remove.add(first_i) + remaining = remaining[1:] + + shift = 2 - min((level for _, level, _ in remaining), default=2) + out: list[str] = [] + for i, line in enumerate(lines): + if i in remove: + continue + match = HEADING.match(line) + if not match: + out.append(line) + continue + level = max(1, len(match.group(1)) + shift) + out.append("#" * level + line[len(match.group(1)):]) + return "\n".join(out) + + +def outline_of(md: str) -> list[tuple[int, str]]: + """Headings in generated markdown, ignoring anything inside fenced code.""" + out, fenced = [], False + for line in md.splitlines(): + if line.lstrip().startswith("```"): + fenced = not fenced + continue + if fenced: + continue + m = re.match(r"^(#{1,6})\s+(.*\S)\s*$", line) + if m: + out.append((len(m.group(1)), clean(m.group(2)))) + return out + + +def normalize_outline(outline: list[tuple[int, str]] | list[list]) -> list[list]: + """Return the ordered, comparable representation used by outline locks.""" + return [[int(level), clean(str(text))] for level, text in outline] + + +def outline_matches(pin: dict, outline: list[tuple[int, str]] | list[list]) -> bool: + """Compare every normalized heading, in order, with a pinned outline.""" + expected = pin.get("outline") + if not isinstance(expected, list): + # Legacy count/level1 pins are intentionally not sufficient locks. + return False + return expected == normalize_outline(outline) + + +def finalize_pdf(meta_title: str, md: str, stem: str) -> tuple[str, str, list]: + """Select the title and derive structure from the body that will be emitted.""" + title = pick_title(meta_title, md, stem) + body = normalize_pdf_headings(drop_repeated_title(md, title)) + return title, body, outline_of(body) + + +PAGE_NUMBER = re.compile(r"^\d+$") +HTML_TABLE = re.compile(r"", re.DOTALL | re.IGNORECASE) + + +def strip_furniture(pages: list[str], threshold: float = 0.6) -> list[str]: + """Remove running headers and footers from per-page markdown. + + Detection is by repetition rather than by position, which keeps it + independent of the producing toolchain -- the corpus spans four Word + versions, XLingPaper/LaTeX, and two Acrobat Distiller variants. A line + qualifies only if it sits at the very top or bottom of its page and recurs, + modulo page numbers, on most pages. + """ + if len(pages) < 3: + return pages + + def key(line: str) -> str: + line = line.strip() + return "#page-number" if PAGE_NUMBER.fullmatch(line) else line + + counts: collections.Counter = collections.Counter() + samples: dict[str, str] = {} + for page in pages: + lines = [ln.strip() for ln in page.splitlines() if ln.strip()] + for line in set(lines[:2] + lines[-2:]): + normalized = key(line) + counts[normalized] += 1 + samples.setdefault(normalized, line) + + # Word repeats the *current section* heading in the running header, so any + # one header text may cover only two pages. Accept two occurrences only for + # explicit Markdown headings and page numbers. Plain prose needs both three + # occurrences and a majority of pages, preventing repeated instructions or + # distinct numbered headings from being collapsed into furniture. + need = max(3, int(threshold * len(pages) + 0.999)) + furniture = { + k for k, n in counts.items() + if ((k == "#page-number" or HEADING.match(samples[k])) and n >= 2) + or n >= need + } + if not furniture: + return pages + + cleaned = [] + for page in pages: + lines = page.splitlines() + # Trim from the ends only; an identical sentence mid-page is content. + while lines and (not lines[0].strip() + or key(lines[0]) in furniture): + lines.pop(0) + while lines and (not lines[-1].strip() + or key(lines[-1]) in furniture): + lines.pop() + cleaned.append("\n".join(lines)) + return cleaned + + +LUA = Path(__file__).parent / "fwhelp.lua" + + +def tables_to_gfm(md: str, unconverted: list | None = None) -> str: + """Convert the HTML tables pymupdf4llm emits into GFM pipe tables. + + Runs through fwhelp.lua, the same filter the CHM side uses. Without it + pandoc emits the table straight back as HTML, because pymupdf4llm attaches +

widths and fixed column widths force a grid table -- the identical + problem RoboHelp's markup causes, and already solved there. + """ + def convert(match: re.Match) -> str: + # A GFM pipe cell cannot hold a line break, so pandoc answers any table + # containing one with raw HTML instead -- which left 43 tables across + # the corpus unconverted. The breaks are where the PDF happened to wrap + # the cell text ("the analysis data control file used by
XAmple."), + # so they encode page width rather than meaning and collapse to a space. + html = re.sub(r"", " ", match.group(0)) + proc = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={LUA}"], + input=html, capture_output=True, text=True, encoding="utf-8", + check=False, + ) + if proc.returncode != 0 or not proc.stdout.strip() or " tuple[str, str, list]: + """Convert one PDF and retain source replacement-character provenance.""" + doc = fitz.open(pdf) + try: + toc = doc.get_toc() + if toc: + hdr, strategy = TocHeaders(doc), "bookmarks" + else: + hdr, strategy = FontHeaders(doc), "font-inference" + + image_dir.mkdir(parents=True, exist_ok=True) + chunks = pymupdf4llm.to_markdown( + doc, + hdr_info=hdr, + # Per page, so running headers/footers can be found by repetition. + # pymupdf4llm's `margins` argument does not reach this content -- + # the footer survives at every margin value tested, including one + # far larger than the band it occupies -- so strip_furniture() + # removes them afterwards instead. + page_chunks=True, + write_images=True, + image_path=str(image_dir), + image_format="png", + # Skip decorative rules and page-furniture fragments; the corpus + # PDFs use them heavily and each would otherwise become a file. + image_size_limit=0.08, + # "lines_strict" only, never "text": the text strategy reports a + # 6-column x 53-row "table" for an ordinary page of prose, and + # treats every dotted-leader contents page as tabular. Verified by + # rendering the pages -- lines_strict finds 11 genuinely ruled + # tables across the corpus, and they are all real. + table_strategy="lines_strict", + # HTML, then pandoc, because pymupdf4llm's markdown table writer + # concatenates spans without separators: "part of Entry by default" + # comes out as "part of Entrybydefault", silently corrupting the SFM + # marker reference tables. Its HTML output keeps the spacing. + table_output="html", + show_progress=False, + ) + pages = doc.page_count + source_replacements = [] + for page_number, page in enumerate(doc, 1): + details = source_replacement_details(page.get_text()) + if details: + source_replacements.append({"page": page_number, **details}) + finally: + doc.close() + + unconverted: list = [] + md = tables_to_gfm("\n\n".join(strip_furniture([c["text"] for c in chunks])), + unconverted) + + # pymupdf4llm builds image references relative to the process's working + # directory rather than to the markdown file that holds them, so each link + # arrives carrying the whole export path ("tools/out/pdf/X_images/...") and + # resolves only when the file is read from that one directory -- or, when + # --out is not under the cwd, as an absolute path that resolves nowhere + # else at all. The images are written beside the markdown, so reduce every + # reference to the sibling folder it actually sits in. + md = re.sub(rf"\(\S*?{re.escape(image_dir.name)}/", f"({image_dir.name}/", md) + + # Remove the document's own contents list and back-of-book index, then push + # its headings down one level so the title supplied by the caller is the + # page's only H1. + md = demote_headings(strip_back_index(strip_contents_sections(strip_toc(md)))) + + md = re.sub(r"\n{4,}", "\n\n\n", md).strip() + "\n" + if source_replacements: + unconverted.append({"kind": "source_replacement", "pages": source_replacements}) + return md, f"{strategy} ({pages}p)", unconverted + + +def frontmatter(fields: dict) -> str: + lines = ["---"] + + def emit(k, v, indent=""): + if v in (None, "", [], {}): + return + if isinstance(v, dict): + lines.append(f"{indent}{k}:") + for child, value in v.items(): + emit(child, value, indent + " ") + elif isinstance(v, list): + lines.append(f"{indent}{k}:") + for item in v: + lines.append(f"{indent} - {yaml_scalar(item)}") + else: + lines.append(f"{indent}{k}: {yaml_scalar(v)}") + + for k, v in fields.items(): + emit(k, v) + return "\n".join(lines + ["---"]) + + +PDF_MANIFEST = ".pdf-converter-manifest.json" +PDF_MANIFEST_SCHEMA = 2 + + +class ManifestError(ValueError): + """A PDF manifest is malformed or does not describe canonical outputs.""" + + +def _validate_lock_path(path: Path) -> Path: + """Reject linked or parent-traversing lock paths before any lock I/O.""" + path = Path(path) + if ".." in path.parts: + raise ManifestError(f"unsafe lock path: {path}") + if first_link_in_path(path) is not None: + raise ManifestError(f"lock path contains symlink/junction: {path}") + return path + + +def _manifest_pairs(pairs): + result = {} + for key, value in pairs: + if key in result: + raise ManifestError(f"duplicate manifest key: {key!r}") + result[key] = value + return result + + +def _safe_manifest_relative(value: object, label: str) -> str: + if (not isinstance(value, str) or not value or "\\" in value + or "\x00" in value or ":" in value): + raise ManifestError(f"unsafe {label} path: {value!r}") + if any(ord(char) < 32 for char in value): + raise ManifestError(f"unsafe {label} path: {value!r}") + path = Path(value) + if value.startswith(("/", "\\")) or path.drive: + raise ManifestError(f"absolute {label} path: {value!r}") + parts = value.split("/") + if any(part in {"", ".", ".."} for part in parts): + raise ManifestError(f"traversal {label} path: {value!r}") + return value + + +def _manifest_destinations(source: str) -> tuple[str, str]: + if not source.casefold().endswith(".pdf"): + raise ManifestError(f"manifest source is not a PDF: {source!r}") + rel_dest = slug_path(source)[:-4] + ".md" + rel_images = (Path(rel_dest).parent / (Path(rel_dest).stem + "_images")).as_posix() + return rel_dest, rel_images + + +def _ownership_key(relative: str) -> str: + """Return the host filesystem's normalized key for a safe relative path.""" + return os.path.normcase(os.path.normpath(relative)) + + +def validate_manifest(manifest: Mapping, out: Path, *, + authenticated_images: set[str] | None = None) -> dict[str, dict]: + """Validate a complete schema-2 manifest before any output mutation.""" + if not isinstance(manifest, dict) or set(manifest) != {"schema", "files"}: + raise ManifestError("manifest must contain exactly schema and files") + if type(manifest["schema"]) is not int or manifest["schema"] != PDF_MANIFEST_SCHEMA: + raise ManifestError("unsupported PDF manifest schema") + files = manifest["files"] + if not isinstance(files, dict): + raise ManifestError("manifest files must be an object") + root = Path(os.path.abspath(os.fspath(out))) + authenticated_image_keys = { + _ownership_key(value) for value in (authenticated_images or set()) + if isinstance(value, str) + } + seen_sources: set[str] = set() + seen_destinations: set[str] = set() + normalized: dict[str, dict] = {} + for source, entry in files.items(): + source = _safe_manifest_relative(source, "source") + source_key = source.casefold() + if source_key in seen_sources: + raise ManifestError(f"case-colliding manifest source: {source!r}") + seen_sources.add(source_key) + if not isinstance(entry, dict) or set(entry) != {"markdown", "images"}: + raise ManifestError(f"invalid manifest entry for {source!r}") + expected_markdown, expected_images = _manifest_destinations(source) + markdown = _safe_manifest_relative(entry["markdown"], "markdown") + if markdown != expected_markdown: + raise ManifestError(f"markdown destination mismatch for {source!r}") + images = entry["images"] + if images is not None: + images = _safe_manifest_relative(images, "images") + if images != expected_images: + raise ManifestError(f"images destination mismatch for {source!r}") + elif first_link_in_path(root / expected_images) is not None: + raise ManifestError(f"images destination contains symlink/junction: {source!r}") + elif (root / expected_images).exists() and _ownership_key(expected_images) not in authenticated_image_keys: + raise ManifestError(f"null images destination has existing output: {source!r}") + for destination in (markdown, images): + if destination is None: + continue + key = destination.casefold() + if key in seen_destinations: + raise ManifestError(f"case-colliding manifest destination: {destination!r}") + seen_destinations.add(key) + if first_link_in_path(root / destination) is not None: + raise ManifestError(f"manifest destination contains symlink/junction: {destination!r}") + if _owned_path(root, destination) is None: + raise ManifestError(f"manifest destination escapes output root: {destination!r}") + normalized[source] = {"markdown": markdown, "images": images} + return normalized + + +def _load_manifest(path: Path, out: Path) -> tuple[dict, dict[str, dict]]: + if first_link_in_path(path) is not None: + raise ManifestError("PDF manifest path contains symlink/junction") + try: + manifest = json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=_manifest_pairs) + except ManifestError: + raise + except (OSError, UnicodeError, json.JSONDecodeError) as exc: + raise ManifestError(f"cannot read PDF manifest: {exc}") from exc + return manifest, validate_manifest(manifest, out) + + +def _normalize_root(path: Path) -> Path: + """Normalize lexical ``..`` segments without resolving symlinks.""" + return Path(os.path.normpath(os.fspath(Path(path).expanduser()))) + + +def discover_pdfs(repo: Path) -> list[Path]: + """Discover PDF inputs independent of filename case on the host OS.""" + repo = _normalize_root(repo) + return discover_source_files( + repo, suffixes={".pdf"}, recursive=True, exclude_dirs={".git"} + ) + + +def destination_collisions(repo: Path, out: Path, + pdfs: list[Path] | None = None) -> list[tuple[str, list[str]]]: + """Find PDFs whose normalized destination path is claimed more than once.""" + repo = _normalize_root(repo) + pdfs = pdfs if pdfs is not None else discover_pdfs(repo) + claimed: dict[str, tuple[str, list[str]]] = {} + for pdf in pdfs: + rel = pdf.relative_to(repo).as_posix() + dest = out / (slug_path(rel)[: -len(".pdf")] + ".md") + display = dest.relative_to(out).as_posix() + key = display.casefold() + if key not in claimed: + claimed[key] = (display, []) + claimed[key][1].append(rel) + return [ + (display, sorted(rels)) + for display, rels in sorted(claimed.values()) if len(rels) > 1 + ] + + +def _owned_path(out: Path, relative: str) -> Path | None: + """Return a lexical in-root path, refusing link chains.""" + candidate = Path(os.path.abspath(os.fspath(Path(out) / relative))) + root = Path(os.path.abspath(os.fspath(out))) + if first_link_in_path(candidate) is not None: + raise ManifestError(f"manifest path contains symlink/junction: {relative!r}") + try: + candidate.relative_to(root) + except ValueError: + return None + return candidate + + +def _validate_regular_file(path: Path, label: str) -> None: + if first_link_in_path(path) is not None or not path.exists() or not path.is_file(): + raise ManifestError(f"{label} must be an existing regular non-link file: {path}") + + +def _validate_clean_directory(path: Path, label: str) -> None: + if first_link_in_path(path) is not None or not path.exists() or not path.is_dir(): + raise ManifestError(f"{label} must be an existing non-link directory: {path}") + pending = [path] + while pending: + current = pending.pop() + try: + entries = list(os.scandir(current)) + except OSError as exc: + raise ManifestError(f"cannot inspect {label}: {current}") from exc + for entry in entries: + child = Path(entry.path) + if first_link_in_path(child) is not None: + raise ManifestError(f"{label} contains symlink/junction: {child}") + if entry.is_dir(follow_symlinks=False): + pending.append(child) + elif not entry.is_file(follow_symlinks=False): + raise ManifestError(f"{label} contains non-regular entry: {child}") + + +def _validate_manifest_outputs(out: Path, stage: Path, + previous: dict[str, dict], current: dict[str, dict]) -> None: + """Validate all filesystem objects before staging or replacing anything.""" + for files in previous.values(): + if not isinstance(files, dict): + continue + markdown = files.get("markdown") + if isinstance(markdown, str): + path = _owned_path(out, markdown) + if path is None: + raise ManifestError(f"prior markdown escapes output: {markdown!r}") + _validate_regular_file(path, "prior markdown") + images = files.get("images") + if isinstance(images, str): + path = _owned_path(out, images) + if path is None: + raise ManifestError(f"prior images escape output: {images!r}") + _validate_clean_directory(path, "prior images") + + prior_path_keys = { + _ownership_key(value) + for files in previous.values() + if isinstance(files, dict) + for value in (files.get("markdown"), files.get("images")) + if isinstance(value, str) + } + for files in current.values(): + if not isinstance(files, dict): + continue + markdown = files.get("markdown") + if isinstance(markdown, str): + staged = _owned_path(stage, markdown) + if staged is None: + raise ManifestError(f"current markdown escapes stage: {markdown!r}") + _validate_regular_file(staged, "current staged markdown") + target = _owned_path(out, markdown) + if target is not None and target.exists() and _ownership_key(markdown) not in prior_path_keys: + raise FileExistsError(f"PDF destination is not converter-owned: {markdown}") + images = files.get("images") + if isinstance(images, str): + staged = _owned_path(stage, images) + if staged is None: + raise ManifestError(f"current images escape stage: {images!r}") + _validate_clean_directory(staged, "current staged images") + target = _owned_path(out, images) + if target is not None and target.exists() and _ownership_key(images) not in prior_path_keys: + raise FileExistsError(f"PDF destination is not converter-owned: {images}") + + +def _remove_owned_entry(out: Path, files: dict) -> None: + for key in ("markdown", "images"): + value = files.get(key) + path = _owned_path(out, value) if isinstance(value, str) else None + if path is None or not path.exists(): + continue + if path.is_dir(): + shutil.rmtree(path) + else: + path.unlink() + + +def _promote_pdf_outputs_locked(out: Path, stage: Path, + previous: dict, current: dict, + lock_path: Path | None = None, + fresh: dict | None = None) -> None: + """Replace the complete converter-owned PDF set from a finished stage.""" + out = Path(out) + if lock_path is not None: + lock_path = _validate_lock_path(lock_path) + manifest_path = out / PDF_MANIFEST + # Validate the on-disk manifest before even creating a backup. The caller's + # parsed view is also checked so a stale/tampered in-memory view cannot + # authorize a different deletion set. + if first_link_in_path(manifest_path) is not None: + raise ManifestError("PDF manifest path contains symlink/junction") + if manifest_path.exists(): + _, actual_previous = _load_manifest(manifest_path, out) + if previous != actual_previous: + raise ManifestError("previous PDF manifest does not match on-disk manifest") + else: + actual_previous = validate_manifest( + {"schema": PDF_MANIFEST_SCHEMA, "files": previous}, out + ) + previous = actual_previous + prior_images = { + _ownership_key(files["images"]) for files in previous.values() + if isinstance(files, dict) and isinstance(files.get("images"), str) + } + validate_manifest( + {"schema": PDF_MANIFEST_SCHEMA, "files": current}, out, + authenticated_images=prior_images, + ) + _validate_manifest_outputs(out, stage, previous, current) + out.mkdir(parents=True, exist_ok=True) + staged_manifest = stage / PDF_MANIFEST + if first_link_in_path(staged_manifest) is not None: + raise ManifestError("staged PDF manifest path contains symlink/junction") + staged_manifest.write_text( + json.dumps({"schema": PDF_MANIFEST_SCHEMA, "files": current}, indent=1, + ensure_ascii=False) + "\n", + encoding="utf-8", + ) + staged_lock = stage / ".pdf-outlines.json" + if lock_path is not None: + _validate_lock_path(lock_path) + if first_link_in_path(staged_lock) is not None: + raise ManifestError("staged PDF lock path contains symlink/junction") + staged_lock.write_text( + json.dumps(fresh or {}, indent=1, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + + # Validate targets before removing anything. Existing paths are safe to + # replace only when the prior manifest claimed them; CHM/unrelated output + # must never be overwritten by a PDF conversion. + prior_paths = { + value + for files in previous.values() + if isinstance(files, dict) + for value in (files.get("markdown"), files.get("images")) + if isinstance(value, str) + } + + backup = Path(tempfile.mkdtemp(prefix=".pdf-backup-", dir=out.parent)) + backed_up: list[tuple[str, Path]] = [] + manifest_backup = backup / PDF_MANIFEST + had_manifest = manifest_path.exists() + lock_backup = backup / "pdf-outlines.json" + if lock_path is not None: + _validate_lock_path(lock_path) + had_lock = lock_path is not None and lock_path.exists() + try: + if had_manifest: + shutil.copy2(manifest_path, manifest_backup) + if had_lock and lock_path is not None: + shutil.copy2(lock_path, lock_backup) + + # Keep a recoverable copy until every staged path has been promoted. + for value in prior_paths: + source = _owned_path(out, value) + target = _owned_path(backup, value) + if source is None or target is None or not source.exists(): + continue + target.parent.mkdir(parents=True, exist_ok=True) + if source.is_dir(): + shutil.copytree(source, target) + else: + shutil.copy2(source, target) + backed_up.append((value, target)) + + # Remove every prior owned path, including the same source's image + # folder: staging is a complete replacement, so omitted files die. + for files in previous.values(): + if isinstance(files, dict): + _remove_owned_entry(out, files) + + for files in current.values(): + if not isinstance(files, dict): + continue + for key in ("markdown", "images"): + value = files.get(key) + if not isinstance(value, str): + continue + source = _owned_path(stage, value) + if source is None or not source.exists(): + continue + if source.is_dir() and not any(source.iterdir()): + continue + target = _owned_path(out, value) + if target is None: + continue + target.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(source), str(target)) + shutil.move(str(staged_manifest), str(manifest_path)) + if lock_path is not None: + _validate_lock_path(lock_path) + lock_path.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(staged_lock), str(lock_path)) + except Exception: + # A failed move must not leave a half-promoted PDF set behind. + for files in current.values(): + if isinstance(files, dict): + _remove_owned_entry(out, files) + for value, source in backed_up: + target = _owned_path(out, value) + if target is None or not source.exists(): + continue + target.parent.mkdir(parents=True, exist_ok=True) + if source.is_dir(): + shutil.copytree(source, target) + else: + shutil.copy2(source, target) + if had_manifest and manifest_backup.exists(): + manifest_path.unlink(missing_ok=True) + shutil.copy2(manifest_backup, manifest_path) + elif not had_manifest: + manifest_path.unlink(missing_ok=True) + if lock_path is not None: + if had_lock and lock_backup.exists(): + lock_path.unlink(missing_ok=True) + shutil.copy2(lock_backup, lock_path) + elif not had_lock: + lock_path.unlink(missing_ok=True) + raise + finally: + shutil.rmtree(backup, ignore_errors=True) + + +def promote_pdf_outputs(out: Path, stage: Path, + previous: dict, current: dict, + lock_path: Path | None = None, + fresh: dict | None = None) -> None: + """Promote PDF output while serializing direct destination mutation.""" + out = Path(out) + if lock_path is not None: + # Keep the same output-then-global order as ``run`` when updating the + # shared outline lock. The locked implementation is used internally + # by ``run`` to avoid recursive acquisition. + with export_locks(out, OUTLINES): + _promote_pdf_outputs_locked( + out, stage, previous, current, lock_path=lock_path, fresh=fresh, + ) + else: + with export_locks(out): + _promote_pdf_outputs_locked( + out, stage, previous, current, lock_path=lock_path, fresh=fresh, + ) + + +def _source_url(rel: str, resolver=None) -> str: + """Resolve a stable source URL/ref through the small orchestration seam.""" + encoded = "/".join(quote(part, safe="") for part in rel.split("/")) + if resolver is None: + value = encoded + elif callable(resolver): + value = str(resolver(rel)) + elif isinstance(resolver, dict): + value = str(resolver.get(rel, encoded)) + else: + value = str(resolver) + if "{path}" in value: + value = value.format(path=encoded) + else: + value = value.rstrip("/") + "/" + encoded + if "{path}" in value: + value = value.format(path=encoded) + parts = urlsplit(value) + if parts.scheme or parts.netloc: + value = urlunsplit(( + parts.scheme, + parts.netloc, + quote(parts.path, safe="/%:@-._~!$&'()*+,;=%"), + parts.query, + parts.fragment, + )) + else: + value = quote(value, safe="/%:@-._~!$&'()*+,;=%") + return value + + +def _sha256(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as source: + for chunk in iter(lambda: source.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def replacement_provenance(source_pages: list[dict], generated_count: int) -> dict: + """Classify replacement characters as source-derived or exporter-created.""" + source_count = sum(int(item.get("count", 0)) for item in source_pages) + return { + "source_count": source_count, + "exporter_count": max(0, generated_count - source_count), + } + + +def source_replacement_details(text: str) -> dict | None: + """Identify source glyphs likely to become replacement characters.""" + chars = [ + char for char in text + if char == "\ufffd" or (ord(char) < 32 and char not in "\t\n\r") + ] + if not chars: + return None + return { + "count": len(chars), + "codepoints": sorted({f"U+{ord(char):04X}" for char in chars}), + } + + +def _run_locked(repo: Path, out: Path, update: bool, source_url=None, + source_ref=None) -> tuple[dict, list[str]]: + """Convert all PDFs, with ``source_url`` as the repository policy seam. + + ``source_url`` may be a URL template containing ``{path}``, a mapping, or + a callable receiving the repository-relative PDF path. ``source_ref`` is a + backwards-compatible alias for callers that prefer that terminology. + """ + repo = _normalize_root(repo) + if source_ref is not None: + source_url = source_ref + _validate_lock_path(OUTLINES) + pins = json.loads(OUTLINES.read_text(encoding="utf-8")) if OUTLINES.exists() else {} + fresh: dict[str, dict] = {} + report: dict[str, list] = collections.defaultdict(list) + lines: list[str] = [] + + pdfs = discover_pdfs(repo) + collisions = destination_collisions(repo, out, pdfs) + if collisions: + report["destination_collisions"].extend(collisions) + lines.extend(f" COLLISION {dest}: {', '.join(rels)}" for dest, rels in collisions) + return {"converted": 0, "report": dict(report), "lines": lines}, lines + + out.parent.mkdir(parents=True, exist_ok=True) + previous: dict = {} + manifest_path = out / PDF_MANIFEST + if first_link_in_path(manifest_path) is not None: + raise ManifestError("PDF manifest path contains symlink/junction") + if manifest_path.exists(): + _, previous = _load_manifest(manifest_path, out) + current: dict = {} + stage = Path(tempfile.mkdtemp(prefix=".pdf-convert-", dir=out.parent)) + + try: + for pdf in pdfs: + rel = pdf.relative_to(repo).as_posix() + rel_dest = slug_path(rel)[: -len(".pdf")] + ".md" + dest = stage / rel_dest + images = dest.parent / (dest.stem + "_images") + + try: + md, strategy, unconverted = convert_pdf(pdf, dest, images) + # A single damaged PDF is reportable while the remaining corpus + # continues through staging; this boundary intentionally catches + # converter/backend errors of varying concrete types. + except Exception as exc: # noqa: BLE001 + report["pdf_failures"].append([rel, f"{type(exc).__name__}: {exc}"]) + lines.append(f" FAILED {rel}: {exc}") + continue + + table_warnings = [item for item in unconverted if not isinstance(item, dict)] + source_warnings = [ + item for item in unconverted + if isinstance(item, dict) and item.get("kind") == "source_replacement" + ] + if table_warnings: + report["html_tables_kept"].append([rel, len(table_warnings)]) + source_pages = [ + page for warning in source_warnings for page in warning["pages"] + ] + provenance = replacement_provenance(source_pages, md.count("\ufffd")) + if provenance["source_count"]: + report["pdf_source_replacements"].append([rel, source_pages]) + if provenance["exporter_count"]: + report["pdf_export_replacements"].append([rel, provenance]) + with fitz.open(pdf) as _doc: + metadata = dict(_doc.metadata or {}) + meta_title = metadata.get("title") or "" + title, body, outline = finalize_pdf(meta_title, md, pdf.stem) + normalized_outline = normalize_outline(outline) + fresh[rel] = {"outline": normalized_outline, "headings": len(outline)} + + pin = pins.get(rel) + if not update: + if pin is None: + report["outline_unpinned"].append(rel) + elif not outline_matches(pin, outline): + expected = pin.get("outline", "") + report["outline_drift"].append([ + rel, + f"expected ordered outline {expected!r}, got {normalized_outline!r}", + ]) + + dest.parent.mkdir(parents=True, exist_ok=True) + fm = frontmatter({ + "title": title, + "source": rel, + "source_url": _source_url(rel, source_url), + "sha256": _sha256(pdf), + "pdf_metadata": metadata, + "type": "pdf", + "outline_count": len(outline), + "structure": strategy, + }) + dest.write_text(f"{fm}\n\n# {title}\n\n{body}", + encoding="utf-8") + n_img = len(list(images.glob("*"))) if images.exists() else 0 + if not n_img and images.exists(): + images.rmdir() + current[rel] = { + "markdown": rel_dest, + "images": ( + (Path(rel_dest).parent / (Path(rel_dest).stem + "_images")).as_posix() + if n_img else None + ), + } + lines.append(f" {strategy:<22} {len(outline):>3} hdrs {n_img:>4} img {rel}") + + # No final output or manifest is touched until every PDF and every lock + # check succeeds. A late failure therefore leaves the prior PDF set. + lock_fatal = (report.get("outline_drift") + or report.get("outline_unpinned")) and not update + fatal = (report.get("pdf_failures") or lock_fatal + or report.get("pdf_export_replacements")) + if not fatal and not report.get("pdf_failures"): + _promote_pdf_outputs_locked( + out, stage, previous, current, + lock_path=OUTLINES if update else None, + fresh=fresh if update else None, + ) + if update: + lines.append(f"\n pinned {len(fresh)} outlines -> {OUTLINES}") + finally: + shutil.rmtree(stage, ignore_errors=True) + + return {"converted": len(fresh), "report": dict(report), "lines": lines}, lines + + +def run_in_private_stage(repo: Path, out: Path, update: bool, source_url=None, + source_ref=None) -> tuple[dict, list[str]]: + """Convert into a caller-owned private stage while locking global state. + + The caller must guarantee that ``out`` is an unshared staging path. Such + a path needs no destination lock; creating one beside it would turn the + lockfile into generated content when the enclosing stage is promoted. + """ + with export_locks(OUTLINES): + return _run_locked( + repo, Path(out), update, source_url=source_url, source_ref=source_ref, + ) + + +def run(repo: Path, out: Path, update: bool, source_url=None, + source_ref=None) -> tuple[dict, list[str]]: + """Convert PDFs under output-then-global-outline locks. + + The global outline lock covers the pinned-outline read and any update + promotion, including the interval between those operations. + """ + out = Path(out) + with export_locks(out, OUTLINES): + return _run_locked( + repo, out, update, source_url=source_url, source_ref=source_ref, + ) + + +def _canonical_report(producer_report: dict) -> Report: + """Render producer findings through the shared issue catalog.""" + issues = [] + for code, entries in producer_report.items(): + for item in entries if isinstance(entries, list) else [entries]: + if isinstance(item, (list, tuple)) and item: + path = str(item[0]) + message = str(item[1]) if len(item) > 1 else "" + else: + path = "" + message = str(item) + issues.append(make_issue(code, message, path, item)) + return Report(issues) + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--repo", default=".", type=Path) + ap.add_argument("--out", default="export", type=Path) + ap.add_argument("--update-outlines", action="store_true", + help="re-pin pdf_outlines.json after reviewing the diff") + args = ap.parse_args() + + result, lines = run(args.repo.resolve(), args.out.resolve(), args.update_outlines) + print("\n".join(lines)) + rep = result["report"] + print(f"\nconverted {result['converted']} PDFs") + canonical = _canonical_report(rep) + print(canonical.to_console()) + return 1 if canonical.fatal else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/pdf_outlines.json b/tools/pdf_outlines.json new file mode 100644 index 00000000..11b10f07 --- /dev/null +++ b/tools/pdf_outlines.json @@ -0,0 +1,1751 @@ +{ + "FieldWorks Writing Systems.pdf": { + "outline": [ + [ + 2, + "FieldWorks" + ], + [ + 3, + "Writing Systems" + ], + [ + 4, + "Add a Writing System to a FieldWorks project" + ], + [ + 4, + "Add a new Writing System using the Writing System Wizard" + ], + [ + 5, + "Language ID: Find the relevant language in the Ethnologue" + ], + [ + 6, + "Result: one match" + ], + [ + 6, + "Result: duplicate language names" + ], + [ + 6, + "Result: many matches" + ], + [ + 6, + "Result: apparent duplicate matches" + ], + [ + 6, + "Result: search by country name" + ], + [ + 6, + "Result: no matches" + ], + [ + 5, + "Writing System: Distinguish the Writing System" + ], + [ + 5, + "Appearance: Default fonts" + ], + [ + 5, + "Input: Keyboard" + ], + [ + 5, + "Additional information" + ], + [ + 6, + "Modifying properties of a Writing System" + ], + [ + 6, + "Full name of a Writing System" + ], + [ + 6, + "Information about keyboards and Keyman" + ] + ], + "headings": 18 + }, + "Language Explorer/Training/Publishing FLEx Dictionaries Using Microsoft Word.pdf": { + "outline": [ + [ + 2, + "1 Introduction" + ], + [ + 2, + "2 Exporting a dictionary to Word" + ], + [ + 3, + "2.1 Preparation in FLEx prior to export" + ], + [ + 4, + "Saving list reference names in the Word document" + ], + [ + 3, + "2.2 Exporting a dictionary from FLEx to Word" + ], + [ + 3, + "2.3 Preparing a Word dictionary for publication" + ], + [ + 4, + "Page layout" + ], + [ + 4, + "Exported styles in Word" + ], + [ + 4, + "2.3.2.1 Context styles for Before, Between, and After text" + ], + [ + 4, + "Add page headers with guidewords" + ], + [ + 4, + "2.3.3.1 Changing Headers and Footers" + ], + [ + 4, + "How to fix incorrect guidewords on some pages" + ], + [ + 5, + "7. The remaining pages should continue with the normal page headers." + ], + [ + 4, + "Convert letter headings to single column" + ], + [ + 4, + "Pictures in Word" + ], + [ + 4, + "Editing the internal Word docx file" + ], + [ + 4, + "Dictionary front matter and back matter" + ], + [ + 2, + "3 Useful features in Word" + ], + [ + 3, + "3.1 Basics of Styles in Word" + ], + [ + 3, + "3.2 Preventing styles from changing: dynamic updating" + ], + [ + 3, + "3.3 Advanced Find and Replace" + ], + [ + 4, + "3.3.1Finding styles" + ], + [ + 4, + "Replacing styles" + ], + [ + 4, + "Special codes" + ], + [ + 4, + "Wild cards" + ], + [ + 4, + "Adding text before or after a style" + ], + [ + 3, + "3.4 Removing unused styles in Word" + ], + [ + 3, + "3.5 Word macros" + ], + [ + 4, + "Example recording a macro and running it through your document" + ], + [ + 3, + "3.6 Editing the internal Word docx file" + ], + [ + 2, + "4 Bidirectional dictionaries in Word" + ], + [ + 3, + "4.1 Enabling right-to-left publications in Word" + ], + [ + 3, + "4.2 Basic RTL layout in FLEx and Word" + ], + [ + 4, + "Settings in FLEx" + ], + [ + 4, + "Settings in Word" + ], + [ + 3, + "4.3 Bidirectional algorithm" + ], + [ + 3, + "4.4 Formatting bidirectional entries in FLEx" + ], + [ + 5, + "Word:" + ] + ], + "headings": 38 + }, + "Language Explorer/Training/Technical Notes on FieldWorks Send-Receive.pdf": { + "outline": [ + [ + 2, + "1 Send/Receive Introduction" + ], + [ + 2, + "2 Getting started" + ], + [ + 3, + "2.1 Starting up a Project (FLEx) Send/Receive" + ], + [ + 3, + "2.2 Starting up a Lexicon (LIFT) Send/Receive in FLEx." + ], + [ + 3, + "2.3 Starting up Send/Receive in WeSay." + ], + [ + 3, + "2.4 Chorus Hub startup" + ], + [ + 3, + "2.5 Lexbox Internet setup" + ], + [ + 2, + "3 How it works, or why it doesn’t" + ], + [ + 3, + "3.1 Lexicon examples" + ], + [ + 3, + "3.2 General concepts" + ], + [ + 3, + "3.3 Interlinear examples" + ], + [ + 3, + "3.4 WeSay/LIFT collaboration" + ], + [ + 3, + "3.5 FieldWorks/Paratext collaboration" + ], + [ + 3, + "3.6 Linked files" + ], + [ + 3, + "3.7 FieldWorks and FLEx Bridge versions" + ], + [ + 4, + "3.7.1 FLEx Bridge mercurial compatibility" + ], + [ + 3, + "3.8 FieldWorks project name" + ], + [ + 2, + "4 Technical details" + ], + [ + 3, + "4.1 Chorus Hub issues" + ], + [ + 3, + "4.2 Using FieldWorks backups and Send/Receive" + ], + [ + 3, + "4.3 FLEx backups and Send/Receive" + ], + [ + 4, + "4.3.1 Using FLEx Backup with Send/Receive data" + ], + [ + 4, + "4.3.2 Using FLEx Backup without Send/Receive data" + ], + [ + 3, + "4.4 Viewing Send/Receive history" + ], + [ + 3, + "4.5 Recovering lost data from S/R" + ], + [ + 4, + "4.5.1 Repairing lost or damaged local repo" + ], + [ + 4, + "4.5.2 Using Repository Utility" + ], + [ + 4, + "4.5.3 Using S/R to repair or change data" + ], + [ + 4, + "4.5.4 Dealing with defective repos" + ], + [ + 4, + "4.5.5 Recovering data in difficult situations" + ], + [ + 3, + "4.6 To switch a WeSay bridge user to another FLEx user" + ], + [ + 3, + "4.7 FieldWorks and WeSay compatibility issues" + ], + [ + 3, + "4.8 Modifying FLEx lists outside of FLEx" + ], + [ + 4, + "4.? Unfinished notes" + ] + ], + "headings": 34 + }, + "Language Explorer/Training/Technical Notes on Interlinear Import.pdf": { + "outline": [ + [ + 2, + "1 Importing SFM Interlinear text" + ], + [ + 3, + "Special intermediate file" + ], + [ + 3, + "Constructing literal translations" + ], + [ + 2, + "2 FieldWorks FLExText Interlinear XML" + ], + [ + 3, + "Dealing with audio in interlinear" + ] + ], + "headings": 5 + }, + "Language Explorer/Training/Technical Notes on LinguaLinks Database Import.pdf": { + "outline": [ + [ + 2, + "1 Data model differences" + ], + [ + 3, + "1.1 Subentry and minor entry differences" + ], + [ + 3, + "1.2 LinguaLinks annotations" + ], + [ + 4, + "Steps: Entry Notes" + ], + [ + 3, + "1.3 Interlinear text sections" + ], + [ + 3, + "1.4 Interlinear text collections" + ], + [ + 3, + "1.5 Thesaurus links" + ], + [ + 3, + "1.6 Nested sense limitations" + ], + [ + 3, + "1.7 LinguaLinks sense media files" + ], + [ + 3, + "1.8 Limitations with interlinear text baselines in multiple scripts" + ], + [ + 3, + "1.9 Pictures and pronunciation files" + ], + [ + 3, + "1.10 LinguaLinks subentry types" + ], + [ + 2, + "2 Exporting data from LinguaLinks" + ], + [ + 3, + "2.1 Steps: LinguaLinks export" + ], + [ + 3, + "2.2 Steps: Listing fonts used in LinguaLinks" + ], + [ + 2, + "3 Importing LinguaLinks data into Language Explorer" + ], + [ + 3, + "3.1 Importing into existing Language Explorer data" + ], + [ + 3, + "3.2 Steps: Importing LinguaLinks data" + ], + [ + 2, + "4 Technical process flow of LinguaLinks import" + ], + [ + 2, + "5 Advanced import problem solving" + ], + [ + 3, + "5.1 Partial processing" + ], + [ + 3, + "5.2 Step: Testing encoding converters" + ], + [ + 3, + "5.3 Really large databases" + ], + [ + 3, + "5.3 Converting bar codes" + ], + [ + 3, + "5.3 Missing writing systems" + ] + ], + "headings": 25 + }, + "Language Explorer/Training/Technical Notes on SFM Database Import.pdf": { + "outline": [ + [ + 2, + "1 Mapping SFM fields to the FieldWorks Language Explorer model" + ], + [ + 3, + "1.1 Basic field mapping" + ], + [ + 3, + "1.2 Text fields" + ], + [ + 3, + "1.3 List references" + ], + [ + 3, + "1.4 Semantic domains" + ], + [ + 3, + "1.5 Repeating fields" + ], + [ + 3, + "1.6 Multiple items per field" + ], + [ + 3, + "1.7 Lexical functions" + ], + [ + 3, + "1.8 References to entries and senses" + ], + [ + 3, + "1.9 Multiple uses of an SFM marker" + ], + [ + 3, + "1.10 In-line markers, or character mapping" + ], + [ + 3, + "1.11 Variants" + ], + [ + 4, + "\\lx abc" + ], + [ + 3, + "1.12 Subentries" + ], + [ + 3, + "1.13 Nested senses" + ], + [ + 3, + "1.14 Allomorph conditions" + ], + [ + 3, + "1.15 Import residue" + ], + [ + 3, + "1.15 Custom fields" + ], + [ + 3, + "1.16 Affix markers" + ], + [ + 3, + "1.17 Picture & Pronunciation files" + ], + [ + 3, + "1.18 MDF supported fields" + ], + [ + 3, + "1.19 Unsupported features" + ], + [ + 2, + "2 Legacy to Unicode encoding conversion" + ], + [ + 2, + "3 Importing SFM data into Language Explorer" + ], + [ + 2, + "4 Resolving errors reported by the SFM readiness check" + ], + [ + 3, + "4.1 Eliminating Internet Explorer warnings when launching ZEdit" + ], + [ + 2, + "5 Technical process flow of SFM import" + ], + [ + 2, + "6 Advanced import problem solving" + ], + [ + 3, + "6.1 Extra processing during importing" + ], + [ + 3, + "6.2 Handling non-MDF standards" + ], + [ + 2, + "7 Language Explorer V3.0 (FieldWorks 6.0) bugs" + ] + ], + "headings": 31 + }, + "Language Explorer/Training/Technical Notes on Writing Systems.pdf": { + "outline": [ + [ + 2, + "1 Ingredients" + ], + [ + 3, + "1.1 Encoding Converter" + ], + [ + 3, + "1.2 Font" + ], + [ + 3, + "1.3 MSKLC or Keyman keyboard" + ], + [ + 2, + "2 Notes and Recommendations" + ], + [ + 3, + "2.1 Encoding Converters" + ], + [ + 3, + "2.2 Same Script for different languages?" + ], + [ + 3, + "2.3 Multiple Writing Systems for the same language" + ], + [ + 4, + "2.3.1 Creating and adding phonetic and phonemic writing systems to your project" + ], + [ + 4, + "2.3.2 Creating an alternative writing system, e.g. Roman for a non-Roman script" + ], + [ + 4, + "2.3.3 Reordering writing systems" + ], + [ + 4, + "2.2.4 Current limitations on multiple scripts for interlinear baseline texts" + ], + [ + 2, + "3 MSKLC and Keyman setup" + ], + [ + 4, + "Keyman 9 and FieldWorks 8.0.6 and later Notes" + ], + [ + 4, + "Notes for older versions of Keyman and FieldWorks" + ], + [ + 4, + "Keyman 7 Notes" + ], + [ + 2, + "4 Writing System Error Message" + ], + [ + 3, + "Scenario 1" + ], + [ + 3, + "Scenario 2" + ], + [ + 3, + "Scenario 3" + ] + ], + "headings": 20 + }, + "Language Explorer/Utilities/AlloGenUserDocumentation.pdf": { + "outline": [ + [ + 2, + "1 Introduction" + ], + [ + 3, + "1.1 Invoking** **_Allomorph Generator_ from within** **_FLEx_" + ], + [ + 3, + "1.2 Appearance" + ], + [ + 2, + "2 Edit Operations tab" + ], + [ + 3, + "2.1 Operation name and description" + ], + [ + 3, + "2.2 Pattern section" + ], + [ + 3, + "_2.2.1 Match_" + ], + [ + 3, + "_2.2.2 Morph Types_" + ], + [ + 3, + "_2.2.3 Category_" + ], + [ + 3, + "2.3 Actions section" + ], + [ + 3, + "_2.3.1 Replace operations_" + ], + [ + 3, + "_2.3.2 Environments_" + ], + [ + 3, + "_2.3.3 Stem Name_" + ], + [ + 3, + "2.4 Apply operations to drop-down box" + ], + [ + 3, + "2.5 Save changes button" + ], + [ + 2, + "3 Run Operations tab" + ], + [ + 2, + "4 Edit Replace Operations tab" + ], + [ + 2, + "5 Applying operations" + ], + [ + 2, + "6 Restarting** **_Allomorph Generator_" + ], + [ + 2, + "7 Error messages" + ], + [ + 2, + "8 Known problems" + ], + [ + 2, + "9 Support" + ] + ], + "headings": 22 + }, + "Language Explorer/Utilities/PcPatrFLExUserDocumentation.pdf": { + "outline": [ + [ + 2, + "1 Introduction" + ], + [ + 3, + "1.1 Invoking** **_Use PC-PATR with FLEx_ from within** **_FLEx_" + ], + [ + 3, + "1.2 Initial invocation" + ], + [ + 3, + "1.3 Appearance" + ], + [ + 2, + "2 Buttons" + ], + [ + 3, + "2.1 PC-PATR grammar Browse button" + ], + [ + 3, + "2.2 Rootgloss choices" + ], + [ + 3, + "2.3 Advanced" + ], + [ + 3, + "2.4 Help button" + ], + [ + 3, + "2.5 Refresh texts button" + ], + [ + 3, + "2.6 Disambiguate button" + ], + [ + 3, + "2.7 Parse button" + ], + [ + 2, + "3 Parsing a segment" + ], + [ + 2, + "4 Disambiguating a text" + ], + [ + 2, + "5 Restarting** **_Use PC-PATR with FLEx_" + ], + [ + 3, + "4. remember which segment in that text you last selected.2" + ], + [ + 2, + "6 Error messages" + ], + [ + 2, + "7 Known problems" + ], + [ + 2, + "8 Support" + ], + [ + 2, + "A. The** **_PcPatr Browser_ tool" + ], + [ + 2, + "A.1 Overview" + ], + [ + 2, + "A.2 Right-to-left script" + ], + [ + 2, + "A.3 Keyboard shortcuts" + ] + ], + "headings": 23 + }, + "Language Explorer/Utilities/silewp2007_002.pdf": { + "outline": [ + [ + 2, + "Abstract" + ], + [ + 2, + "1 Introduction" + ], + [ + 2, + "2 Phonological Concepts" + ], + [ + 2, + "3 Implementation Issues" + ], + [ + 3, + "3.1 Syllabification and TBUs" + ], + [ + 3, + "3.2 Lexical Representation" + ], + [ + 3, + "3.3 Tone Rules" + ], + [ + 4, + "3.3.1 Operations" + ], + [ + 4, + "3.3.2 Operation Parameters" + ], + [ + 4, + "3.3.3 Rule Application" + ], + [ + 4, + "3.3.4 Edge Rules" + ], + [ + 4, + "3.3.5 Conditions on Rules" + ], + [ + 4, + "3.3.6 Example" + ], + [ + 3, + "3.4 Orthographic Output" + ], + [ + 2, + "4 Using TonePars with AMPLE" + ], + [ + 4, + "(41) AMPLE **TonePars" + ], + [ + 2, + "5 Conclusion" + ], + [ + 2, + "Appendix A: List of Field Codes" + ], + [ + 2, + "Appendix B: Annotated Syntax for Tone Rules" + ], + [ + 2, + "Endnotes" + ], + [ + 2, + "References" + ] + ], + "headings": 21 + }, + "Language Explorer/Utilities/ToneParsFLExUserDocumentation.pdf": { + "outline": [ + [ + 2, + "1 Introduction" + ], + [ + 3, + "1.1 Invoking** **_Use TonePars with FLEx_ from within** **_FLEx_" + ], + [ + 3, + "1.2 Initial invocation" + ], + [ + 3, + "1.3 Appearance" + ], + [ + 2, + "2 Buttons and check boxes" + ], + [ + 3, + "2.3 Trace Tone Processing check box" + ], + [ + 3, + "2.4 Tracing Options button" + ], + [ + 3, + "2.5 Show Log button" + ], + [ + 3, + "2.6 Help button" + ], + [ + 3, + "2.7 Verify Control File Information check box" + ], + [ + 3, + "2.8 Ignore Context check box" + ], + [ + 3, + "2.9 Refresh Texts button" + ], + [ + 3, + "2.10 Parse this text button" + ], + [ + 3, + "2.11 Parse this segment button" + ], + [ + 2, + "3 Maximum analyses setting for** **_XAmple_" + ], + [ + 2, + "4 Restarting** **_Use TonePars with FLEx_" + ], + [ + 2, + "5 Known problems" + ], + [ + 2, + "6 Output files" + ], + [ + 2, + "7 Error messages" + ], + [ + 2, + "8 Support" + ] + ], + "headings": 20 + }, + "Language Explorer/Utilities/VarGenUserDocumentation.pdf": { + "outline": [ + [ + 2, + "1 Introduction" + ], + [ + 3, + "1.1 Invoking** **_Variant Generator_ from within** **_FLEx_" + ], + [ + 3, + "1.2 Appearance" + ], + [ + 2, + "2 Edit Operations tab" + ], + [ + 3, + "2.1 Operation name and description" + ], + [ + 3, + "2.2 Pattern section" + ], + [ + 3, + "_2.2.1 Match_" + ], + [ + 3, + "_2.2.2 Morph Types_" + ], + [ + 3, + "_2.2.3 Category_" + ], + [ + 3, + "2.3 Actions section" + ], + [ + 3, + "_2.3.1 Replace operations_" + ], + [ + 4, + "Publish entry in" + ], + [ + 3, + "_2.3.2 Variant types_" + ], + [ + 3, + "_2.3.3 Show minor entry_" + ], + [ + 3, + "_2.3.4 Publish entry in_" + ], + [ + 3, + "2.4 Apply operations to drop-down box" + ], + [ + 3, + "2.5 Save changes button" + ], + [ + 2, + "3 Run Operations tab" + ], + [ + 2, + "4 Edit Replace Operations tab" + ], + [ + 2, + "5 Applying operations" + ], + [ + 2, + "6 Restarting** **_Variant Generator_" + ], + [ + 2, + "7 Error messages" + ], + [ + 2, + "8 Known problems" + ], + [ + 2, + "9 Support" + ] + ], + "headings": 24 + }, + "WW-ConceptualIntro/ConceptualIntroFLEx.pdf": { + "outline": [ + [ + 2, + "Abbreviations" + ], + [ + 2, + "1 Introduction" + ], + [ + 3, + "1.1 Key issues" + ], + [ + 4, + "1.1.1 Inflection" + ], + [ + 4, + "1.1.2 Derivation" + ], + [ + 4, + "1.1.3 Ambiguity" + ], + [ + 4, + "1.1.4 Epenthesis" + ], + [ + 4, + "1.1.5 Discontinuous morphemes" + ], + [ + 4, + "1.1.6 Infixation" + ], + [ + 4, + "1.1.7 Reduplication" + ], + [ + 4, + "1.1.8 Root and pattern morphology" + ], + [ + 4, + "1.1.9 Metathesis" + ], + [ + 4, + "1.1.10 Morphemes that may be null" + ], + [ + 3, + "1.2 Tasks for any morphological parser" + ], + [ + 2, + "2 Morphotactics" + ], + [ + 3, + "2.1 Affixation" + ], + [ + 4, + "2.1.1 Unclassified affixes" + ], + [ + 4, + "2.1.2 Inflectional affixes" + ], + [ + 4, + "_2.1.2.1 Simple example_" + ], + [ + 4, + "_2.1.2.2 Optional affix slots_" + ], + [ + 4, + "_2.1.2.3 Multiple templates_" + ], + [ + 4, + "_2.1.2.4 Discontinuous morpheme_" + ], + [ + 4, + "_2.1.2.5 Inflection and categories considerations_" + ], + [ + 4, + "_2.1.2.6 Inflection classes_" + ], + [ + 4, + "2.1.2.6.1 Inflection subclasses" + ], + [ + 4, + "Figure 5. How to Create Isthmus Zapotec Inflection Classes." + ], + [ + 4, + "2.1.2.6.2 Inflection classes and category organization" + ], + [ + 4, + "_2.1.2.7 Agreement and other inflection features_" + ], + [ + 4, + "_2.1.2.8 Inflection classes versus inflection features_" + ], + [ + 4, + "_2.1.2.9 Underspecified inflectional affixes_" + ], + [ + 4, + "2.1.3 Derivational affixes" + ], + [ + 4, + "_2.1.3.1 Major category-changing derivational affixes_" + ], + [ + 4, + "_2.1.3.2 Sub-category-changing derivational affixes_" + ], + [ + 4, + "_2.1.3.3 Non-category-changing derivational affixes_" + ], + [ + 4, + "_2.1.3.4 Inflection class and derivational affixes_" + ], + [ + 4, + "2.1.3.4.1 Inflection class may change" + ], + [ + 4, + "2.1.3.4.2 Inflection class does not change" + ], + [ + 4, + "_2.1.3.5 Inflection Features and Derivational Affixes_" + ], + [ + 4, + "_2.1.3.6 Category-changing derivational affixes and category organization_" + ], + [ + 4, + "_2.1.3.7 Underspecified derivational affixes_" + ], + [ + 4, + "2.1.4 Derivation outside of inflection" + ], + [ + 4, + "2.1.5 Derivation versus inflection" + ], + [ + 4, + "2.1.6 Exception “features”" + ], + [ + 3, + "2.2 Stem compounding" + ], + [ + 4, + "2.2.1 Headed compounds" + ], + [ + 4, + "2.2.2 Non-headed compounds" + ], + [ + 4, + "2.2.3 Incorporation" + ], + [ + 4, + "_2.2.3.1 Incorporation as a simple headed compound_" + ], + [ + 4, + "_2.2.3.2 Incorporation as a headed compound with override_" + ], + [ + 4, + "2.2.4 Affixes between roots in compounds" + ], + [ + 4, + "2.2.5 Compound rules and categories considerations" + ], + [ + 4, + "2.2.6 Restricting the productivity of a compound rule" + ], + [ + 3, + "2.3 Clitics" + ], + [ + 4, + "Figure 23. How to Create a Clitic." + ], + [ + 3, + "2.4 Ad hoc morpheme-oriented rules" + ], + [ + 4, + "2.4.1 Creating morpheme-oriented ad hoc rules" + ], + [ + 4, + "2.4.2 Grouping ad hoc morpheme rules" + ], + [ + 4, + "Figure 25. How to Create a Group of Morpheme-oriented Ad Hoc Rules." + ], + [ + 2, + "3 Morphophonemics" + ], + [ + 3, + "3.1 Overview" + ], + [ + 4, + "3.1.1 Phoneme sets" + ], + [ + 4, + "_3.1.1.1 Phonological features_" + ], + [ + 4, + "_3.1.1.2 Digraphs_" + ], + [ + 4, + "_3.1.1.3 Tones_" + ], + [ + 4, + "3.1.1.3.1 No forms conditioned by tone" + ], + [ + 4, + "3.1.1.3.2 Forms conditioned by tone" + ], + [ + 4, + "3.1.2 Natural classes" + ], + [ + 4, + "Figure 29. How to Create Natural Classes." + ], + [ + 4, + "3.1.3 Allomorph environments" + ], + [ + 4, + "(65) / _ [C] / _ #" + ], + [ + 4, + "3.1.4 Allomorph ordering" + ], + [ + 4, + "_3.1.4.1 Free fluctuation_" + ], + [ + 3, + "3.2 Reduplication" + ], + [ + 4, + "3.2.1 Full reduplication" + ], + [ + 4, + "_3.2.1.1 Writing the pattern for full reduplication_" + ], + [ + 4, + "3.2.2 Partial reduplication" + ], + [ + 4, + "_3.2.2.1 Writing the pattern for partial reduplication_" + ], + [ + 3, + "3.3 Infixation" + ], + [ + 4, + "3.3.1 Writing the infixation environment" + ], + [ + 4, + "3.3.2 Infixation and root and pattern morphology" + ], + [ + 3, + "3.4 Epenthesis" + ], + [ + 3, + "3.5 Metathesis" + ], + [ + 3, + "3.6 Morphemes that may be null" + ], + [ + 3, + "3.7 Non-phonologically conditioned allomorphy" + ], + [ + 4, + "3.7.1 Stem allomorphs conditioned by morpho-syntactic features" + ], + [ + 4, + "Figure 38. How to Create and Use Stem Allomorph Labels." + ], + [ + 4, + "3.7.2 Affix allomorphs conditioned by morpho-syntactic features" + ], + [ + 3, + "3.8 Irregularly inflected forms" + ], + [ + 3, + "3.9 Coalescence" + ], + [ + 3, + "3.10 Ad hoc allomorph-oriented rules" + ], + [ + 4, + "3.10.1 Creating ad hoc allomorph-oriented rules" + ], + [ + 4, + "3.10.2 Grouping ad hoc allomorph rules" + ], + [ + 2, + "4 Lexical entry considerations" + ], + [ + 3, + "4.1 Allomorphs" + ], + [ + 4, + "4.1.1 Null allomorphs" + ], + [ + 4, + "4.1.2 Order of allomorphs within a lexical entry" + ], + [ + 4, + "Figure 43. How to Create a Null Allomorph." + ], + [ + 3, + "4.2 Morpheme types" + ], + [ + 3, + "4.3 Circumfixes" + ], + [ + 3, + "4.4 Senses/Glosses" + ], + [ + 2, + "5 Other considerations" + ], + [ + 3, + "5.1 Exceptional Case for Compound Rules" + ], + [ + 2, + "6 The phonological rule-based parser" + ], + [ + 3, + "6.1 Item and process" + ], + [ + 4, + "6.1.1 Affix process rules" + ], + [ + 4, + "_6.1.1.1 Reduplication as a process_" + ], + [ + 4, + "6.1.1.1.1 Full reduplication as a process" + ], + [ + 4, + "6.1.1.1.2 Partial reduplication as a process" + ], + [ + 4, + "_6.1.1.2 Infixation as a process_" + ], + [ + 4, + "_6.1.1.3 Circumfixation as a process_" + ], + [ + 4, + "6.1.2 Phonological rules" + ], + [ + 4, + "_6.1.2.1 “Regular” phonological rules_" + ], + [ + 4, + "6.1.2.1.1 Epenthesis" + ], + [ + 4, + "Figure 45. Selaru Epenthesis Rule." + ], + [ + 4, + "6.1.2.1.2 Glide becomes a vowel" + ], + [ + 4, + "Figure 46. Selaru Glide Vowel Rule." + ], + [ + 4, + "6.1.2.1.3 Tone processing" + ], + [ + 4, + "Figure 47. How to Handle Awngi Tone." + ], + [ + 4, + "Figure 48. Awngi Docking Rule." + ], + [ + 4, + "Figure 49. Awngi Deletion Rule." + ], + [ + 4, + "6.1.2.1.4 Nasal assimilation" + ], + [ + 4, + "Figure 51. How to Handle the Unspecified Nasal in Indonesian." + ], + [ + 4, + "Figure 57. How to Use an Exception “Feature” for Unspecified Nasal in Indonesian." + ], + [ + 4, + "_6.1.2.2 Constraining application of “regular” phonological rules_" + ], + [ + 4, + "6.1.2.2.1 Rule applies only with certain categories" + ], + [ + 4, + "6.1.2.2.2 Rule applies only with certain properties" + ], + [ + 4, + "_6.1.2.3 Phonological metathesis rules_" + ], + [ + 3, + "6.2 Tips for making the phonological rule-based parser work effectively." + ], + [ + 4, + "6.2.1 Every phoneme used in the orthography must be defined as a phoneme" + ], + [ + 4, + "6.2.2 The phonological features need to uniquely identify each phoneme" + ], + [ + 4, + "6.2.3 Fully specify each phoneme" + ], + [ + 4, + "6.2.4 Features used in a rule should be explicit" + ], + [ + 4, + "6.2.5 Avoid using archiphonemes that are uppercase equivalents of a character in your orthography" + ], + [ + 4, + "6.2.6 Make sure every affix process rule is complete" + ], + [ + 4, + "6.2.7 Natural classes defined by phonemes may not work as expected" + ], + [ + 3, + "6.3 Known limitations" + ], + [ + 4, + "6.3.1 Affixes are tried only once per word" + ], + [ + 4, + "6.3.2 Natural classes defined by segments may or may not work as expected" + ], + [ + 4, + "6.3.3 Ambiguous digraphs and multigraphs may not work as expected" + ], + [ + 2, + "References" + ] + ], + "headings": 140 + } +} diff --git a/tools/reporting.py b/tools/reporting.py new file mode 100644 index 00000000..c7cce459 --- /dev/null +++ b/tools/reporting.py @@ -0,0 +1,213 @@ +"""Canonical quality report model and renderers.""" + +from __future__ import annotations + +import json +from collections.abc import Iterable +from dataclasses import asdict, dataclass +from types import MappingProxyType + +from issue_catalog import ISSUE_CATALOG, policy_for + +LABELS = MappingProxyType({code: policy.label for code, policy in ISSUE_CATALOG.items()}) + + +@dataclass(frozen=True) +class Issue: + code: str + message: str + path: str = "" + fatal: bool | None = None + provenance: str | None = None + detail: object = None + + def __post_init__(self) -> None: + original_code = self.code + code, policy = policy_for(original_code) + unknown = code == "unknown_issue" and original_code != code + if unknown: + object.__setattr__(self, "message", f"[{original_code}] {self.message}") + object.__setattr__(self, "code", code) + # The legacy constructor accepts these fields for source compatibility, + # but policy is always selected solely by the catalog code. + object.__setattr__(self, "fatal", policy.fatal) + object.__setattr__(self, "provenance", policy.provenance) + + @property + def severity(self) -> str: + return "error" if self.fatal else "warning" + + @property + def label(self) -> str: + return ISSUE_CATALOG[self.code].label + + def as_dict(self) -> dict: + value = asdict(self) + value.update(label=self.label, severity=self.severity) + return value + + +class Report: + def __init__(self, issues: Iterable[Issue] = (), metadata: dict | None = None) -> None: + self.issues = list(issues) + self.metadata = dict(metadata or {}) + + def add(self, issue: Issue) -> Issue: + self.issues.append(issue) + return issue + + def extend(self, issues: Iterable[Issue]) -> None: + self.issues.extend(issues) + + @property + def fatal(self) -> bool: + return any(issue.fatal for issue in self.issues) + + def as_dict(self) -> dict: + by_code: dict[str, int] = {} + for issue in self.issues: + by_code[issue.code] = by_code.get(issue.code, 0) + 1 + return { + "corpus": self.metadata, + "summary": { + "total": len(self.issues), + "fatal": sum(issue.fatal for issue in self.issues), + "advisory": sum(not issue.fatal for issue in self.issues), + "by_code": by_code, + }, + "issues": [issue.as_dict() for issue in self.issues], + } + + def to_readme(self) -> str: + lines = ["## Quality report", "", "| Check | Severity | Count |", "| --- | --- | ---: |"] + counts: dict[tuple[str, str], int] = {} + for issue in self.issues: + key = (issue.label, issue.severity) + counts[key] = counts.get(key, 0) + 1 + for (label, severity), count in sorted(counts.items()): + lines.append(f"| {label} | {severity} | {count} |") + if not counts: + lines.append("| None | — | 0 |") + return "\n".join(lines) + + def to_json(self) -> str: + return json.dumps(self.as_dict(), indent=2, ensure_ascii=False) + "\n" + + @staticmethod + def _markdown_cell(value: object, *, code: bool = False) -> str: + if value is None or value == "": + return "—" + if not isinstance(value, str): + value = json.dumps(value, ensure_ascii=False) + escaped = (value.replace("&", "&").replace("<", "<") + .replace(">", ">").replace("\\", "\") + .replace("`", "`").replace("[", "[") + .replace("]", "]").replace("|", "\\|") + .replace("\r\n", "
").replace("\r", "
") + .replace("\n", "
")) + if code: + return f"`{escaped}`" + return escaped + + @staticmethod + def _repair_context(issue: Issue) -> str: + if issue.provenance != "source": + return "exporter" + normalized = issue.path.replace("\\", "/").casefold() + if normalized.startswith("pdf/") or normalized.endswith(".pdf"): + return "PDF source" + return "RoboHelp" + + def to_markdown(self) -> str: + """Render the canonical report for source authors and maintainers.""" + data = self.as_dict() + corpus = data["corpus"] + summary = data["summary"] + lines = [ + "# Author quality report", "", + ( + "This report explains every source or export finding and how to repair it. " + "Machine-readable detail is in [author-report.json](author-report.json)." + ), "", + "## Corpus", "", "| Item | Value |", "| --- | ---: |", + f"| Source ref | {self._markdown_cell(corpus.get('source_ref'), code=True)} |", + f"| CHMs | {corpus.get('chm_count', 0)} |", + f"| Topics | {corpus.get('topic_count', 0)} |", + f"| Images | {corpus.get('image_count', 0)} |", + f"| PDFs | {corpus.get('pdf_count', 0)} |", "", + "## Summary", "", "| Severity | Count |", "| --- | ---: |", + f"| Fatal errors | {summary['fatal']} |", + f"| Advisories | {summary['advisory']} |", + f"| Total | {summary['total']} |", "", + ] + grouped: dict[str, list[Issue]] = {} + for issue in self.issues: + grouped.setdefault(issue.code, []).append(issue) + for code in sorted( + grouped, + key=lambda item: ( + not ISSUE_CATALOG[item].fatal, ISSUE_CATALOG[item].label.casefold(), item, + ), + ): + policy = ISSUE_CATALOG[code] + issues = grouped[code] + contexts = {self._repair_context(issue) for issue in issues} + where = "/".join(sorted(contexts)) + lines.extend([ + f"## {policy.label} (`{code}`)", "", + f"- **Severity:** {'fatal error' if policy.fatal else 'advisory'}", + f"- **Owner:** {where}", + f"- **Count:** {len(issues)}", + f"- **How to fix in {where}:** {policy.guidance}", "", + "| Source or generated path | Problem | Evidence |", + "| --- | --- | --- |", + ]) + for issue in issues: + lines.append( + f"| {self._markdown_cell(issue.path, code=True)} " + f"| {self._markdown_cell(issue.message)} " + f"| {self._markdown_cell(issue.detail)} |" + ) + lines.append("") + if not grouped: + lines.extend(["## Findings", "", "No findings.", ""]) + return "\n".join(lines) + + def to_console(self) -> str: + fatal = [issue for issue in self.issues if issue.fatal] + advisory = [issue for issue in self.issues if not issue.fatal] + lines = [( + f"quality report: {len(self.issues)} issue(s), " + f"{len(fatal)} fatal, {len(advisory)} advisory" + )] + lines.append(f"fatal issues ({len(fatal)}):") + if not fatal: + lines.append(" none") + for issue in fatal: + where = f" [{issue.path}]" if issue.path else "" + lines.append(f" FATAL {issue.label}{where}: {issue.message}") + + counts: dict[tuple[str, str], int] = {} + for issue in advisory: + key = (issue.code, issue.label) + counts[key] = counts.get(key, 0) + 1 + if counts: + lines.append( + f"advisories: {len(advisory)} issue(s) in {len(counts)} kind(s); " + "see author-report.json for details" + ) + for (_, label), count in sorted( + counts.items(), key=lambda item: (item[0][1], item[0][0]) + ): + lines.append(f" WARN {label}: {count}") + else: + lines.append("advisories: none") + return "\n".join(lines) + + +def make_issue(code: str, message: str, path: str = "", detail: object = None) -> Issue: + """Construct an issue using the catalog, safely handling new producer codes.""" + return Issue(str(code), str(message), str(path), detail=detail) + + +__all__ = ["LABELS", "Issue", "Report", "make_issue"] diff --git a/tools/source_safety.py b/tools/source_safety.py new file mode 100644 index 00000000..7f7f9bcc --- /dev/null +++ b/tools/source_safety.py @@ -0,0 +1,158 @@ +"""Safe, deterministic discovery of repository source files.""" + +from __future__ import annotations + +import os +import stat +from pathlib import Path + + +class SourceSafetyError(ValueError): + """A source root or source input violates the repository boundary.""" + + +def _absolute_lexical(path: Path | str) -> Path: + return Path(os.path.abspath(os.fspath(Path(path).expanduser()))) + + +def _is_link(path: Path) -> bool: + is_junction = getattr(path, "is_junction", lambda: False) + return path.is_symlink() or is_junction() + + +def _first_link(path: Path) -> Path | None: + current = Path(path.anchor) + for part in path.parts[1:]: + current /= part + if _is_link(current): + return current + return None + + +def first_link_in_path(path: Path | str) -> Path | None: + """Return the first symlink/junction in a lexical path chain.""" + return _first_link(_absolute_lexical(path)) + + +def validate_source_tree(root: Path) -> Path: + """Validate a lexical tree without following links or escaping root.""" + lexical_root = _absolute_lexical(root) + if (link := first_link_in_path(lexical_root)) is not None: + raise SourceSafetyError(f"refusing symlink/junction tree root: {link}") + try: + resolved_root = lexical_root.resolve(strict=True) + except OSError as exc: + raise SourceSafetyError(f"source tree does not exist: {root}") from exc + if not resolved_root.is_dir(): + raise SourceSafetyError(f"source tree is not a directory: {root}") + + pending = [lexical_root] + while pending: + current = pending.pop() + try: + entries = sorted(current.iterdir(), key=lambda item: (item.name.casefold(), item.name)) + except OSError as exc: + raise SourceSafetyError(f"cannot inspect source tree: {current}") from exc + for entry in entries: + if (link := first_link_in_path(entry)) is not None: + raise SourceSafetyError(f"refusing symlink/junction in source tree: {link}") + try: + mode = entry.stat(follow_symlinks=False).st_mode + except OSError as exc: + raise SourceSafetyError(f"cannot inspect source tree entry: {entry}") from exc + if stat.S_ISDIR(mode): + _resolved_inside(entry, resolved_root) + pending.append(entry) + elif stat.S_ISREG(mode): + _resolved_inside(entry, resolved_root) + return lexical_root + + +def _resolved_inside(path: Path, root: Path) -> Path: + try: + resolved = path.resolve(strict=True) + except OSError as exc: + raise SourceSafetyError(f"cannot resolve source path: {path}") from exc + if resolved != root and root not in resolved.parents: + raise SourceSafetyError(f"source path resolves outside repository root: {path}") + return resolved + + +def discover_source_files( + root: Path, + *, + suffixes: set[str], + recursive: bool, + exclude_dirs: set[str] | None = None, +) -> list[Path]: + """Discover regular source files without following links. + + Candidate files and directories that will be traversed are checked for + links and for a resolved path outside the resolved root. Irrelevant files + and excluded directories are ignored before those checks, so an unrelated + link cannot abort discovery or become an input boundary. + """ + output_root = ( + Path(os.path.normpath(os.fspath(Path(root).expanduser()))) + if not Path(root).expanduser().is_absolute() + else _absolute_lexical(root) + ) + lexical_root = _absolute_lexical(root) + if _first_link(lexical_root) is not None: + raise SourceSafetyError(f"refusing symlink/junction source root: {root}") + try: + resolved_root = lexical_root.resolve(strict=True) + except OSError as exc: + raise SourceSafetyError(f"source root does not exist: {root}") from exc + if not resolved_root.is_dir(): + raise SourceSafetyError(f"source root is not a directory: {root}") + + wanted = {suffix.casefold() if suffix.startswith(".") else f".{suffix.casefold()}" + for suffix in suffixes} + # Repository metadata is never a source tree, even when callers do not + # provide an explicit exclusion set. Additional exclusions are additive; + # callers cannot accidentally opt .git back into traversal. + excluded = {".git"} | {name.casefold() for name in (exclude_dirs or set())} + pending = [lexical_root] + discovered: list[tuple[Path, Path]] = [] + while pending: + current = pending.pop() + try: + entries = sorted(current.iterdir(), key=lambda item: (item.name.casefold(), item.name)) + except OSError as exc: + raise SourceSafetyError(f"cannot inspect source directory: {current}") from exc + for entry in entries: + if entry.name.casefold() in excluded: + continue + if _is_link(entry): + if entry.suffix.casefold() in wanted and not entry.is_dir(): + raise SourceSafetyError(f"refusing symlink/junction source input: {entry}") + continue + mode = entry.stat(follow_symlinks=False).st_mode + if stat.S_ISDIR(mode): + if recursive: + _resolved_inside(entry, resolved_root) + pending.append(entry) + continue + if not stat.S_ISREG(mode): + continue + if entry.suffix.casefold() not in wanted: + continue + _resolved_inside(entry, resolved_root) + discovered.append((entry.relative_to(lexical_root), output_root / entry.relative_to(lexical_root))) + + return sorted( + (path for _, path in discovered), + key=lambda item: ( + item.relative_to(output_root).as_posix().casefold(), + item.relative_to(output_root).as_posix(), + ), + ) + + +__all__ = [ + "SourceSafetyError", + "discover_source_files", + "first_link_in_path", + "validate_source_tree", +] diff --git a/tools/survey.py b/tools/survey.py new file mode 100644 index 00000000..1f7ce4e6 --- /dev/null +++ b/tools/survey.py @@ -0,0 +1,429 @@ +"""Survey the FwHelps corpus: what is actually in the CHM and the PDFs. + +This is a read-only reconnaissance pass, not the converter. It answers the +questions we need settled before designing the markdown export: + - how many topics, how big, how deep is the TOC + - which topics are reachable from the TOC and which are orphans + - what HTML constructs and CSS classes actually appear (what must map to md) + - how many internal links resolve, and where the broken ones point + - which PDFs carry real text vs. scanned images + +Usage: python tools/survey.py [--repo .] [--work DIR] [--json report.json] +""" + +from __future__ import annotations + +import argparse +import collections +import html +import json +import re +import sys +from html.parser import HTMLParser +from pathlib import Path +from urllib.parse import unquote, urldefrag + +sys.path.insert(0, str(Path(__file__).parent)) +from chm_extract import extract + +# ---------------------------------------------------------------- sitemap --- + +class SitemapParser(HTMLParser): + """Parses the HTML Help sitemap format shared by .hhc (TOC) and .hhk (index). + + Structure is
  • ...
      ...
  • . + Nesting of
      gives TOC depth; each holds one or more Name/Local + pairs. + """ + + def __init__(self): + super().__init__(convert_charrefs=True) + self.depth = 0 + self.entries = [] # {depth, params: [(name, value)]} + self._current = None + + def handle_starttag(self, tag, attrs): + a = {k.lower(): (v or "") for k, v in attrs} + if tag == "ul": + self.depth += 1 + elif tag == "object": + self._current = [] + elif tag == "param" and self._current is not None: + self._current.append((a.get("name", "").lower(), a.get("value", ""))) + + def handle_endtag(self, tag): + if tag == "ul": + self.depth = max(0, self.depth - 1) + elif tag == "object" and self._current is not None: + if self._current: + self.entries.append({"depth": self.depth, "params": self._current}) + self._current = None + + +def parse_toc(path): + """Return flat TOC nodes with a breadcrumb trail assembled from
        depth.""" + p = SitemapParser() + p.feed(path.read_text(encoding="cp1252", errors="replace")) + + nodes, trail = [], {} + for e in p.entries: + params = dict(e["params"]) + name = params.get("name", "").strip() + local = unquote(params.get("local", "")).replace("\\", "/").strip() + depth = e["depth"] + trail[depth] = name + for d in [d for d in trail if d > depth]: + del trail[d] + nodes.append({ + "title": name, + "href": urldefrag(local)[0] if local else "", + "depth": depth, + "breadcrumb": [trail[d] for d in sorted(trail) if trail.get(d)], + "is_container": not local, + }) + return nodes + + +def parse_index(path): + """Return .hhk keyword entries: {keyword, targets: [(label, href)]}.""" + p = SitemapParser() + p.feed(path.read_text(encoding="cp1252", errors="replace")) + + out = [] + for e in p.entries: + keyword, targets, pending = None, [], None + for k, v in e["params"]: + if k == "name": + if keyword is None: + keyword = v.strip() + else: + pending = v.strip() + elif k == "local": + href = unquote(v).replace("\\", "/").strip() + targets.append((pending or keyword or "", urldefrag(href)[0])) + pending = None + if keyword: + out.append({"keyword": keyword, "targets": targets}) + return out + + +# ------------------------------------------------------------------ topic --- + +class TopicParser(HTMLParser): + """Collects structural facts about one topic: tags, classes, links, text.""" + + def __init__(self): + super().__init__(convert_charrefs=True) + self.title = "" + self.meta = {} + self.tags = collections.Counter() + self.classes = collections.Counter() + self.styles = collections.Counter() + self.links = [] + self.images = [] + self.headings = [] + self.scripts = [] + self.text_parts = [] + self._in_title = False + self._in_body = False + self._heading = None + self._heading_text = [] + + def handle_starttag(self, tag, attrs): + a = {k.lower(): (v or "") for k, v in attrs} + if tag == "title": + self._in_title = True + return + if tag == "body": + self._in_body = True + if tag == "meta": + key = a.get("name") or a.get("http-equiv") + if key: + self.meta[key.lower()] = a.get("content", "") + return + if tag == "script": + self.scripts.append(a.get("src", "")) + return + if not self._in_body: + return + + self.tags[tag] += 1 + for cls in a.get("class", "").split(): + self.classes[tag + "." + cls] += 1 + for decl in a.get("style", "").split(";"): + prop = decl.split(":")[0].strip().lower() + if prop: + self.styles[prop] += 1 + if tag == "a" and a.get("href"): + self.links.append(a["href"]) + if tag == "img": + self.images.append(a.get("src", "")) + if re.fullmatch(r"h[1-6]", tag): + self._heading = tag + self._heading_text = [] + + def handle_endtag(self, tag): + if tag == "title": + self._in_title = False + elif tag == self._heading: + self.headings.append([tag, "".join(self._heading_text).strip()]) + self._heading = None + + def handle_data(self, data): + if self._in_title: + self.title += data + elif self._in_body: + self.text_parts.append(data) + if self._heading: + self._heading_text.append(data) + + @property + def text(self): + return re.sub(r"\s+", " ", "".join(self.text_parts)).strip() + + +def survey_topics(root): + topics = [] + agg = { + "tags": collections.Counter(), + "classes": collections.Counter(), + "styles": collections.Counter(), + "scripts": collections.Counter(), + "charsets": collections.Counter(), + "generators": collections.Counter(), + "meta_names": collections.Counter(), + } + for f in sorted(root.rglob("*.htm")): + raw = f.read_bytes() + p = TopicParser() + p.feed(raw.decode("cp1252", errors="replace")) + + charset = "" + m = re.search(rb"charset=([\w-]+)", raw, re.IGNORECASE) + if m: + charset = m.group(1).decode("ascii", "replace").lower() + + kw = p.meta.get("rh-index-keywords", "") + topics.append({ + "path": f.relative_to(root).as_posix(), + "title": html.unescape(p.title).strip(), + "bytes": len(raw), + "words": len(p.text.split()), + "keywords": [k.strip() for k in kw.split(",") if k.strip()], + "headings": p.headings, + "links": p.links, + "images": p.images, + "tables": p.tags.get("table", 0), + "charset": charset, + "non_ascii_bytes": sum(1 for b in raw if b > 0x7F), + }) + agg["tags"].update(p.tags) + agg["classes"].update(p.classes) + agg["styles"].update(p.styles) + agg["scripts"].update(p.scripts) + agg["meta_names"].update(p.meta.keys()) + agg["charsets"][charset or "(none)"] += 1 + agg["generators"][p.meta.get("generator", "(none)")] += 1 + return topics, agg + + +def classify_links(topics, root): + kinds = collections.Counter() + broken = [] + external = collections.Counter() + for t in topics: + base = (root / t["path"]).parent + for href in t["links"]: + low = href.lower() + if low.startswith(("http://", "https://")): + kinds["external"] += 1 + external[re.sub(r"^https?://([^/]+).*", r"\1", href, flags=re.IGNORECASE)] += 1 + elif low.startswith("mailto:"): + kinds["mailto"] += 1 + elif low.startswith(("javascript:", "#")): + kinds["anchor/script"] += 1 + else: + kinds["internal"] += 1 + target = urldefrag(unquote(href))[0] + if target and not (base / target).exists(): + broken.append([t["path"], href]) + return {"kinds": dict(kinds), "broken": broken, "external_hosts": dict(external)} + + +# -------------------------------------------------------------------- pdf --- + +def survey_pdfs(repo): + try: + import fitz # PyMuPDF + except ImportError: + return [{"error": "PyMuPDF (fitz) not installed; skipping PDF survey"}] + + out = [] + for f in sorted(repo.rglob("*.pdf")): + if ".git" in f.parts: + continue + rec = {"path": f.relative_to(repo).as_posix(), "bytes": f.stat().st_size} + try: + with fitz.open(f) as doc: + rec["pages"] = doc.page_count + rec["toc_entries"] = len(doc.get_toc()) + rec["encrypted"] = doc.is_encrypted + sample = list(range(min(doc.page_count, 12))) + chars = images = 0 + for i in sample: + page = doc[i] + chars += len(page.get_text("text").strip()) + images += len(page.get_images(full=True)) + rec["chars_per_page"] = round(chars / max(1, len(sample))) + rec["images_per_page"] = round(images / max(1, len(sample)), 1) + rec["text_layer"] = ( + "yes" if rec["chars_per_page"] > 200 + else "sparse" if rec["chars_per_page"] > 20 + else "NONE (scanned?)" + ) + rec["metadata_title"] = (doc.metadata or {}).get("title") or "" + except (ValueError, RuntimeError, TypeError) as exc: + rec["error"] = f"{type(exc).__name__}: {exc}" + out.append(rec) + return out + + +# ------------------------------------------------------------------- main --- + +W = 78 + + +def rule(title=""): + print("\n" + ((("-- " + title + " ").ljust(W, "-")) if title else "-" * W)) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--repo", default=".", type=Path) + ap.add_argument("--work", default=None, type=Path, + help="scratch dir for CHM extraction (default: /.chm-work)") + ap.add_argument("--json", default=None, type=Path) + ap.add_argument("--reuse", action="store_true", + help="reuse an existing extraction instead of re-extracting") + args = ap.parse_args() + + repo = args.repo.resolve() + work = (args.work or repo / ".chm-work").resolve() + chm = repo / "FieldWorks_Language_Explorer_Help.chm" + + if args.reuse and any(work.rglob("*.htm")): + tool = "(reused existing extraction)" + else: + tool = extract(chm, work) + print(f"CHM extracted with: {tool}\n -> {work}") + + hhc = next(work.rglob("*.hhc"), None) + hhk = next(work.rglob("*.hhk"), None) + toc = parse_toc(hhc) if hhc else [] + idx = parse_index(hhk) if hhk else [] + + topics, agg = survey_topics(work) + links = classify_links(topics, work) + pdfs = survey_pdfs(repo) + + by_path = {t["path"]: t for t in topics} + toc_hrefs = {n["href"] for n in toc if n["href"]} + orphans = sorted(set(by_path) - toc_hrefs) + dangling = sorted(h for h in toc_hrefs if h not in by_path) + + version = "" + if hhk: + m = re.search(r"_(\d+\.\d+)\.hhk$", hhk.name) + version = m.group(1) if m else "" + + words = sorted(t["words"] for t in topics) + + rule("CORPUS") + print(" help version (from .hhk) {}".format(version or "?")) + print(f" topics (.htm) {len(topics)}") + print(f" total body words {sum(words):,}") + print(" total bytes {:,}".format(sum(t["bytes"] for t in topics))) + print(f" words/topic min/med/max {words[0]} / {words[len(words) // 2]} / {words[-1]}") + biggest = max(topics, key=lambda t: t["words"]) + print(" largest topic {} words {}".format(biggest["words"], biggest["path"])) + print(f" topics under 30 words {sum(1 for w in words if w < 30)}") + print(f" topics over 1000 words {sum(1 for w in words if w > 1000)}") + print(" distinct images referenced {}".format(len({i for t in topics for i in t["images"]}))) + print(" topics containing tables {}".format(sum(1 for t in topics if t["tables"]))) + + rule("TOC (.hhc)") + print(" nodes {} ({} containers, {} topic links)".format( + len(toc), sum(1 for n in toc if n["is_container"]), len(toc_hrefs))) + print(" max depth {}".format(max((n["depth"] for n in toc), default=0))) + print(f" topics NOT in TOC {len(orphans)}") + print(f" TOC links with no file {len(dangling)}") + for o in orphans[:10]: + print(f" orphan: {o}") + for d in dangling[:10]: + print(f" dangling: {d}") + + rule("INDEX (.hhk)") + print(f" keywords {len(idx)}") + print(" keyword -> topic refs {}".format(sum(len(e["targets"]) for e in idx))) + print(" topics w/ rh-index-keywords {} / {}".format( + sum(1 for t in topics if t["keywords"]), len(topics))) + allkw = collections.Counter(k for t in topics for k in t["keywords"]) + print(f" distinct meta keywords {len(allkw)}") + print(" most common: " + ", ".join(f"{k}({n})" for k, n in allkw.most_common(6))) + + rule("HTML CONSTRUCTS (what the converter must handle)") + print(" tags: " + ", ".join(f"{t}:{n}" for t, n in agg["tags"].most_common(24))) + print(" classes: " + ", ".join(f"{c}:{n}" for c, n in agg["classes"].most_common(20))) + print(" inline styles: " + ", ".join(f"{s}:{n}" for s, n in agg["styles"].most_common(10))) + print(" charsets: " + ", ".join(f"{c}:{n}" for c, n in agg["charsets"].most_common())) + print(" generators: " + ", ".join(f"{g}:{n}" for g, n in agg["generators"].most_common(3))) + print(" scripts: " + ", ".join(f"{s}:{n}" for s, n in agg["scripts"].most_common(5))) + print(" meta names: " + ", ".join(f"{m}:{n}" for m, n in agg["meta_names"].most_common(14))) + print(" topics with non-ASCII bytes: {}".format(sum(1 for t in topics if t["non_ascii_bytes"]))) + print(" total non-ASCII bytes: {:,}".format(sum(t["non_ascii_bytes"] for t in topics))) + + rule("LINKS") + for k, v in sorted(links["kinds"].items(), key=lambda kv: -kv[1]): + print(f" {k:<16} {v}") + print(" BROKEN internal {}".format(len(links["broken"]))) + for src, href in links["broken"][:10]: + print(f" {src} -> {href}") + print(" external hosts: " + ", ".join( + f"{h}({n})" + for h, n in sorted(links["external_hosts"].items(), key=lambda kv: -kv[1])[:8])) + + rule("PDFs") + print(" {:>5} {:>6} {:>7} {:<14} {}".format("pages", "ch/pg", "img/pg", "text", "path")) + for p in pdfs: + if "pages" not in p: + print(" {:>5} {:>6} {:>7} {:<14} {} {}".format( + "?", "?", "?", "ERROR", p.get("path", ""), p.get("error", ""))) + continue + print(" {:>5} {:>6} {:>7} {:<14} {}".format( + p["pages"], p["chars_per_page"], p["images_per_page"], p["text_layer"], p["path"])) + ok = [p for p in pdfs if p.get("pages")] + if ok: + print(" total pages {}, with PDF bookmarks: {}/{}".format( + sum(p["pages"] for p in ok), + sum(1 for p in ok if p.get("toc_entries")), len(ok))) + + rule() + if args.json: + args.json.parent.mkdir(parents=True, exist_ok=True) + args.json.write_text(json.dumps({ + "version": version, + "toc": toc, + "index": idx, + "topics": topics, + "orphans": orphans, + "dangling": dangling, + "links": links, + "pdfs": pdfs, + "aggregate": {k: dict(v) for k, v in agg.items()}, + }, indent=1), encoding="utf-8") + print(f"full report -> {args.json}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/test_chm_extract.py b/tools/test_chm_extract.py new file mode 100644 index 00000000..e4b2990b --- /dev/null +++ b/tools/test_chm_extract.py @@ -0,0 +1,437 @@ +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +import chm_extract +from output_fs import OutputPathError +from source_safety import SourceSafetyError + + +class ChmExtractionIsolationTests(unittest.TestCase): + def _chm(self, directory: Path) -> Path: + chm = directory / "sample.chm" + chm.write_bytes(b"not a real archive") + return chm + + def test_rejects_destination_equal_to_source_parent(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + with self.assertRaises(OutputPathError): + chm_extract.extract(chm, root) + + def test_rejects_destination_tree_that_contains_source_chm(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + with self.assertRaises(OutputPathError): + chm_extract.extract(chm, root / ".." / root.name) + + def test_rejects_destination_link_chain_before_staging(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "work" / "out" + with mock.patch.object( + chm_extract, "first_link_in_path", side_effect=[None, destination.parent] + ), self.assertRaises(OutputPathError): + chm_extract._validate_extract_paths(chm, destination) + + def test_rejects_source_chm_with_linked_ancestor(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + with mock.patch.object(chm_extract, "first_link_in_path", return_value=root.parent), \ + self.assertRaises(chm_extract.ExtractError): + chm_extract._validate_extract_paths(chm, root / "work") + + def test_rejects_actual_destination_symlink_chain_when_available(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + outside = root.parent / (root.name + "-destination") + outside.mkdir() + link = root / "linked-work" + try: + link.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + with self.assertRaises(OutputPathError): + chm_extract._validate_extract_paths(chm, link / "out") + finally: + outside.rmdir() + + def test_rejects_actual_source_symlink_ancestor_when_available(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + outside = root.parent / (root.name + "-source") + outside.mkdir() + (outside / "sample.chm").write_bytes(b"not a real archive") + link = root / "linked-source" + try: + link.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + (outside / "sample.chm").unlink() + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + with self.assertRaises(chm_extract.ExtractError): + chm_extract._validate_extract_paths(link / "sample.chm", root / "work") + finally: + (outside / "sample.chm").unlink() + outside.rmdir() + + def test_allows_descendant_destination_that_does_not_contain_source(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "work" / "extracted" + + def backend(_chm: Path, outdir: Path) -> str: + (outdir / "index.htm").write_text("ok", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("tool", chm_extract.extract(chm, destination)) + + def test_brs_companion_artifact_is_known_but_unknown_filename_is_fatal(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + valid = root / "valid" + valid.mkdir() + (valid / "topic.htm").write_text("topic", encoding="utf-8") + (valid / "Using_Help.brs").write_text("companion", encoding="utf-8") + (valid / "toc.hhc").write_text( + '', encoding="cp1252" + ) + fatal, _ = chm_extract.validate(valid) + self.assertEqual([], fatal) + + invalid = root / "invalid" + invalid.mkdir() + (invalid / "topic.htm").write_text("topic", encoding="utf-8") + (invalid / "Using_Help.brsx").write_text("unknown", encoding="utf-8") + (invalid / "toc.hhc").write_text( + '', encoding="cp1252" + ) + fatal, _ = chm_extract.validate(invalid) + self.assertEqual([], fatal) + + def test_hhc_local_param_is_casefolded_and_attribute_order_independent(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Texts_&_Words.htm").write_text("topic", encoding="cp1252") + (root / "book.hhc").write_text( + "" + "", encoding="cp1252" + ) + fatal, advisory = chm_extract.validate(root) + self.assertEqual([], fatal) + self.assertEqual([], advisory) + + def test_html_href_and_src_references_detect_missing_and_truncated_targets(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.htm").write_text( + "missing" + "", encoding="cp1252" + ) + (root / "clip-image.pn").write_bytes(b"truncated") + (root / "book.hhc").write_text( + "", encoding="cp1252" + ) + fatal, advisory = chm_extract.validate(root) + self.assertTrue(any("clip-image.png" in item for item in fatal)) + self.assertTrue(any("missing.htm" in item for item in advisory)) + + def test_unrelated_asset_extensions_are_not_truncation_failures(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.htm").write_text("page", encoding="cp1252") + for name in ("font.woff", "picture.webp", "data.json", "bundle.map"): + (root / name).write_text("asset", encoding="utf-8") + (root / "book.hhc").write_text( + "", encoding="cp1252" + ) + fatal, _ = chm_extract.validate(root) + self.assertEqual([], fatal) + + def test_unsafe_and_absolute_references_are_fatal_with_specific_codes(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.htm").write_text( + "x" + "" + "x", encoding="cp1252" + ) + (root / "book.hhc").write_text( + "", encoding="cp1252" + ) + fatal, advisory = chm_extract.validate(root) + self.assertFalse(any("unsafe_uri" in item or "path_escape" in item for item in fatal)) + self.assertTrue(any("source_unsafe_uri" in item for item in advisory)) + self.assertTrue(any("source_path_escape" in item for item in advisory)) + + def test_encoded_uri_separators_cannot_hide_unsafe_schemes(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.htm").write_text( + "x" + "x", + encoding="cp1252", + ) + (root / "book.hhc").write_text( + "", encoding="cp1252" + ) + fatal, advisory = chm_extract.validate(root) + self.assertEqual([], fatal) + self.assertGreaterEqual(sum("source_unsafe_uri" in item for item in advisory), 2) + + def test_backends_get_private_empty_dirs_and_only_valid_result_is_promoted(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + seen: list[tuple[str, Path, list[str]]] = [] + + def failed_backend(_chm: Path, outdir: Path) -> str: + seen.append(("failed", outdir, sorted(p.name for p in outdir.iterdir()))) + (outdir / "contamination.htm").write_text("bad", encoding="utf-8") + return "failed-tool" + + def successful_backend(_chm: Path, outdir: Path) -> str: + seen.append(("successful", outdir, sorted(p.name for p in outdir.iterdir()))) + (outdir / "index.htm").write_text("good", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "successful-tool" + + with mock.patch.object(chm_extract, "_sevenzip", failed_backend), \ + mock.patch.object(chm_extract, "_chmlib", successful_backend), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("successful-tool", chm_extract.extract(chm, destination)) + + self.assertEqual(["failed", "successful"], [item[0] for item in seen]) + self.assertEqual([[], []], [item[2] for item in seen]) + self.assertNotEqual(seen[0][1], seen[1][1]) + self.assertTrue(all(item[1].parent == destination.parent for item in seen)) + self.assertEqual("good", (destination / "index.htm").read_text(encoding="utf-8")) + self.assertFalse((destination / "contamination.htm").exists()) + + def test_failed_attempts_leave_existing_destination_untouched(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + destination.mkdir() + marker = destination / "previous.htm" + marker.write_text("keep", encoding="utf-8") + + def invalid_backend(_chm: Path, outdir: Path) -> str: + (outdir / "new.htm").write_text("invalid", encoding="utf-8") + return "invalid-tool" + + with ( + mock.patch.object(chm_extract, "_sevenzip", invalid_backend), + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), + mock.patch.object(chm_extract, "_hh", lambda *_: None), + self.assertRaises(chm_extract.ExtractError), + ): + chm_extract.extract(chm, destination) + + self.assertEqual("keep", marker.read_text(encoding="utf-8")) + self.assertFalse((destination / "new.htm").exists()) + + def test_clean_false_keeps_existing_extraction_files(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + destination.mkdir() + (destination / "old.htm").write_text("old", encoding="utf-8") + + def backend(_chm: Path, outdir: Path) -> str: + (outdir / "new.htm").write_text("new", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("tool", chm_extract.extract(chm, destination, clean=False)) + + self.assertTrue((destination / "old.htm").exists()) + self.assertTrue((destination / "new.htm").exists()) + + def test_clean_false_rejects_linked_existing_descendant_before_copy(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + destination.mkdir() + outside = root.parent / (root.name + "-existing-link") + outside.mkdir() + linked = destination / "linked" + try: + linked.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + outside.rmdir() + self.skipTest("symlinks unavailable") + + def backend(_chm, outdir): + (outdir / "new.htm").write_text("new", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + try: + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None), \ + self.assertRaises(SourceSafetyError): + chm_extract.extract(chm, destination, clean=False) + self.assertFalse((destination / "new.htm").exists()) + finally: + linked.unlink(missing_ok=True) + outside.rmdir() + + def test_advisory_metadata_survives_successful_promotion(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + + def backend(_chm: Path, outdir: Path) -> str: + (outdir / "new.htm").write_text("new", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("tool", chm_extract.extract(chm, destination)) + + self.assertEqual( + ["TOC points at a topic that does not exist: missing.htm"], + chm_extract.extract.advisory, + ) + + def test_called_process_error_falls_back_to_next_backend(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + + def failed_backend(_chm: Path, _outdir: Path) -> str: + raise chm_extract.subprocess.CalledProcessError(7, "fake") + + def successful_backend(_chm: Path, outdir: Path) -> str: + (outdir / "index.htm").write_text("good", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "fallback-tool" + + with mock.patch.object(chm_extract, "_sevenzip", failed_backend), \ + mock.patch.object(chm_extract, "_chmlib", successful_backend), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("fallback-tool", chm_extract.extract(chm, destination)) + + def test_validation_fatal_falls_back_and_truncated_name_is_not_promoted(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + + def truncated_backend(_chm: Path, outdir: Path) -> str: + (outdir / "topic").write_text("truncated", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "truncated-tool" + + def successful_backend(_chm: Path, outdir: Path) -> str: + (outdir / "index.html").write_text("good", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "html-tool" + + with mock.patch.object(chm_extract, "_sevenzip", truncated_backend), \ + mock.patch.object(chm_extract, "_chmlib", successful_backend), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None): + self.assertEqual("html-tool", chm_extract.extract(chm, destination)) + + self.assertTrue((destination / "index.html").exists()) + self.assertFalse((destination / "topic").exists()) + + def test_clean_false_merge_failure_preserves_previous_destination(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + destination.mkdir() + marker = destination / "previous.htm" + marker.write_text("keep", encoding="utf-8") + + def backend(_chm: Path, outdir: Path) -> str: + (outdir / "new.htm").write_text("new", encoding="utf-8") + (outdir / "toc.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None), \ + mock.patch.object( + chm_extract.shutil, + "copytree", + side_effect=[None, OSError("merge failed")], + ), self.assertRaises(OSError): + chm_extract.extract(chm, destination, clean=False) + + self.assertEqual("keep", marker.read_text(encoding="utf-8")) + self.assertFalse((destination / "new.htm").exists()) + + def test_fresh_backend_tree_safety_failure_preserves_existing_destination(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = self._chm(root) + destination = root / "extracted" + destination.mkdir() + (destination / "previous.htm").write_text("keep", encoding="utf-8") + + def backend(_chm: Path, outdir: Path) -> str: + (outdir / "new.htm").write_text("new", encoding="utf-8") + return "unsafe-tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None), \ + mock.patch.object( + chm_extract, "validate_source_tree", + side_effect=SourceSafetyError("unsafe tree"), + ), self.assertRaises(SourceSafetyError): + chm_extract.extract(chm, destination) + + self.assertEqual("keep", (destination / "previous.htm").read_text(encoding="utf-8")) + self.assertFalse((destination / "new.htm").exists()) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_chm_metadata.py b/tools/test_chm_metadata.py new file mode 100644 index 00000000..d68bf581 --- /dev/null +++ b/tools/test_chm_metadata.py @@ -0,0 +1,53 @@ +import tempfile +import unittest +from pathlib import Path + +from chm_metadata import TopicMeta, parse_toc, safe_stem + + +class ChmMetadataTests(unittest.TestCase): + def test_safe_stem_is_stable_and_filesystem_safe(self): + self.assertEqual("Using_Help", safe_stem("Using Help.chm")) + self.assertEqual("FieldWorks_Language_Explorer_Help", safe_stem("FieldWorks_Language_Explorer_Help.chm")) + + def test_topic_meta_collects_heading_links_images_and_related_links(self): + meta = TopicMeta() + meta.feed("""Title +

        Reader heading

        Next + Site +

        Related Topics

        Other + """) + self.assertEqual("Title", meta.title) + self.assertEqual("Reader heading", meta.page_heading) + self.assertEqual(["next.htm", "https://example.test", "other.htm"], meta.links) + self.assertEqual(["img/a.png"], meta.images) + self.assertEqual([("Other", "other.htm")], meta.related) + + def test_parse_toc_returns_deterministic_breadcrumbs(self): + with tempfile.TemporaryDirectory() as raw: + path = Path(raw) / "book.hhc" + path.write_text( + '
        • ' + '
            ' + '
          • ' + '
        ', + encoding="cp1252", + ) + nodes = parse_toc(path) + self.assertEqual(["Root"], nodes[0]["breadcrumb"]) + self.assertEqual(["Root", "Child"], nodes[1]["breadcrumb"]) + + def test_parse_toc_accepts_reversed_casefolded_param_attributes_and_escaped_local(self): + with tempfile.TemporaryDirectory() as raw: + path = Path(raw) / "book.hhc" + path.write_text( + "
        • " + "
        ", encoding="cp1252" + ) + nodes = parse_toc(path) + self.assertEqual("Texts_&_Words.htm", nodes[0]["href"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_convert.py b/tools/test_convert.py new file mode 100644 index 00000000..37d3c481 --- /dev/null +++ b/tools/test_convert.py @@ -0,0 +1,1155 @@ +import hashlib +import json +import os +import shutil +import subprocess +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +import chm_convert +import chm_extract +import convert +import source_safety +from output_fs import ExportBusyError, ExportLock, OutputPathError +from source_safety import SourceSafetyError + + +class ConvertOrchestrationTests(unittest.TestCase): + def test_chm_conversions_sharing_work_root_serialize_extraction(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction_lock = root / "work" / "Using_Help" + with ExportLock(extraction_lock), self.assertRaises(ExportBusyError): + chm_convert.convert_chm( + chm, + root / "work", + root / "first-output", + extractor=lambda *_args: self.fail("extractor should not run"), + ) + + def test_chm_pandoc_scratch_paths_are_unique_and_cleaned(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + seen: list[Path] = [] + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True, exist_ok=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + + def fake_pandoc(_html, scratch): + seen.append(scratch) + return "# source\n", [] + + with mock.patch.object(chm_convert, "run_pandoc", side_effect=fake_pandoc): + chm_convert.convert_chm( + chm, root / "work", root / "first-output", extractor=fake_extract + ) + chm_convert.convert_chm( + chm, root / "work", root / "second-output", extractor=fake_extract + ) + self.assertEqual(2, len(seen)) + self.assertEqual(2, len(set(seen))) + self.assertTrue(all(not path.exists() for path in seen)) + + def test_fatal_build_writes_external_diagnostics_without_promoting(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Using_Help.chm").write_bytes(b"fixture") + diagnostics = root.parent / "diagnostics.json" + + def unsafe_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "topic.md").write_text("# Topic\n", encoding="utf-8") + return { + "chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, + "report": {"unsafe_uri": [["topic.htm", "javascript:x"]]}, + } + + with mock.patch.object(convert, "run_in_private_stage", side_effect=unsafe_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build( + root, root / "export", root / "work", diagnostics=diagnostics + ) + + self.assertFalse(result["promoted"]) + self.assertTrue(diagnostics.exists()) + report = json.loads(diagnostics.read_text(encoding="utf-8")) + self.assertTrue(report["summary"]["fatal"]) + self.assertTrue(any(i["code"] == "unsafe_uri" for i in report["issues"])) + self.assertFalse((root / "export").exists()) + self.assertNotIn(".output-stage-", diagnostics.read_text(encoding="utf-8")) + + def test_handled_conversion_error_writes_sanitized_diagnostics(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Using_Help.chm").write_bytes(b"fixture") + work = root / "work" + diagnostics = root.parent / "conversion-diagnostics.json" + + def failed_chm(_chm, extraction, _destination, **_kwargs): + raise RuntimeError(f"failed in {extraction}") + + with mock.patch.object(convert, "run_in_private_stage", side_effect=failed_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(root, root / "export", work, diagnostics=diagnostics) + + self.assertFalse(result["promoted"]) + report_text = diagnostics.read_text(encoding="utf-8") + report = json.loads(report_text) + self.assertTrue(any(i["code"] == "chm_failure" for i in report["issues"])) + self.assertNotIn(str(work), report_text) + + def test_diagnostics_path_cannot_overlap_repo_output_or_work(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Using_Help.chm").write_bytes(b"fixture") + for diagnostics in ( + root / "inside.json", root / "out" / "diagnostics.json", + root / "work" / "diagnostics.json", + ): + with self.subTest(diagnostics=diagnostics), self.assertRaises(OutputPathError): + convert.build( + root, root / "out", root / "work", diagnostics=diagnostics + ) + + def test_diagnostics_path_parent_is_created_atomically(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Using_Help.chm").write_bytes(b"fixture") + diagnostics = root.parent / "nested" / "diagnostics.json" + with mock.patch.object(convert, "run_in_private_stage", return_value={ + "chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 0, "images": 0, "report": {}, + }), mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + convert.build(root, root / "export", root / "work", diagnostics=diagnostics) + self.assertTrue(diagnostics.is_file()) + self.assertFalse(list(diagnostics.parent.glob(".*diagnostics*"))) + def test_discovers_all_root_chms_in_case_insensitive_order(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"") + (repo / "FieldWorks_Language_Explorer_Help.chm").write_bytes(b"") + (repo / "nested").mkdir() + (repo / "nested" / "ignored.chm").write_bytes(b"") + self.assertEqual( + ["FieldWorks_Language_Explorer_Help.chm", "Using_Help.chm"], + [p.name for p in convert.discover_chms(repo)], + ) + + def test_rejects_symlink_root_chm(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + target = repo / "real.chm" + target.write_bytes(b"") + link = repo / "linked.chm" + try: + link.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(ValueError): + convert.discover_chms(repo) + + def test_source_url_template_contains_source_ref_for_pdf_call(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"") + with mock.patch.object(convert, "run_in_private_stage", return_value={ + "chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 0, "images": 0, "report": {}, + }) as chm, \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )) as pdf: + result = convert.build(repo, repo / "export", repo / "work", source_ref="feature/x") + self.assertEqual(1, chm.call_count) + self.assertTrue(pdf.call_args.kwargs["source_url"].startswith( + "https://github.com/sillsdev/FwHelps/blob/feature/x/" + )) + self.assertIn("report", result) + self.assertEqual(1, result["report"]["corpus"]["chm_count"]) + + def test_two_chm_fixture_is_emitted_under_separate_namespaces(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + for name in ("FieldWorks_Language_Explorer_Help.chm", "Using_Help.chm"): + (repo / name).write_bytes(b"fixture") + + def fake_chm(chm, _work, destination, **_kwargs): + (destination / "index.md").parent.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text(f"# {chm.stem}\n", encoding="utf-8") + return {"chm": chm.name, "stem": destination.name, "toc": [], + "topics": 1, "images": 0, "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=fake_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, repo / "export", repo / "work") + self.assertTrue(result["promoted"]) + self.assertTrue((repo / "export" / "chm" / "Using_Help" / "index.md").exists()) + self.assertTrue((repo / "export" / "chm" / "FieldWorks_Language_Explorer_Help" / "index.md").exists()) + readme = (repo / "export" / "README.md").read_text(encoding="utf-8") + self.assertIn("Root CHMs (auto-discovered)", readme) + self.assertIn("FieldWorks_Language_Explorer_Help.chm", readme) + self.assertIn("Using_Help.chm", readme) + + def test_successful_build_does_not_publish_pdf_staging_lockfile(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def fake_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text("# Topic\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=fake_chm): + result = convert.build(repo, repo / "export", repo / "work") + + self.assertTrue(result["promoted"]) + self.assertFalse(list((repo / "export").rglob(".fwhelps-export-*.lock"))) + self.assertTrue((repo / "export" / "author-report.md").is_file()) + self.assertTrue((repo / "export" / "author-report.json").is_file()) + author_report = (repo / "export" / "author-report.md").read_text(encoding="utf-8") + self.assertIn("# Author quality report", author_report) + self.assertIn("## Summary", author_report) + self.assertNotIn("Build validation is in progress.", author_report) + readme = (repo / "export" / "README.md").read_text(encoding="utf-8") + self.assertIn("[author-report.md](author-report.md)", readme) + self.assertIn("[author-report.json](author-report.json)", readme) + + def test_chm_private_stage_does_not_publish_destination_lockfile(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True, exist_ok=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# Topic\n", [])): + chm_convert.run_in_private_stage( + chm, root / "work", root / "stage" / "chm" / "Using_Help", + extractor=fake_extract, + ) + + self.assertFalse(list((root / "stage" / "chm").rglob(".fwhelps-export-*.lock"))) + + def test_direct_chm_conversion_preserves_related_images_and_disambiguates_titles(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "a.htm").write_text( + "Same

        First heading

        " + "

        Related Topics

        Other" + "", encoding="cp1252" + ) + (extraction / "b.htm").write_text( + "Same

        Second heading

        ", encoding="cp1252" + ) + (extraction / "pic.png").write_bytes(b"png") + (extraction / "book.hhc").write_text( + '
        • ' + '
        ', encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=( + "# source\n\n- - child\n", [] + )): + result = chm_convert.convert_chm(chm, root / "work", root / "out", + extractor=fake_extract, + source_url_base="https://docs.example/help") + a = (root / "out" / "a.md").read_text(encoding="utf-8") + b = (root / "out" / "b.md").read_text(encoding="utf-8") + self.assertIn("# First heading", a) + self.assertIn("# Second heading", b) + self.assertIn("related:", a) + self.assertIn( + 'source_url: "https://docs.example/help/index.htm#t=a.htm"', a + ) + self.assertIn("b.md", a) + self.assertNotIn("- -", a) + self.assertIn("child", a) + self.assertEqual(b"png", (root / "out" / "pic.png").read_bytes()) + self.assertEqual(2, result["topics"]) + + def test_failed_validation_preserves_previous_output_and_stage_is_cleaned(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + out, work = repo / "export", repo / "work" + out.mkdir() + (out / "sentinel.txt").write_text("previous", encoding="utf-8") + + def bad_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "bad.md").write_text("# One\n# Two\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=bad_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, out, work) + self.assertFalse(result["promoted"]) + self.assertEqual("previous", (out / "sentinel.txt").read_text(encoding="utf-8")) + self.assertTrue(not work.exists() or not any( + path.name.startswith(".output-stage-") for path in work.iterdir() + )) + + def test_broken_readme_navigation_is_a_fatal_validation_issue(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + out = repo / "export" + + def bad_nav(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text("# Page\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [{ + "title": "Missing", "href": "missing.htm", "depth": 1, + }], "topics": 1, "images": 0, "topics_paths": ["missing.htm"], "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=bad_nav), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, out, repo / "work") + self.assertFalse(result["promoted"]) + self.assertTrue(any(issue["code"] == "missing_link" for issue in result["report"]["issues"])) + + def test_stale_toc_entry_is_safe_text_and_source_advisory(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def stale(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text("# Page\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [{ + "title": "About Strata Sequences", "href": "About_Strata_Sequences.htm", "depth": 1, + }], "topics": 1, "images": 0, "topics_paths": ["index.htm"], + "report": {"stale_toc_entries": [["About_Strata_Sequences.htm", "missing"]]}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=stale), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, repo / "export", repo / "work") + self.assertTrue(result["promoted"]) + readme = (repo / "export" / "README.md").read_text(encoding="utf-8") + self.assertIn("- **About Strata Sequences**", readme) + self.assertNotIn("About_Strata_Sequences.htm)", readme) + stale_issues = [issue for issue in result["report"]["issues"] + if issue["code"] == "stale_toc_entries"] + self.assertTrue(stale_issues) + + def test_readme_renders_qualified_chm_href_as_safe_text(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def qualified(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text("# Page\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [{ + "title": "Qualified", "href": "Using_Help.chm::/Using_Help.hhc", "depth": 1, + }], "topics": 1, "images": 0, "topics_paths": ["index.htm"], "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=qualified), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, repo / "export", repo / "work") + self.assertTrue(result["promoted"]) + readme = (repo / "export" / "README.md").read_text(encoding="utf-8") + self.assertIn("- **Qualified**", readme) + self.assertNotIn("Using_Help.chm%3A%3A", readme) + + def test_repeated_fixture_builds_have_identical_emitted_bytes(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def fake_chm(chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text(f"# {chm.stem}\n", encoding="utf-8") + return {"chm": chm.name, "stem": destination.name, "toc": [], + "topics": 1, "images": 0, "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=fake_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + first = convert.build(repo, repo / "export", repo / "work") + files = sorted(path.relative_to(repo / "export").as_posix() for path in (repo / "export").rglob("*") if path.is_file()) + before = {name: (repo / "export" / name).read_bytes() for name in files} + second = convert.build(repo, repo / "export", repo / "work") + after = {name: (repo / "export" / name).read_bytes() for name in files} + self.assertTrue(first["promoted"] and second["promoted"]) + self.assertEqual(before, after) + self.assertFalse((repo / ".export.staging").exists()) + + def test_reuse_revalidates_and_preserves_extraction_advisory(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text("Topic

        Topic

        ", encoding="cp1252") + (extraction / "book.hhc").write_text( + '', encoding="cp1252" + ) + (extraction / ".chm-extraction-manifest.json").write_text(json.dumps({ + "schema": 1, + "source_name": chm.name, + "source_sha256": hashlib.sha256(chm.read_bytes()).hexdigest(), + }), encoding="utf-8") + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# Topic\n", [])): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", reuse=True + ) + self.assertTrue(result["report"]["stale_toc_entries"]) + + def test_fresh_and_reused_extraction_preserve_the_same_advisories(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def backend(_chm, outdir): + (outdir / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + (outdir / "book.hhc").write_text( + '', encoding="cp1252" + ) + return "tool" + + with mock.patch.object(chm_extract, "_sevenzip", backend), \ + mock.patch.object(chm_extract, "_chmlib", lambda *_: None), \ + mock.patch.object(chm_extract, "_hh", lambda *_: None), \ + mock.patch.object(chm_convert, "run_pandoc", return_value=("# Topic\n", [])): + fresh = chm_convert.convert_chm(chm, root / "work", root / "fresh") + reused = chm_convert.convert_chm( + chm, root / "work", root / "reused", reuse=True + ) + + self.assertEqual( + fresh["report"].get("stale_toc_entries"), + reused["report"].get("stale_toc_entries"), + ) + self.assertTrue(fresh["report"].get("stale_toc_entries")) + + def test_fresh_extraction_writes_authenticated_manifest_after_success(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# Topic\n", [])): + chm_convert.convert_chm(chm, root / "work", root / "out", extractor=fake_extract) + + self.assertEqual({ + "schema": 1, + "source_name": chm.name, + "source_sha256": hashlib.sha256(chm.read_bytes()).hexdigest(), + }, json.loads((root / "work" / "Using_Help" / ".chm-extraction-manifest.json").read_text())) + + def test_reuse_with_missing_manifest_forces_fresh_extraction(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + (extraction / "old.htm").write_text("old", encoding="cp1252") + calls = [] + + def fake_extract(_chm, fresh_extraction): + calls.append(fresh_extraction) + shutil.rmtree(fresh_extraction, ignore_errors=True) + fresh_extraction.mkdir(parents=True) + (fresh_extraction / "new.htm").write_text( + "New

        New

        ", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# New\n", [])): + chm_convert.convert_chm( + chm, root / "work", root / "out", reuse=True, extractor=fake_extract + ) + + self.assertEqual([extraction], calls) + self.assertFalse((extraction / "old.htm").exists()) + + def test_reuse_with_mismatched_manifest_forces_fresh_extraction(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + (extraction / "old.htm").write_text("old", encoding="cp1252") + (extraction / ".chm-extraction-manifest.json").write_text(json.dumps({ + "schema": 1, "source_name": chm.name, "source_sha256": "0" * 64, + }), encoding="utf-8") + calls = [] + + def fake_extract(_chm, fresh_extraction): + calls.append(fresh_extraction) + shutil.rmtree(fresh_extraction, ignore_errors=True) + fresh_extraction.mkdir(parents=True) + (fresh_extraction / "new.htm").write_text( + "New

        New

        ", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# New\n", [])): + chm_convert.convert_chm( + chm, root / "work", root / "out", reuse=True, extractor=fake_extract + ) + + self.assertEqual([extraction], calls) + + def test_reuse_with_malformed_or_invalid_manifest_forces_fresh_extraction(self): + invalid_manifests = [ + "{", + json.dumps({"schema": 2, "source_name": "Using_Help.chm", "source_sha256": "0" * 64}), + json.dumps({"schema": 1, "source_name": "Other.chm", "source_sha256": "0" * 64}), + json.dumps({"schema": 1, "source_name": "Using_Help.chm", "source_sha256": "A" * 64}), + json.dumps({"schema": 1, "source_name": "Using_Help.chm", "source_sha256": "g" * 64}), + json.dumps({"schema": 1, "source_name": "Using_Help.chm", "source_sha256": "abc"}), + ] + for manifest_text in invalid_manifests: + with self.subTest(manifest=manifest_text), tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + (extraction / "old.htm").write_text("old", encoding="cp1252") + (extraction / ".chm-extraction-manifest.json").write_text( + manifest_text, encoding="utf-8" + ) + calls = [] + + def fake_extract(_chm, fresh_extraction, calls=calls): + calls.append(fresh_extraction) + shutil.rmtree(fresh_extraction, ignore_errors=True) + fresh_extraction.mkdir(parents=True) + (fresh_extraction / "new.htm").write_text( + "New

        New

        ", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# New\n", [])): + chm_convert.convert_chm( + chm, root / "work", root / "out", reuse=True, extractor=fake_extract + ) + self.assertEqual([extraction], calls) + + def test_reuse_rejects_source_path_link_before_hashing(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + with mock.patch.object( + chm_convert, "first_link_in_path", return_value=root / "linked-source" + ), self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, root / "work", root / "out", reuse=True) + + def test_reuse_rejects_extraction_path_link_before_scanning(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + with mock.patch.object( + chm_convert, "first_link_in_path", side_effect=[None, root / "linked-work"] + ), self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, root / "work", root / "out", reuse=True) + + def test_reuse_rejects_manifest_link_before_parsing(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text("topic", encoding="cp1252") + manifest = extraction / ".chm-extraction-manifest.json" + manifest.write_text("{}", encoding="utf-8") + + def fake_first_link(path): + return path if Path(path).name == manifest.name else None + + with mock.patch.object(source_safety, "first_link_in_path", side_effect=fake_first_link), \ + self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, root / "work", root / "out", reuse=True) + + def test_reuse_rejects_relevant_extraction_file_link_before_scanning(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + extraction.mkdir(parents=True) + topic = extraction / "topic.htm" + topic.write_text("topic", encoding="cp1252") + + def fake_first_link(path): + return path if Path(path) == topic else None + + with mock.patch.object(source_safety, "first_link_in_path", side_effect=fake_first_link), \ + self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, root / "work", root / "out", reuse=True) + + def test_reuse_rejects_actual_linked_chm_when_available(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + outside = root.parent / (root.name + "-chm-source") + outside.mkdir() + real_chm = outside / "Using_Help.chm" + real_chm.write_bytes(b"fixture") + linked_chm = root / "Using_Help.chm" + try: + linked_chm.symlink_to(real_chm) + except (OSError, NotImplementedError): + real_chm.unlink() + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + with self.assertRaises(SourceSafetyError): + chm_convert.convert_chm( + linked_chm, root / "work", root / "out", reuse=True + ) + finally: + real_chm.unlink() + outside.rmdir() + + def test_reuse_rejects_actual_linked_extraction_directory_when_available(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + outside = root.parent / (root.name + "-extraction") + outside.mkdir() + work = root / "work" + work.mkdir() + linked_extraction = work / "Using_Help" + try: + linked_extraction.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + with self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, work, root / "out", reuse=True) + finally: + outside.rmdir() + + def test_fresh_extraction_validation_failure_writes_no_manifest(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text("topic", encoding="cp1252") + + with mock.patch.object( + chm_convert, "validate_source_tree", side_effect=SourceSafetyError("unsafe tree") + ), self.assertRaises(SourceSafetyError): + chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + + self.assertFalse( + (root / "work" / "Using_Help" / ".chm-extraction-manifest.json").exists() + ) + + def test_chm_conversion_discovers_htm_and_html_topics_case_insensitively(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "one.HTML").write_text("One

        One

        ", encoding="cp1252") + (extraction / "two.htm").write_text("Two

        Two

        ", encoding="cp1252") + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# body\n", [])): + result = chm_convert.convert_chm(chm, root / "work", root / "out", extractor=fake_extract) + self.assertEqual(2, result["topics"]) + self.assertTrue((root / "out" / "one.md").exists()) + self.assertTrue((root / "out" / "two.md").exists()) + + def test_casefolded_topic_and_asset_destinations_are_rejected_before_writes(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + + with mock.patch.object( + chm_convert, "_topic_files", + side_effect=lambda extraction: [ + extraction / "Icon.htm", extraction / "icon.HTM" + ], + ): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + self.assertTrue(result["report"]["destination_collisions"]) + self.assertEqual([], list((root / "out").rglob("*"))) + + def test_distinct_casefolded_images_are_rejected_before_conversion_writes(self): + if os.path.normcase("Icon.png") == os.path.normcase("icon.PNG"): + self.skipTest("filesystem cannot represent case-distinct image names") + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "Icon.png").write_bytes(b"upper") + (extraction / "icon.PNG").write_bytes(b"lower") + (extraction / "topic.htm").write_text( + "Topic

        Topic

        ", encoding="cp1252" + ) + + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + self.assertTrue(result["report"]["destination_collisions"]) + self.assertEqual([], [ + path for path in (root / "out").rglob("*") if path.is_file() + ]) + + def test_frontmatter_uses_json_yaml_scalars_and_escapes_c0_controls(self): + rendered = chm_convert.frontmatter({ + "title": 'Unicode ☃ "quoted" \\ path', "tab": "a\tb" + }) + self.assertIn('title: "Unicode ☃ \\"quoted\\" \\\\ path"', rendered) + self.assertIn('tab: "a\\tb"', rendered) + for value in ("bad\x00value", "bad\nvalue", "bad\rvalue", "bad\x1fvalue"): + with self.subTest(value=repr(value)): + output = chm_convert.frontmatter({"title": value}) + self.assertNotIn(value, output) + self.assertIn("\\", output) + + def test_chm_conversion_escapes_control_chars_in_metadata_frontmatter(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_bytes( + b"Bad\x01Title

        Topic

        " + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=("# Topic\n", [])): + chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + output = (root / "out" / "topic.md").read_text(encoding="utf-8") + self.assertNotIn("\x01", output) + self.assertIn("\\u0001", output) + + def test_converter_neutralizes_unsafe_and_absolute_targets_but_preserves_allowed_schemes(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        " + "bad" + "good", encoding="cp1252" + ) + + with mock.patch.object(chm_convert, "run_pandoc", return_value=( + ("# Topic\n\n[bad](javascript:alert(1)) [file](/x) " + "[good](https://example.test) [mail](mailto:a@example.test)\n"), [] + )): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + output = (root / "out" / "topic.md").read_text(encoding="utf-8") + self.assertNotIn("javascript:", output.lower()) + self.assertNotIn("/x)", output) + self.assertIn("https://example.test", output) + self.assertIn("mailto:a@example.test", output) + self.assertTrue(result["report"].get("source_unsafe_uri")) + self.assertTrue(result["report"].get("path_escape")) + + def test_direct_conversion_rejects_destination_overlapping_extraction_before_writes(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + extraction = root / "work" / "Using_Help" + with self.assertRaises(SourceSafetyError): + chm_convert.convert_chm(chm, root / "work", extraction) + self.assertFalse(extraction.exists()) + + def test_internal_topic_matching_is_case_insensitive(self): + self.assertEqual([], chm_convert._check_links( + ["SubTopic.HTM"], "Index.HTM", {"subtopic.htm"} + )) + + def test_converter_neutralizes_encoded_separator_unsafe_targets(self): + report: dict[str, list] = {} + sanitized = chm_convert._sanitize_markdown_targets( + "[bad](java%09script:alert(1)) [drive](C:%5Coutside) " + "[unc](%5C%5Cserver%5Cshare)", report, "topic.htm" + ) + self.assertNotIn("script:", sanitized.lower()) + self.assertTrue(report.get("unsafe_uri")) + self.assertGreaterEqual(len(report.get("path_escape", [])), 2) + + def test_build_treats_converter_uri_and_collision_reports_as_fatal(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def unsafe_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "topic.md").write_text("# Topic\n", encoding="utf-8") + return { + "chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, + "report": {"unsafe_uri": [["topic.htm", "javascript:x"]]}, + } + + with mock.patch.object(convert, "run_in_private_stage", side_effect=unsafe_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, repo / "export", repo / "work") + self.assertFalse(result["promoted"]) + self.assertTrue(any( + issue["code"] == "unsafe_uri" and issue["fatal"] + for issue in result["report"]["issues"] + )) + + def test_chm_conversion_records_exact_source_replacement_provenance(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "source.htm").write_bytes( + b"Source

        Source

        \x81" + ) + (extraction / "clean.htm").write_text( + "Clean

        Clean

        ", encoding="cp1252" + ) + + def fake_pandoc(source, _tmp): + title = "Source" if "Source" in source else "Clean" + return (f"# {title}\n\ncontains �\n" if "\ufffd" in source else "# Clean\n", []) + + with mock.patch.object(chm_convert, "run_pandoc", side_effect=fake_pandoc): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + self.assertEqual(["source.htm"], result["source_replacement_paths"]) + + def test_chm_source_normalizes_robohelp_cp1252_nbsp_before_pandoc(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + seen = [] + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text( + "Topic

        Topic

        " + "

        word\u00a0boundary

" + "

K\u00a0

 " + "

entity boundary again

", + encoding="cp1252", + ) + + def fake_pandoc(source, _tmp): + seen.append(source) + return ("# Topic\n", []) + + with mock.patch.object(chm_convert, "run_pandoc", side_effect=fake_pandoc): + chm_convert.convert_chm(chm, root / "work", root / "out", extractor=fake_extract) + self.assertEqual(1, len(seen)) + self.assertNotIn("\u00a0", seen[0]) + self.assertNotIn(" ", seen[0]) + self.assertNotIn("\ufffd", seen[0]) + self.assertIn("word boundary", seen[0]) + self.assertIn("entity boundary again", seen[0]) + + def test_fwhelp_declares_robohelp_presentation_classes_and_note_marker(self): + lua = Path(convert.__file__).with_name("fwhelp.lua").read_text(encoding="utf-8") + for name in ("hcp1", "hcp2", "hcp3", "hcp4"): + self.assertIn(f"{name} = \"plain\"", lua) + self.assertIn('note[_-]?icon%.gif', lua) + + def test_build_reports_allowlisted_source_replacement_as_advisory(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def source_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "source.md").write_text("# Source\n\ufffd\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, "topics_paths": ["source.htm"], + "source_replacement_paths": ["source.htm"], "report": {}} + + with mock.patch.object(convert, "run_in_private_stage", side_effect=source_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=( + {"converted": 0, "report": {}}, [] + )): + result = convert.build(repo, repo / "export", repo / "work") + self.assertTrue(result["promoted"]) + replacement = [issue for issue in result["report"]["issues"] + if issue["code"] == "source_replacement_character"] + self.assertEqual(1, len(replacement)) + self.assertFalse(replacement[0]["fatal"]) + + def test_pdf_replacement_reports_authored_and_generated_paths(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / "Using_Help.chm").write_bytes(b"fixture") + + def clean_chm(_chm, _work, destination, **_kwargs): + destination.mkdir(parents=True, exist_ok=True) + (destination / "index.md").write_text("# Page\n", encoding="utf-8") + return {"chm": "Using_Help.chm", "stem": "Using_Help", "toc": [], + "topics": 1, "images": 0, "report": {}} + + pdf_result = {"converted": 2, "report": { + "pdf_source_replacements": [[ + "docs/Guide With Spaces.pdf", + [{"page": 3, "count": 1, "codepoints": ["U+001F"]}], + ]], + "pdf_export_replacements": [[ + "docs/Generated.pdf", {"source_count": 0, "exporter_count": 1}, + ]], + }} + with mock.patch.object(convert, "run_in_private_stage", side_effect=clean_chm), \ + mock.patch.object(convert.pdf_convert, "run_in_private_stage", return_value=(pdf_result, [])): + result = convert.build(repo, repo / "export", repo / "work") + source = [issue for issue in result["report"]["issues"] + if issue["code"] == "source_replacement_character" + and issue["path"] == "docs/Guide With Spaces.pdf"] + self.assertEqual(1, len(source)) + self.assertFalse(source[0]["fatal"]) + self.assertEqual("source", source[0]["provenance"]) + self.assertEqual( + "pdf/docs/Guide_With_Spaces.md", + source[0]["detail"]["generated_markdown"], + ) + exporter = [issue for issue in result["report"]["issues"] + if issue["path"] == "docs/Generated.pdf"] + self.assertTrue(exporter) + self.assertTrue(exporter[0]["fatal"]) + self.assertFalse(result["promoted"]) + + def test_unmapped_span_report_retains_affected_robohelp_topic(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Using_Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + extraction.mkdir(parents=True) + (extraction / "topic.htm").write_text( + "Topic

Topic

", encoding="cp1252" + ) + + with mock.patch.object( + chm_convert, "run_pandoc", return_value=("# Topic\n", ["MysteryClass"]) + ): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + + self.assertEqual( + ["MysteryClass", 1, ["topic.htm"]], + result["report"]["unmapped_span_classes"][0], + ) + + def test_producer_issue_normalization_uses_authored_topic_paths(self): + unmapped = convert._report_issue( + "unmapped_span_classes", + ["MysteryClass", 2, ["a.htm", "folder/b.htm"]], + "Using_Help.chm", + ) + duplicate = convert._report_issue( + "duplicate_titles", ["Same title", ["one.htm", "two.htm"]] + ) + + self.assertEqual("a.htm", unmapped.path) + self.assertEqual( + {"class": "MysteryClass", "count": 2, "topics": ["a.htm", "folder/b.htm"]}, + unmapped.detail, + ) + self.assertEqual("one.htm", duplicate.path) + self.assertEqual( + {"title": "Same title", "topics": ["one.htm", "two.htm"]}, + duplicate.detail, + ) + + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_pandoc_lua_keeps_nested_lists_and_parenthesized_image_targets(self): + html = ( + "

Topic

  • Parent
    • Child
" + "

" + ) + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(convert.__file__).with_name('fwhelp.lua')}"], + input=html, text=True, encoding="utf-8", capture_output=True, check=True, + ) + self.assertNotIn("- -", result.stdout) + self.assertIn("Child", result.stdout) + self.assertTrue("images/a%20(1).png" in result.stdout or "images/a\\ (1).png" in result.stdout) + + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_pandoc_lua_drops_decorative_note_icon_but_keeps_normal_images(self): + html = ( + "

Tip

" + "

" + ) + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(convert.__file__).with_name('fwhelp.lua')}"], + input=html, text=True, encoding="utf-8", capture_output=True, check=True, + ) + self.assertNotIn("\ufffd", result.stdout) + self.assertIn("images/check.png", result.stdout) + self.assertIn("[!TIP]", result.stdout) + + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_pandoc_lua_inventories_unknown_class_even_with_supported_class(self): + html = "

Text

" + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(convert.__file__).with_name('fwhelp.lua')}"], + input=html, text=True, encoding="utf-8", capture_output=True, check=True, + ) + self.assertIn("**Text**", result.stdout) + self.assertIn("FWHELP_UNMAPPED_SPAN NewSemantic=1", result.stderr) + + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_pandoc_lua_neutralizes_html_character_ref_unsafe_scheme(self): + html = "

Bad

" + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(convert.__file__).with_name('fwhelp.lua')}"], + input=html, text=True, encoding="utf-8", capture_output=True, check=True, + ) + self.assertNotIn("script:", result.stdout.lower()) + + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_pandoc_lua_neutralizes_percent_encoded_drive_and_unc_paths(self): + html = ( + "

Drive " + "UNC

" + ) + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(convert.__file__).with_name('fwhelp.lua')}"], + input=html, text=True, encoding="utf-8", capture_output=True, check=True, + ) + self.assertNotIn("outside", result.stdout.lower()) + self.assertNotIn("server", result.stdout.lower()) + self.assertGreaterEqual(result.stderr.count("FWHELP_PATH_ESCAPE"), 2) + + +class LinkCaseTests(unittest.TestCase): + def test_authored_link_case_is_corrected_and_reported(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + chm = root / "Help.chm" + chm.write_bytes(b"fixture") + + def fake_extract(_chm, extraction): + (extraction / "Topics" / "A_&_B").mkdir(parents=True) + (extraction / "Topics" / "Start.htm").write_text( + "Start" + "wrong case" + "right case" + "absent", + encoding="cp1252", + ) + (extraction / "Topics" / "A_&_B" / "Target.htm").write_text( + "Target", encoding="cp1252" + ) + + captured: list[str] = [] + + def capture(source, tmp): + captured.append(source) + return "# converted\n", [] + + with mock.patch.object(chm_convert, "run_pandoc", side_effect=capture): + result = chm_convert.convert_chm( + chm, root / "work", root / "out", extractor=fake_extract + ) + + start = next(text for text in captured if "wrong case" in text) + # The authored case is republished as the case the file really has, so + # the link survives on a case-sensitive host. + self.assertIn("href='A_%26_B/Target.htm'>wrong case", start) + # Links that already match are left exactly as authored. + self.assertIn("href='A_%26_B/Target.htm'>right case", start) + # A target that does not exist in any case is not invented. + self.assertIn("href='A_&_B/Missing.htm'>absent", start) + mismatches = result["report"]["link_case_mismatches"] + self.assertEqual([["Topics/Start.htm", "a_&_b/target.htm"]], mismatches) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_corpus_validation.py b/tools/test_corpus_validation.py new file mode 100644 index 00000000..d0967b0f --- /dev/null +++ b/tools/test_corpus_validation.py @@ -0,0 +1,166 @@ +import tempfile +import unittest +from pathlib import Path + +from corpus_validation import validate_corpus + + +class CorpusValidationTests(unittest.TestCase): + def test_resolves_local_links_and_reports_fatal_exporter_errors(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "ok.md").write_text( + "# One\n\n[ok](two.md) ![image](img.png)\n\n- parent\n - child\n", + encoding="utf-8", + ) + (root / "two.md").write_text("# Two\n", encoding="utf-8") + (root / "img.png").write_bytes(b"image") + issues = validate_corpus(root) + self.assertEqual([], issues) + + def test_missing_image_and_malformed_list_are_fatal(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + "# Page\n\n![missing](missing.png)\n- - literal\n", + encoding="utf-8", + ) + (root / "missing.md").write_text("# Target\n", encoding="utf-8") + issues = validate_corpus(root) + by_code = {item.code: item for item in issues} + self.assertTrue(by_code["missing_image"].fatal) + self.assertTrue(by_code["malformed_list"].fatal) + + def test_source_missing_link_is_advisory_and_duplicate_title_is_reported(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "one.md").write_text( + "---\nsource: a.htm\n---\n# Same\n\n[old](missing.htm)\n", + encoding="utf-8", + ) + (root / "two.md").write_text("# Same\n", encoding="utf-8") + issues = validate_corpus(root, advisory_links={("one.md", "missing.htm")}) + self.assertTrue(any(i.code == "source_missing_link" and not i.fatal for i in issues)) + self.assertTrue(any(i.code == "duplicate_title" for i in issues)) + + def test_unallowlisted_link_is_fatal_even_when_source_metadata_exists(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + "---\nsource: page.htm\n---\n# Page\n\n[rewritten](missing.md)\n", + encoding="utf-8", + ) + issues = validate_corpus(root, advisory_links={("page.md", "other.md")}) + self.assertTrue(any(i.code == "missing_link" and i.fatal for i in issues)) + + def test_external_anchors_title_attributes_and_escaped_brackets_are_ignored(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + '# Page\n\n[external](https://example.test/no.md) [anchor](#x) ' + '[label](missing.md "title") \\[literal\\]\n', encoding="utf-8" + ) + (root / "missing.md").write_text("# Target\n", encoding="utf-8") + issues = validate_corpus(root) + self.assertFalse(any(i.code == "missing_link" for i in issues)) + + def test_balances_parentheses_angle_targets_and_ignores_fenced_code(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Export_(LIFT).md").write_text("# Target\n", encoding="utf-8") + (root / "space name.md").write_text("# Space\n", encoding="utf-8") + (root / "page.md").write_text( + "# Page\n\n[paren](Export_(LIFT).md) [angle]()\n" + "```\n[ignored](nope.md)\n```\n", encoding="utf-8" + ) + issues = validate_corpus(root) + self.assertFalse(any(i.code == "missing_link" for i in issues)) + + def test_rejects_a_local_target_that_escapes_the_corpus_root(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text("# Page\n\n[out](../outside.md)\n", encoding="utf-8") + issues = validate_corpus(root) + self.assertTrue(any(i.code == "missing_link" and i.fatal for i in issues)) + + def test_source_replacement_character_is_advisory_only_when_allowlisted(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "source.md").write_text("# Source\n\ufffd\n", encoding="utf-8") + (root / "generated.md").write_text("# Generated\n\ufffd\n", encoding="utf-8") + issues = validate_corpus(root, source_replacement_paths={"source.md"}) + by_path = {issue.path: issue for issue in issues + if issue.code in {"source_replacement_character", "replacement_character"}} + self.assertFalse(by_path["source.md"].fatal) + self.assertTrue(by_path["generated.md"].fatal) + + def test_duplicate_titles_are_case_insensitive(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "one.md").write_text("# Same\n", encoding="utf-8") + (root / "two.md").write_text("# same\n", encoding="utf-8") + issues = validate_corpus(root) + self.assertTrue(any(issue.code == "duplicate_title" for issue in issues)) + + def test_allowlisted_source_image_is_advisory(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text("# Page\n\n![source](missing.png)\n", encoding="utf-8") + issues = validate_corpus(root, advisory_images={("page.md", "missing.png")}) + self.assertTrue(any(issue.code == "source_missing_image" and not issue.fatal for issue in issues)) + + def test_unsafe_uri_and_absolute_local_variants_are_fatal(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + "# Page\n\n[j](javascript:alert(1)) [f](file:///tmp/x) " + "![d](data:image/png;base64,AAAA) [abs](/x) " + "[drive](C:\\\\x) [unc](\\\\server\\share\\x)\n", + encoding="utf-8", + ) + issues = validate_corpus(root) + self.assertTrue(any(i.code == "unsafe_uri" and i.fatal for i in issues)) + self.assertTrue(any(i.code == "path_escape" and i.fatal for i in issues)) + + def test_allowed_external_uri_schemes_are_not_flagged(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + "# Page\n\n[https](https://example.test/x) " + "[http](http://example.test/x) [mail](mailto:a@example.test)\n", + encoding="utf-8", + ) + issues = validate_corpus(root) + self.assertEqual([], [issue for issue in issues if issue.code in {"unsafe_uri", "path_escape"}]) + + def test_link_case_must_match_its_target_on_every_platform(self): + # A case-insensitive filesystem resolves this link and a case-sensitive + # host does not, so validating with the host's rules would pass on + # Windows and publish a 404 to GitHub. + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "Topics").mkdir() + (root / "Topics" / "Target.md").write_text("# Target\n", encoding="utf-8") + (root / "page.md").write_text( + "# Page\n\n[wrong case](topics/target.md) " + "[right case](Topics/Target.md)\n", + encoding="utf-8", + ) + issues = validate_corpus(root) + broken = [issue for issue in issues if issue.code == "missing_link"] + self.assertEqual(1, len(broken), [issue.message for issue in broken]) + self.assertIn("topics/target.md", broken[0].message) + + def test_encoded_uri_separators_are_unsafe(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "page.md").write_text( + "# Page\n\n[tab](java%09script:alert(1)) " + "[cr](java%0dscript:alert(1))\n", encoding="utf-8" + ) + issues = validate_corpus(root) + self.assertGreaterEqual(sum(issue.code == "unsafe_uri" for issue in issues), 2) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_issue_catalog.py b/tools/test_issue_catalog.py new file mode 100644 index 00000000..76d8ef13 --- /dev/null +++ b/tools/test_issue_catalog.py @@ -0,0 +1,68 @@ +import unittest + +from issue_catalog import ISSUE_ALIASES, ISSUE_CATALOG, IssuePolicy +from reporting import LABELS, Issue, make_issue + + +class IssueCatalogTests(unittest.TestCase): + def test_every_producer_alias_resolves_to_one_catalog_policy(self): + self.assertTrue(ISSUE_ALIASES) + for code in ISSUE_ALIASES.values(): + self.assertIn(code, ISSUE_CATALOG) + + def test_every_emitted_issue_code_has_exactly_one_policy(self): + emitted = { + "missing_link", "source_missing_link", "missing_image", "source_missing_image", + "source_link_case", "duplicate_title", "malformed_list", "replacement_character", + "source_replacement_character", "raw_html", "one_h1", "destination_collision", + "unsafe_uri", "path_escape", "source_unsafe_uri", "source_path_escape", + "pandoc_failure", "unmapped_span", "pdf_failure", + "outline_drift", "outline_unpinned", "stale_toc_entries", "not_in_toc", + "chm_failure", "chm_discovery", "unknown_issue", + } + self.assertEqual(emitted, set(ISSUE_CATALOG)) + + def test_policies_are_immutable_and_cover_representative_issues(self): + self.assertIsInstance(ISSUE_CATALOG["pdf_failure"], IssuePolicy) + self.assertTrue(all(policy.guidance.strip() for policy in ISSUE_CATALOG.values())) + self.assertTrue(ISSUE_CATALOG["pdf_failure"].fatal) + self.assertEqual("exporter", ISSUE_CATALOG["pdf_failure"].provenance) + self.assertFalse(ISSUE_CATALOG["raw_html"].fatal) + self.assertEqual("source", ISSUE_CATALOG["raw_html"].provenance) + with self.assertRaises((AttributeError, TypeError)): + ISSUE_CATALOG["raw_html"].fatal = True + with self.assertRaises(TypeError): + LABELS["raw_html"] = "changed" + + def test_public_policy_constructor_remains_source_compatible(self): + policy = IssuePolicy("Example", False, "source") + self.assertEqual("", policy.guidance) + + def test_make_issue_uses_catalog_policy(self): + issue = make_issue("html_tables_kept", "table kept", "doc.pdf") + self.assertEqual("raw_html", issue.code) + self.assertEqual("Raw HTML retained", issue.label) + self.assertFalse(issue.fatal) + self.assertEqual("source", issue.provenance) + + def test_unknown_issue_is_fatal_and_does_not_raise_key_error(self): + issue = make_issue("future_producer_code", "new failure", "page.md") + self.assertEqual("unknown_issue", issue.code) + self.assertTrue(issue.fatal) + self.assertEqual("exporter", issue.provenance) + self.assertIn("future_producer_code", issue.message) + self.assertTrue(Issue("another_future_code", "failure", fatal=False).fatal) + + def test_legacy_severity_arguments_cannot_override_catalog_policy(self): + issue = Issue("missing_link", "generated missing link", fatal=False, provenance="source") + self.assertTrue(issue.fatal) + self.assertEqual("exporter", issue.provenance) + + def test_direct_issue_also_gets_catalog_defaults(self): + issue = Issue("outline_drift", "changed", "guide.pdf") + self.assertTrue(issue.fatal) + self.assertEqual("PDF outline drift", issue.label) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_output_fs.py b/tools/test_output_fs.py new file mode 100644 index 00000000..55986287 --- /dev/null +++ b/tools/test_output_fs.py @@ -0,0 +1,230 @@ +import os +import subprocess +import tempfile +import unittest +from pathlib import Path + +from output_fs import ( + ExportBusyError, + ExportLock, + OutputPathError, + OutputStaging, + export_locks, +) + + +class OutputStagingSafetyTests(unittest.TestCase): + def test_rejects_filesystem_root_before_touching_it(self): + with self.assertRaises(OutputPathError): + OutputStaging(Path(Path.cwd().anchor)) + + def test_uses_destination_parent_for_owned_stage_when_work_is_omitted(self): + with tempfile.TemporaryDirectory() as raw: + destination = Path(raw) / "output" + with OutputStaging(destination) as staged: + self.assertEqual(destination.parent, staged.path.parent) + (staged / "fresh.txt").write_text("fresh", encoding="utf-8") + staged.promote() + self.assertEqual("fresh", (destination / "fresh.txt").read_text(encoding="utf-8")) + + def test_implicit_sibling_stage_is_allowed_under_protected_repo_root(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) / "repo" + repo.mkdir() + destination = repo / "out" + with OutputStaging(destination, repo_root=repo, source_root=repo) as staged: + self.assertEqual(repo, staged.path.parent) + (staged / "fresh.txt").write_text("fresh", encoding="utf-8") + staged.promote() + self.assertEqual("fresh", (destination / "fresh.txt").read_text(encoding="utf-8")) + + def test_explicit_repo_root_work_directory_is_rejected(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + repo = root / "repo" + repo.mkdir() + with self.assertRaises(OutputPathError): + OutputStaging(root / "out", work_dir=repo, repo_root=repo, source_root=repo) + + def test_rejects_symlink_destination_and_symlink_ancestor_before_resolution(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + real = root / "real" + real.mkdir() + link = root / "link" + try: + link.symlink_to(real, target_is_directory=True) + except (OSError, NotImplementedError): + if os.name == "nt": + junction = subprocess.run( + ["cmd.exe", "/c", "mklink", "/J", str(link), str(real)], + capture_output=True, + text=True, + check=False, + ) + if junction.returncode != 0: + self.skipTest("directory links are unavailable") + else: + self.skipTest("directory symlinks are unavailable") + + try: + for destination in (link, link / "generated"): + with self.subTest(destination=destination), self.assertRaises(OutputPathError): + OutputStaging(destination) + self.assertTrue(real.is_dir()) + finally: + if link.is_symlink(): + link.unlink() + elif link.exists(): + link.rmdir() + + def test_rejects_repository_and_source_roots_before_removal(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + repo = root / "repo" + source = root / "source" + repo.mkdir() + source.mkdir() + for forbidden in (repo, source): + with self.subTest(forbidden=forbidden): + marker = forbidden / "keep.txt" + marker.write_text("keep", encoding="utf-8") + with self.assertRaises(OutputPathError): + OutputStaging(forbidden, repo_root=repo, source_root=source) + self.assertTrue(marker.exists()) + + def test_rejects_overlapping_work_and_output_paths(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + work = root / "work" + output = root / "output" + work.mkdir() + output.mkdir() + with self.assertRaises(OutputPathError): + OutputStaging(output / "nested", work_dir=output, repo_root=root / "repo") + with self.assertRaises(OutputPathError): + OutputStaging(output, work_dir=output / "nested", repo_root=root / "repo") + + def test_promote_replaces_destination_and_failed_context_preserves_it(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + work = root / "work" + destination = root / "output" + work.mkdir() + destination.mkdir() + (destination / "stale.txt").write_text("stale", encoding="utf-8") + + with OutputStaging(destination, work_dir=work, repo_root=root / "repo") as staged: + self.assertEqual([], list(staged.iterdir())) + (staged / "fresh.txt").write_text("fresh", encoding="utf-8") + staged.promote() + + self.assertEqual("fresh", (destination / "fresh.txt").read_text(encoding="utf-8")) + self.assertFalse((destination / "stale.txt").exists()) + self.assertFalse(any(p.name.startswith(".output-stage-") for p in work.iterdir())) + + destination.joinpath("keep.txt").write_text("keep", encoding="utf-8") + with self.assertRaises(RuntimeError), OutputStaging( + destination, work_dir=work, repo_root=root / "repo" + ) as staged: + (staged / "discarded.txt").write_text("discard", encoding="utf-8") + raise RuntimeError("build failed") + self.assertEqual("keep", (destination / "keep.txt").read_text(encoding="utf-8")) + self.assertFalse((destination / "discarded.txt").exists()) + self.assertFalse(any(p.name.startswith(".output-stage-") for p in work.iterdir())) + + +class ExportLockTests(unittest.TestCase): + def test_lock_rejects_another_invocation_for_same_destination(self): + with tempfile.TemporaryDirectory() as raw: + destination = Path(raw) / "output" + with ExportLock(destination) as held: + with self.assertRaisesRegex(ExportBusyError, "export destination is busy"): + ExportLock(destination).acquire() + self.assertEqual(held.lock_path, ExportLock(destination).lock_path) + + def test_different_destinations_do_not_contend(self): + with tempfile.TemporaryDirectory() as raw: + first = Path(raw) / "first" + second = Path(raw) / "second" + with ExportLock(first), ExportLock(second): + pass + + def test_exception_releases_lock(self): + with tempfile.TemporaryDirectory() as raw: + destination = Path(raw) / "output" + with self.assertRaises(RuntimeError), ExportLock(destination): + raise RuntimeError("failure") + with ExportLock(destination): + pass + + def test_stale_lockfile_does_not_block_acquisition(self): + with tempfile.TemporaryDirectory() as raw: + destination = Path(raw) / "output" + lock = ExportLock(destination) + lock.lock_path.write_text("stale owner metadata\n", encoding="utf-8") + with ExportLock(destination): + pass + self.assertEqual("stale owner metadata\n", lock.lock_path.read_text(encoding="utf-8")) + + def test_linked_lock_parent_is_rejected_before_open(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + real = root / "real" + real.mkdir() + linked = root / "linked" + try: + linked.symlink_to(real, target_is_directory=True) + except (OSError, NotImplementedError): + self.skipTest("directory links are unavailable") + with self.assertRaises(OutputPathError): + ExportLock(linked / "output").acquire() + + def test_export_locks_deduplicates_targets_and_releases_partial_acquisition(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + first, second = root / "first", root / "second" + with ( + ExportLock(second), + self.assertRaises(ExportBusyError), + export_locks(first, second, first), + ): + pass + with export_locks(first, first) as locks: + self.assertEqual(1, len(locks)) + with ExportLock(first): + pass + + def test_export_locks_uses_stable_order_for_reversed_targets(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + first, second = root / "first", root / "second" + expected = sorted( + ( + os.path.normcase(os.path.normpath(os.path.abspath(os.fspath(first)))), + os.path.normcase(os.path.normpath(os.path.abspath(os.fspath(second)))), + ) + ) + with export_locks(second, first) as locks: + actual = [ + os.path.normcase(os.path.normpath(os.fspath(lock.destination))) + for lock in locks + ] + self.assertEqual(expected, actual) + + def test_reversed_contention_releases_any_lock_acquired_before_busy_target(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + first, second = root / "first", root / "second" + with ( + ExportLock(first), + self.assertRaises(ExportBusyError), + export_locks(second, first), + ): + pass + with ExportLock(second): + pass + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_pdf_convert.py b/tools/test_pdf_convert.py new file mode 100644 index 00000000..e3df15a6 --- /dev/null +++ b/tools/test_pdf_convert.py @@ -0,0 +1,1271 @@ +import json +import os +import shutil +import subprocess +import sys +import tempfile +import unittest +from contextlib import redirect_stdout +from io import StringIO +from pathlib import Path +from unittest import mock + +import pdf_convert +from frontmatter import yaml_scalar +from output_fs import ExportBusyError, ExportLock + + +class StandaloneConsoleTests(unittest.TestCase): + def test_main_renders_catalog_labels_and_preserves_source_provenance(self): + producer_report = { + "pdf_failures": [["broken.pdf", "backend failed"]], + "pdf_source_replacements": [["source.pdf", [{"page": 2, "count": 1}]]], + } + output = StringIO() + with mock.patch.object( + pdf_convert, "run", return_value=({"converted": 1, "report": producer_report}, []) + ), mock.patch.object( + sys, "argv", ["pdf_convert.py", "--repo", ".", "--out", "export"] + ), redirect_stdout(output): + status = pdf_convert.main() + + rendered = output.getvalue() + self.assertEqual(1, status) + self.assertIn("PDF conversion failure", rendered) + self.assertIn("Replacement character", rendered) + self.assertIn("FATAL", rendered) + self.assertIn("WARN", rendered) + issues = pdf_convert._canonical_report(producer_report).issues + source_issue = next(issue for issue in issues if issue.code == "source_replacement_character") + self.assertEqual("source", source_issue.provenance) + + +class StripContentsSectionsTests(unittest.TestCase): + def test_removes_front_matter_contents_table_as_one_section(self): + markdown = """Preface text. + +## Contents + +
\u00a0
+ +
Introduction1
+ +## 1 Introduction + +Body text. +""" + + self.assertEqual( + "Preface text.\n\n## 1 Introduction\n\nBody text.", + pdf_convert.strip_contents_sections(markdown), + ) + + def test_keeps_a_contents_section_in_the_body(self): + prefix = "\n".join(f"body line {i}" for i in range(20)) + markdown = f"""{prefix} + +## Contents + +This section explains package contents. +""" + + self.assertIn("This section explains package contents.", + pdf_convert.strip_contents_sections(markdown)) + + def test_discover_pdfs_rejects_symlink_inputs(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + target = repo / "real.pdf" + target.write_bytes(b"") + link = repo / "linked.pdf" + try: + link.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(ValueError): + pdf_convert.discover_pdfs(repo) + + def test_discover_pdfs_excludes_git_directory(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + (repo / ".git").mkdir() + (repo / ".git" / "hidden.pdf").write_bytes(b"") + self.assertEqual([], pdf_convert.discover_pdfs(repo)) + + def test_discover_pdfs_ignores_irrelevant_git_symlink(self): + with tempfile.TemporaryDirectory() as raw: + repo = Path(raw) + outside = repo.parent / (repo.name + "-git") + outside.mkdir() + (outside / "hidden.pdf").write_bytes(b"") + link = repo / ".git" + try: + link.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + outside.joinpath("hidden.pdf").unlink() + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + self.assertEqual([], pdf_convert.discover_pdfs(repo)) + finally: + (outside / "hidden.pdf").unlink() + outside.rmdir() + + def test_discover_pdfs_preserves_relative_root_style(self): + with tempfile.TemporaryDirectory() as raw: + absolute_root = Path(raw) + (absolute_root / "guide.PDF").write_bytes(b"") + relative_root = Path(os.path.relpath(absolute_root, Path.cwd())) + found = pdf_convert.discover_pdfs(relative_root) + self.assertFalse(found[0].is_absolute()) + self.assertEqual(relative_root, found[0].parent) + + def test_discover_pdfs_normalizes_root_with_dotdot_for_collisions(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + nested = root / "nested" + nested.mkdir() + (root / "guide.pdf").write_bytes(b"") + caller_root = nested / ".." + found = pdf_convert.discover_pdfs(caller_root) + self.assertEqual([root / "guide.pdf"], found) + self.assertEqual([], pdf_convert.destination_collisions(caller_root, root / "out", found)) + + def test_colon_form_stops_before_editorial_front_matter(self): + markdown = """## Contents: + +1 Introduction ........ 2 + +Editor's note: This edition was converted from HTML. + +## Abstract + +Abstract body. +""" + + result = pdf_convert.strip_contents_sections(markdown) + + self.assertNotIn("1 Introduction", result) + self.assertIn("Editor's note: This edition was converted from HTML.", result) + self.assertIn("## Abstract", result) + + def test_keeps_section_when_no_safe_closing_boundary_exists(self): + markdown = """# Contents + +Chapter One ........ 2 + +## Chapter One + +Real body text. +""" + + self.assertEqual(markdown.rstrip(), + pdf_convert.strip_contents_sections(markdown)) + + +class StripTocTests(unittest.TestCase): + def test_keeps_dotted_reference_line_outside_front_matter(self): + prefix = "\n".join(f"body line {i}" for i in range(20)) + markdown = f"""{prefix} + +Example ................................ 42 + +More body. +""" + + self.assertIn("Example ................................ 42", + pdf_convert.strip_toc(markdown)) + + +class PickTitleTests(unittest.TestCase): + def test_rejects_stale_readme_metadata_for_document_heading(self): + self.assertEqual( + "Technical Notes on FieldWorks Send-Receive", + pdf_convert.pick_title( + "FieldWorks Language Explorer beta 0.8 ReadMe", + "# **Technical Notes on FieldWorks Send-Receive**\n\nBody.", + "Technical Notes on FieldWorks Send-Receive", + ), + ) + + def test_rejects_machine_generated_word_metadata(self): + self.assertEqual( + "Parsing With Left-Corner and Head-Driven Strategies", + pdf_convert.pick_title( + "Microsoft Word - ESR 041silwp1997-007r.doc", + "# Parsing With Left-Corner and Head-Driven Strategies\n", + "silewp2007_002", + ), + ) + + def test_prefers_descriptive_filename_to_truncated_metadata(self): + self.assertEqual( + "FieldWorks Writing Systems", + pdf_convert.pick_title( + "Writing Systems", + "Body without a heading.", + "FieldWorks Writing Systems", + ), + ) + + def test_uses_title_line_before_a_generic_contents_bookmark(self): + self.assertEqual( + "TonePars: A Computational Tool for Exploring Autosegmental Tonology", + pdf_convert.pick_title( + "Microsoft Word - ESR 041silwp1997-007r.doc", + """**TonePars** : A Computational Tool for Exploring Autosegmental Tonology + +H. Andrew Black + +#### Contents: +""", + "silewp2007_002", + ), + ) + + def test_ignores_numbered_running_header_before_document_title(self): + self.assertEqual( + "Technical Notes on Writing Systems", + pdf_convert.pick_title( + "FieldWorks Language Explorer beta 0.8 ReadMe", + """1 Ingredients + +1 + +## Technical Notes on Writing Systems +""", + "Technical Notes on Writing Systems", + ), + ) + + +class DropRepeatedTitleTests(unittest.TestCase): + def test_removes_plain_emphasized_title_line(self): + markdown = """**TonePars** : A Computational Tool for Exploring Autosegmental Tonology + +H. Andrew Black + +#### Contents: +""" + + self.assertEqual( + "H. Andrew Black\n\n#### Contents:\n", + pdf_convert.drop_repeated_title( + markdown, + "TonePars: A Computational Tool for Exploring Autosegmental Tonology", + ), + ) + + +class FinalizePdfTests(unittest.TestCase): + def test_outline_describes_body_after_repeated_title_removal(self): + title, body, outline = pdf_convert.finalize_pdf( + "Document Title", + "## Document Title\n\n## 1 Introduction\n\nBody.\n", + "Document Title", + ) + + self.assertEqual("Document Title", title) + self.assertNotIn("## Document Title", body) + self.assertEqual([(2, "1 Introduction")], outline) + + def test_removes_every_copy_of_a_title_split_across_lines(self): + markdown = """## A Conceptual Introduction to Morphological Parsing for + +**FieldWorks Language Explorer** + +## A Conceptual Introduction to Morphological Parsing for + +**FieldWorks Language Explorer** + +H. Andrew Black +""" + + self.assertEqual( + "H. Andrew Black\n", + pdf_convert.drop_repeated_title( + markdown, + "A Conceptual Introduction to Morphological Parsing for FieldWorks Language Explorer", + ), + ) + + def test_normalizes_author_and_top_level_heading(self): + title, body, outline = pdf_convert.finalize_pdf( + "Publishing", + "##### Ken Zook\n\n### 1 Introduction\n\nBody.\n", + "Publishing", + ) + self.assertEqual("Publishing", title) + self.assertNotIn("Ken Zook", body) + self.assertEqual([(2, "1 Introduction")], outline) + + def test_promotes_documents_that_begin_at_h3_without_losing_nested_levels(self): + _, body, outline = pdf_convert.finalize_pdf( + "Variant Generator", + "### 1 Introduction\n\n#### 1.1 Appearance\n\nBody.\n", + "VarGen", + ) + self.assertIn("## 1 Introduction", body) + self.assertEqual([(2, "1 Introduction"), (3, "1.1 Appearance")], outline) + + def test_drops_standalone_equation_labels_but_keeps_prose_headings(self): + _, body, outline = pdf_convert.finalize_pdf( + "Tone", + "## 2 Concepts\n\n#### (4)\n\nEquation prose.\n\n## 3 Results\n", + "silewp2007_002", + ) + self.assertNotIn("#### (4)", body) + self.assertEqual([(2, "2 Concepts"), (2, "3 Results")], outline) + + +class OutlineLockTests(unittest.TestCase): + def test_run_waits_on_global_outline_lock_even_for_different_output(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + with ExportLock(pdf_convert.OUTLINES), self.assertRaises(ExportBusyError): + pdf_convert.run(repo, out, update=False) + self.assertFalse(out.exists()) + + def test_update_and_reader_share_global_outline_lock(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + repo.mkdir() + first = Path(tmp) / "first" + second = Path(tmp) / "second" + with ExportLock(pdf_convert.OUTLINES): + with self.assertRaises(ExportBusyError): + pdf_convert.run(repo, first, update=True) + with self.assertRaises(ExportBusyError): + pdf_convert.run(repo, second, update=False) + def test_slug_path_replaces_windows_unsafe_and_reserved_names(self): + self.assertEqual("bad_name_.pdf", pdf_convert.slug_path("bad:name?.pdf")) + self.assertEqual("CON_.pdf", pdf_convert.slug_path("CON .pdf")) + self.assertEqual("_CON.txt", pdf_convert.slug_path("CON.txt")) + self.assertEqual("name.pdf_", pdf_convert.slug_path("name.pdf ")) + + def test_slug_sanitization_collisions_are_reported(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + first = repo / "CON.pdf" + second = repo / "_CON.pdf" + first.write_bytes(b"pdf") + second.write_bytes(b"pdf") + collisions = pdf_convert.destination_collisions(repo, out, [first, second]) + self.assertEqual(1, len(collisions)) + + def test_run_rejects_linked_outlines_before_reading_or_output_mutation(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + linked = repo / "locks.json" + linked.write_text("{}", encoding="utf-8") + + def mocked_link(path): + return path if path == linked else None + + with mock.patch.object(pdf_convert, "OUTLINES", linked), \ + mock.patch.object(pdf_convert, "first_link_in_path", side_effect=mocked_link), \ + mock.patch.object(Path, "read_text", side_effect=AssertionError("lock was read")), \ + self.assertRaises(pdf_convert.ManifestError): + pdf_convert.run(repo, out, update=False) + self.assertFalse(out.exists()) + + def test_promote_rejects_linked_lock_before_output_mutation(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + target = Path(tmp) / "keep-lock.json" + target.write_text('{"keep":true}', encoding="utf-8") + lock = Path(tmp) / "locks.json" + + def mocked_link(path): + return path if path == lock else None + + with mock.patch.object(pdf_convert, "first_link_in_path", side_effect=mocked_link), \ + self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, lock_path=lock, + ) + self.assertEqual('{"keep":true}', target.read_text(encoding="utf-8")) + + def test_promote_rejects_real_linked_lock_before_output_mutation(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + target = Path(tmp) / "keep-lock.json" + target.write_text('{"keep":true}', encoding="utf-8") + lock = Path(tmp) / "locks.json" + try: + lock.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, lock_path=lock, + ) + self.assertEqual('{"keep":true}', target.read_text(encoding="utf-8")) + + def test_promote_rejects_parent_traversing_lock_before_output_mutation(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + lock = out / ".." / "locks.json" + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, lock_path=lock, + ) + self.assertFalse((Path(tmp) / "locks.json").exists()) + + def test_normalized_outline_preserves_order_and_text(self): + self.assertEqual( + [[2, "Introduction"], [3, "A heading"], [2, "Introduction"]], + pdf_convert.normalize_outline( + [(2, " Introduction "), (3, "A heading"), (2, "Introduction")] + ), + ) + + def test_outline_lock_compares_complete_ordered_outline(self): + pin = {"outline": [[2, "One"], [2, "Two"]]} + self.assertTrue(pdf_convert.outline_matches(pin, [(2, "One"), (2, "Two")])) + self.assertFalse(pdf_convert.outline_matches(pin, [(2, "One"), (2, "Other")])) + self.assertFalse(pdf_convert.outline_matches(pin, [(2, "Two"), (2, "One")])) + self.assertFalse(pdf_convert.outline_matches(pin, [(3, "One"), (2, "Two")])) + self.assertFalse(pdf_convert.outline_matches(pin, [(2, "One"), (2, "Two"), (2, "Three")])) + self.assertFalse(pdf_convert.outline_matches(pin, [(2, "One")])) + + +class PdfTraceabilityTests(unittest.TestCase): + def test_yaml_scalar_escapes_controls_and_preserves_unicode(self): + for value in ('line\nnext\r\x00"é', "quote\\slash"): + encoded = yaml_scalar(value) + self.assertEqual(value, json.loads(encoded)) + self.assertNotRegex(encoded, r"[\x00-\x1f]") + + def test_frontmatter_uses_safe_scalars_at_nested_levels(self): + result = pdf_convert.frontmatter({ + "title": 'line\nnext\x00"é', + "metadata": {"author": "A\rB"}, + "tags": ["one\n two", "é"], + }) + self.assertIn('title: "line\\nnext\\u0000\\\"é"', result) + self.assertIn(' author: "A\\rB"', result) + self.assertIn(' - "one\\n two"', result) + self.assertNotRegex(result, r"[\x00\r\x01-\x08\x0b\x0c\x0e-\x1f]") + + def test_source_url_encodes_relative_path_segments(self): + self.assertEqual( + "https://example.test/blob/main/docs/My%20file%20%5Bx%5D.pdf", + pdf_convert._source_url( + "docs/My file [x].pdf", + "https://example.test/blob/main/{path}", + ), + ) + + def test_frontmatter_serializes_pdf_traceability_fields(self): + result = pdf_convert.frontmatter( + { + "source": "docs/example.pdf", + "source_url": "https://example.test/blob/main/docs/example.pdf", + "sha256": "abc123", + "pdf_metadata": {"title": "Example", "author": "A"}, + "structure": "bookmarks (2p)", + "outline_count": 3, + } + ) + self.assertIn('source_url: "https://example.test/blob/main/docs/example.pdf"', result) + self.assertIn('sha256: "abc123"', result) + self.assertIn('pdf_metadata:', result) + self.assertIn(' title: "Example"', result) + + +class ReplacementCharacterTests(unittest.TestCase): + def test_source_control_glyph_is_provenanced_as_replacement_risk(self): + self.assertEqual( + {"count": 1, "codepoints": ["U+001F"]}, + pdf_convert.source_replacement_details("valid\x1f text"), + ) + + def test_source_replacements_are_provenanced_but_exporter_replacements_are_fatal(self): + source = [{"page": 21, "count": 1}] + self.assertEqual( + {"source_count": 1, "exporter_count": 0}, + pdf_convert.replacement_provenance(source, 1), + ) + self.assertEqual( + {"source_count": 0, "exporter_count": 1}, + pdf_convert.replacement_provenance([], 1), + ) + + def test_run_rejects_exporter_created_replacement_character(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + (repo / "one.pdf").write_bytes(b"one") + + class FakeDoc: + def __init__(self): + self.metadata = {"title": "One"} + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + with mock.patch.object( + pdf_convert, "convert_pdf", + return_value=("## New\n�\n", "font-inference (1p)", []), + ), mock.patch.object(pdf_convert.fitz, "open", return_value=FakeDoc()), \ + mock.patch.object(pdf_convert, "OUTLINES", repo / "locks.json"): + (repo / "locks.json").write_text("{}", encoding="utf-8") + result, _ = pdf_convert.run(repo, out, update=True) + self.assertEqual(1, result["report"]["pdf_export_replacements"][0][1]["exporter_count"]) + self.assertFalse((out / "one.md").exists()) + + +class PdfOutputSafetyTests(unittest.TestCase): + def test_schema_one_manifest_is_rejected_before_mutation(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + victim = out / "victim.md" + victim.write_text("keep", encoding="utf-8") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text( + '{"files":{"victim.pdf":{"markdown":"victim.md",' + '"images":"victim_images"}}}', encoding="utf-8" + ) + before = sorted( + (p.relative_to(out).as_posix(), p.read_bytes()) + for p in out.rglob("*") if p.is_file() + ) + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + after = sorted( + (p.relative_to(out).as_posix(), p.read_bytes()) + for p in out.rglob("*") if p.is_file() + ) + self.assertEqual(before, after) + + def test_corrupt_manifest_cannot_delete_unrelated_directory(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + victim = out / "unrelated" + victim.mkdir() + (victim / "keep.txt").write_text("keep", encoding="utf-8") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text( + '{"schema":2,"files":{"guide.pdf":{"markdown":"guide.md",' + '"images":"unrelated"}}}', encoding="utf-8" + ) + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + self.assertTrue((victim / "keep.txt").exists()) + + def test_corrupt_manifest_cannot_replace_unrelated_file(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + victim = out / "unrelated.md" + victim.write_text("keep", encoding="utf-8") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text( + '{"schema":2,"files":{"guide.pdf":{"markdown":"unrelated.md",' + '"images":"guide_images"}}}', encoding="utf-8" + ) + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + self.assertEqual("keep", victim.read_text(encoding="utf-8")) + + def test_manifest_rejects_source_destination_mismatch(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + (out / pdf_convert.PDF_MANIFEST).write_text( + '{"schema":2,"files":{"guide.pdf":{"markdown":"other.md",' + '"images":"guide_images"}}}', encoding="utf-8" + ) + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + + def test_null_images_rejects_unclaimed_existing_canonical_directory(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + victim = out / "doc_images" + victim.mkdir() + (victim / "keep.txt").write_text("keep", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + self.assertEqual("keep", (victim / "keep.txt").read_text(encoding="utf-8")) + + def test_canonical_markdown_symlink_cannot_redirect_promotion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + keep = out / "keep.md" + keep.write_text("keep", encoding="utf-8") + link = out / "doc.md" + try: + link.symlink_to(keep) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + self.assertEqual("keep", keep.read_text(encoding="utf-8")) + self.assertTrue(link.is_symlink()) + + def test_canonical_image_symlink_cannot_redirect_promotion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + keep = out / "keep_images" + keep.mkdir() + (keep / "keep.txt").write_text("keep", encoding="utf-8") + link = out / "doc_images" + try: + link.symlink_to(keep, target_is_directory=True) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}}, + ) + self.assertEqual("keep", (keep / "keep.txt").read_text(encoding="utf-8")) + + def test_manifest_rejects_mocked_destination_link_chain_without_symlink_privilege(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + with mock.patch.object(pdf_convert, "first_link_in_path", return_value=out / "doc.md"), \ + self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + + def test_null_images_rejects_mocked_dangling_image_link_before_exists_check(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + + def mocked_link(path): + return path if path.name == "doc_images" else None + + with mock.patch.object(pdf_convert, "first_link_in_path", side_effect=mocked_link), \ + mock.patch.object(Path, "read_text", side_effect=AssertionError("manifest was read")), \ + self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + + def test_dangling_canonical_image_symlink_is_rejected(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + link = out / "doc_images" + try: + link.symlink_to(out / "does-not-exist", target_is_directory=True) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + + def test_symlinked_manifest_file_is_rejected_before_reading_json(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + target = out / "real-manifest.json" + target.write_text( + '{"schema":2,"files":{"doc.pdf":{"markdown":"doc.md",' + '"images":null}}}', encoding="utf-8" + ) + manifest = out / pdf_convert.PDF_MANIFEST + try: + manifest.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + + def test_mocked_symlinked_manifest_file_is_rejected_before_reading_json(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text("not json", encoding="utf-8") + + def mocked_link(path): + return path if path == manifest else None + + with mock.patch.object(pdf_convert, "first_link_in_path", side_effect=mocked_link), \ + mock.patch.object(Path, "read_text", side_effect=AssertionError("manifest was read")), \ + self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs(out, Path(tmp) / "stage", {}, {}) + + def test_case_only_source_rename_keeps_previous_output_owned(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + (out / "Guide.md").write_text("old", encoding="utf-8") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "guide.md").write_text("new", encoding="utf-8") + pdf_convert.promote_pdf_outputs( + out, stage, + {"Guide.pdf": {"markdown": "Guide.md", "images": None}}, + {"guide.pdf": {"markdown": "guide.md", "images": None}}, + ) + self.assertEqual("new", (out / "guide.md").read_text(encoding="utf-8")) + manifest = json.loads((out / pdf_convert.PDF_MANIFEST).read_text(encoding="utf-8")) + self.assertEqual("guide.md", manifest["files"]["guide.pdf"]["markdown"]) + + def test_case_only_image_rename_allows_authenticated_image_shrink(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + (out / "Guide.md").write_text("old", encoding="utf-8") + old_images = out / "Guide_images" + old_images.mkdir() + (old_images / "old.png").write_bytes(b"old") + # On a case-sensitive filesystem, keep a second spelling to make + # the null-image existence check exercise filesystem ownership. + # On a case-insensitive filesystem both names address the same output. + lower_images = out / "guide_images" + try: + lower_images.mkdir() + except FileExistsError: + pass + (lower_images / "keep.txt").write_text("keep", encoding="utf-8") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "guide.md").write_text("new", encoding="utf-8") + + if os.path.normcase("Guide_images") != os.path.normcase("guide_images"): + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"Guide.pdf": {"markdown": "Guide.md", "images": "Guide_images"}}, + {"guide.pdf": {"markdown": "guide.md", "images": None}}, + ) + self.assertTrue(old_images.exists()) + self.assertEqual("keep", (lower_images / "keep.txt").read_text(encoding="utf-8")) + return + + pdf_convert.promote_pdf_outputs( + out, stage, + {"Guide.pdf": {"markdown": "Guide.md", "images": "Guide_images"}}, + {"guide.pdf": {"markdown": "guide.md", "images": None}}, + ) + self.assertFalse(old_images.exists()) + self.assertEqual("new", (out / "guide.md").read_text(encoding="utf-8")) + + def test_missing_staged_markdown_rejects_before_prior_deletion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "old.md" + old.write_text("old", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", + {"old.pdf": {"markdown": "old.md", "images": None}}, + {"new.pdf": {"markdown": "new.md", "images": None}}, + ) + self.assertEqual("old", old.read_text(encoding="utf-8")) + + def test_nonregular_staged_markdown_rejects_before_prior_deletion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "old.md" + old.write_text("old", encoding="utf-8") + stage = Path(tmp) / "stage" + (stage / "new.md").mkdir(parents=True) + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": None}}, + {"new.pdf": {"markdown": "new.md", "images": None}}, + ) + self.assertEqual("old", old.read_text(encoding="utf-8")) + + def test_current_images_must_be_existing_directory_before_prior_deletion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "old.md" + old.write_text("old", encoding="utf-8") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "new.md").write_text("new", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": None}}, + {"new.pdf": {"markdown": "new.md", "images": "new_images"}}, + ) + self.assertEqual("old", old.read_text(encoding="utf-8")) + + def test_current_images_must_be_directory_before_prior_deletion(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "old.md" + old.write_text("old", encoding="utf-8") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "new.md").write_text("new", encoding="utf-8") + (stage / "new_images").write_text("not a directory", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": None}}, + {"new.pdf": {"markdown": "new.md", "images": "new_images"}}, + ) + self.assertEqual("old", old.read_text(encoding="utf-8")) + + def test_missing_prior_markdown_rejects_before_staging(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + stage = Path(tmp) / "stage" + (stage / "new.md").parent.mkdir(parents=True) + (stage / "new.md").write_text("new", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": None}}, + {"new.pdf": {"markdown": "new.md", "images": None}}, + ) + self.assertFalse((out / pdf_convert.PDF_MANIFEST).exists()) + + def test_prior_image_tree_link_is_rejected_before_backup(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + (out / "old.md").write_text("old", encoding="utf-8") + images = out / "old_images" + images.mkdir() + (images / "real.png").write_bytes(b"real") + nested = images / "nested" + nested.mkdir() + (nested / "real.png").write_bytes(b"real") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "new.md").write_text("new", encoding="utf-8") + with mock.patch.object( + pdf_convert, "first_link_in_path", + side_effect=lambda path: path if path.name == "nested" else None, + ), self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": "old_images"}}, + {"new.pdf": {"markdown": "new.md", "images": None}}, + ) + self.assertEqual("old", (out / "old.md").read_text(encoding="utf-8")) + self.assertEqual(b"real", (images / "real.png").read_bytes()) + + def test_prior_image_nested_symlink_is_rejected_when_supported(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + (out / "old.md").write_text("old", encoding="utf-8") + images = out / "old_images" + images.mkdir() + target = out / "keep.txt" + target.write_text("keep", encoding="utf-8") + link = images / "nested-link" + try: + link.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + stage = Path(tmp) / "stage" + stage.mkdir() + (stage / "new.md").write_text("new", encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, stage, + {"old.pdf": {"markdown": "old.md", "images": "old_images"}}, + {"new.pdf": {"markdown": "new.md", "images": None}}, + ) + self.assertEqual("keep", target.read_text(encoding="utf-8")) + + def test_manifest_rejects_traversal_and_casefold_collisions(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + for value in ( + ('{"schema":2,"files":{"guide.pdf":{"markdown":"../x.md",' + '"images":"guide_images"}}}'), + ('{"schema":2,"files":{"A.pdf":{"markdown":"a.md",' + '"images":"a_images"},"a.pdf":{"markdown":"a.md",' + '"images":"a_images"}}}'), + ): + (out / pdf_convert.PDF_MANIFEST).write_text(value, encoding="utf-8") + with self.assertRaises(pdf_convert.ManifestError): + pdf_convert.promote_pdf_outputs( + out, Path(tmp) / "stage", {}, {}, + ) + + def test_destination_collisions_are_reported_before_conversion(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + (repo / "a").mkdir(parents=True) + (repo / "a b.pdf").write_bytes(b"pdf") + (repo / "a_b.pdf").write_bytes(b"pdf") + collisions = pdf_convert.destination_collisions(repo, out) + self.assertEqual(1, len(collisions)) + self.assertEqual({"a b.pdf", "a_b.pdf"}, set(collisions[0][1])) + + def test_destination_collisions_casefold_output_keys(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + collisions = pdf_convert.destination_collisions( + repo, out, [repo / "A.pdf", repo / "a.pdf"] + ) + self.assertEqual(1, len(collisions)) + + def test_promotion_replaces_same_source_image_directory_when_images_shrink(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + old_images = out / "doc_images" + old_images.mkdir(parents=True) + (out / "doc.md").write_text("old", encoding="utf-8") + (old_images / "stale.png").write_bytes(b"old") + stage = Path(tmp) / "stage" + (stage / "doc_images").mkdir(parents=True) + (stage / "doc.md").write_text("new", encoding="utf-8") + pdf_convert.promote_pdf_outputs( + out, + stage, + {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}}, + {"doc.pdf": {"markdown": "doc.md", "images": None}}, + ) + self.assertFalse((old_images / "stale.png").exists()) + self.assertFalse(old_images.exists()) + manifest = json.loads( + (out / pdf_convert.PDF_MANIFEST).read_text(encoding="utf-8") + ) + self.assertIsNone(manifest["files"]["doc.pdf"]["images"]) + + def test_promotion_removes_old_destination_for_same_source(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + old = out / "old.md" + old.parent.mkdir(parents=True) + old.write_text("old", encoding="utf-8") + old_images = out / "old_images" + old_images.mkdir(parents=True) + stage = Path(tmp) / "stage" + (stage / "new.md").parent.mkdir(parents=True) + (stage / "new.md").write_text("new", encoding="utf-8") + (stage / "new_images").mkdir(parents=True) + pdf_convert.promote_pdf_outputs( + out, + stage, + {"old.pdf": {"markdown": "old.md", "images": "old_images"}}, + {"new.pdf": {"markdown": "new.md", "images": "new_images"}}, + ) + self.assertFalse(old.exists()) + self.assertEqual("new", (out / "new.md").read_text(encoding="utf-8")) + + def test_late_pdf_failure_preserves_previous_pdf_outputs(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + (repo / "one.pdf").write_bytes(b"one") + (repo / "two.pdf").write_bytes(b"two") + out.mkdir() + old = out / "one.md" + old.write_text("previous", encoding="utf-8") + (out / pdf_convert.PDF_MANIFEST).write_text( + '{"schema":2,"files":{"one.pdf":{"markdown":"one.md",' + '"images":"one_images"}}}', + encoding="utf-8", + ) + + def fake_convert(pdf, out_md, image_dir): + if pdf.name == "two.pdf": + raise RuntimeError("late failure") + return "## New\n", "font-inference (1p)", [] + + class FakeDoc: + def __init__(self): + self.metadata = {"title": "One"} + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + with mock.patch.object(pdf_convert, "convert_pdf", fake_convert), \ + mock.patch.object(pdf_convert.fitz, "open", return_value=FakeDoc()), \ + mock.patch.object(pdf_convert, "OUTLINES", repo / "locks.json"): + (repo / "locks.json").write_text( + '{"one.pdf":{"outline":[[2,"New"]]},' + '"two.pdf":{"outline":[[2,"New"]]}}', + encoding="utf-8", + ) + result, _ = pdf_convert.run(repo, out, update=False) + self.assertIn("two.pdf", [item[0] for item in result["report"]["pdf_failures"]]) + self.assertEqual("previous", old.read_text(encoding="utf-8")) + + def test_missing_lock_is_fatal_before_promoting_staged_output(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + (repo / "ONE.PDF").write_bytes(b"one") + + class FakeDoc: + def __init__(self): + self.metadata = {"title": "One"} + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + with mock.patch.object(pdf_convert, "convert_pdf", + return_value=("## New\n", "font-inference (1p)", [])), \ + mock.patch.object(pdf_convert.fitz, "open", return_value=FakeDoc()), \ + mock.patch.object(pdf_convert, "OUTLINES", repo / "locks.json"): + (repo / "locks.json").write_text("{}", encoding="utf-8") + result, _ = pdf_convert.run(repo, out, update=False) + self.assertIn("ONE.PDF", result["report"]["outline_unpinned"]) + self.assertFalse((out / "ONE.md").exists()) + + def test_changed_pdf_set_prunes_stale_outputs_and_keeps_chm_sentinel(self): + with tempfile.TemporaryDirectory() as tmp: + repo = Path(tmp) / "repo" + out = Path(tmp) / "out" + repo.mkdir() + one = repo / "one.pdf" + two = repo / "two.pdf" + one.write_bytes(b"one") + two.write_bytes(b"two") + sentinel = out / "CHM" / "topic.md" + sentinel.parent.mkdir(parents=True) + sentinel.write_text("keep", encoding="utf-8") + + class FakeDoc: + def __init__(self): + self.metadata = {"title": "Document"} + + def __enter__(self): + return self + + def __exit__(self, *args): + return False + + def fake_convert(pdf, out_md, image_dir): + image_dir.mkdir(parents=True, exist_ok=True) + (image_dir / "figure.png").write_bytes(pdf.read_bytes()) + return "## New\n", "font-inference (1p)", [] + + with mock.patch.object(pdf_convert, "convert_pdf", fake_convert), \ + mock.patch.object(pdf_convert.fitz, "open", return_value=FakeDoc()), \ + mock.patch.object(pdf_convert, "OUTLINES", repo / "locks.json"): + (repo / "locks.json").write_text("{}", encoding="utf-8") + pdf_convert.run(repo, out, update=True) + written_manifest = json.loads( + (out / pdf_convert.PDF_MANIFEST).read_text(encoding="utf-8") + ) + self.assertEqual(2, written_manifest["schema"]) + two.unlink() + pdf_convert.run(repo, out, update=True) + self.assertTrue((out / "one.md").exists()) + self.assertFalse((out / "two.md").exists()) + self.assertFalse((out / "two_images").exists()) + self.assertEqual("keep", sentinel.read_text(encoding="utf-8")) + + def test_promotion_failure_restores_previous_pdf_files(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "doc.md" + old_images = out / "doc_images" + old_images.mkdir() + old.write_text("old", encoding="utf-8") + (old_images / "old.png").write_bytes(b"old") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text('{"schema":2,"files":{"doc.pdf":{"markdown":"doc.md",' + '"images":"doc_images"}}}', + encoding="utf-8") + old_manifest = manifest.read_text(encoding="utf-8") + stage = Path(tmp) / "stage" + (stage / "doc_images").mkdir(parents=True) + (stage / "doc.md").write_text("new", encoding="utf-8") + previous = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + current = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + def fail_move(*args, **kwargs): + raise OSError("disk full") + + with mock.patch.object(pdf_convert.shutil, "move", fail_move), self.assertRaises(OSError): + pdf_convert.promote_pdf_outputs(out, stage, previous, current) + self.assertEqual("old", old.read_text(encoding="utf-8")) + self.assertEqual(b"old", (old_images / "old.png").read_bytes()) + self.assertEqual(old_manifest, manifest.read_text(encoding="utf-8")) + + def test_manifest_promotion_failure_restores_previous_pdf_set(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "doc.md" + old_images = out / "doc_images" + old_images.mkdir() + old.write_text("old", encoding="utf-8") + (old_images / "old.png").write_bytes(b"old") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text('{"schema":2,"files":{"doc.pdf":{"markdown":"doc.md",' + '"images":"doc_images"}}}', + encoding="utf-8") + old_manifest = manifest.read_text(encoding="utf-8") + stage = Path(tmp) / "stage" + (stage / "doc_images").mkdir(parents=True) + (stage / "doc.md").write_text("new", encoding="utf-8") + previous = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + current = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + original_move = pdf_convert.shutil.move + + def fail_manifest(source, target): + if Path(target) == manifest: + raise OSError("manifest disk full") + return original_move(source, target) + + with mock.patch.object(pdf_convert.shutil, "move", fail_manifest), self.assertRaises(OSError): + pdf_convert.promote_pdf_outputs(out, stage, previous, current) + self.assertEqual("old", old.read_text(encoding="utf-8")) + self.assertEqual(b"old", (old_images / "old.png").read_bytes()) + self.assertEqual(old_manifest, manifest.read_text(encoding="utf-8")) + + def test_lock_promotion_failure_restores_outputs_manifest_and_old_lock(self): + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "out" + out.mkdir() + old = out / "doc.md" + old_images = out / "doc_images" + old_images.mkdir() + old.write_text("old", encoding="utf-8") + (old_images / "old.png").write_bytes(b"old") + manifest = out / pdf_convert.PDF_MANIFEST + manifest.write_text('{"schema":2,"files":{"doc.pdf":{"markdown":"doc.md",' + '"images":"doc_images"}}}', + encoding="utf-8") + old_manifest = manifest.read_text(encoding="utf-8") + lock = Path(tmp) / "locks.json" + lock.write_text('{"old":true}\n', encoding="utf-8") + old_lock = lock.read_text(encoding="utf-8") + stage = Path(tmp) / "stage" + (stage / "doc_images").mkdir(parents=True) + (stage / "doc.md").write_text("new", encoding="utf-8") + previous = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + current = {"doc.pdf": {"markdown": "doc.md", "images": "doc_images"}} + original_move = pdf_convert.shutil.move + + def fail_lock(source, target): + if Path(target) == lock: + raise OSError("lock disk full") + return original_move(source, target) + + with mock.patch.object(pdf_convert.shutil, "move", fail_lock), self.assertRaises(OSError): + pdf_convert.promote_pdf_outputs( + out, stage, previous, current, lock_path=lock, + fresh={"doc.pdf": {"outline": [[2, "New"]]}}, + ) + self.assertEqual("old", old.read_text(encoding="utf-8")) + self.assertEqual(b"old", (old_images / "old.png").read_bytes()) + self.assertEqual(old_manifest, manifest.read_text(encoding="utf-8")) + self.assertEqual(old_lock, lock.read_text(encoding="utf-8")) + + +class StripFurnitureTests(unittest.TestCase): + def test_removes_a_running_header_used_on_only_two_pages(self): + pages = [ + "# **1 Ingredients**\n\n1\n\nFirst page body.", + "# **1 Ingredients**\n\n2\n\nSecond page body.", + ] + [f"# **Section {i}**\n\n{i}\n\nPage {i} body." for i in range(3, 14)] + + cleaned = pdf_convert.strip_furniture(pages) + + self.assertEqual("First page body.", cleaned[0]) + self.assertEqual("Second page body.", cleaned[1]) + + def test_keeps_distinct_numbered_headings_at_page_boundaries(self): + pages = [ + "Scenario 1\n\nFirst scenario body.", + "Scenario 2\n\nSecond scenario body.", + "Different heading\n\nThird body.", + ] + + self.assertTrue(pdf_convert.strip_furniture(pages)[0].startswith("Scenario 1")) + self.assertTrue(pdf_convert.strip_furniture(pages)[1].startswith("Scenario 2")) + + def test_keeps_repeated_boundary_prose(self): + pages = [ + "First page body.\n\nClick OK.", + "Second page body.\n\nClick OK.", + "Third page body.\n\nDifferent ending.", + ] + + cleaned = pdf_convert.strip_furniture(pages) + + self.assertTrue(cleaned[0].endswith("Click OK.")) + self.assertTrue(cleaned[1].endswith("Click OK.")) + + +class LuaTableTests(unittest.TestCase): + @unittest.skipUnless(shutil.which("pandoc"), "pandoc is required") + def test_headerless_table_keeps_a_neutral_header(self): + html = ("" + "
Alpha1
Beta2
") + result = subprocess.run( + ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none", + f"--lua-filter={Path(__file__).with_name('fwhelp.lua')}"], + input=html, + capture_output=True, + check=True, + text=True, + encoding="utf-8", + ) + + self.assertRegex(result.stdout.splitlines()[0], r"^\|\s*\|\s*\|$") + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_reporting.py b/tools/test_reporting.py new file mode 100644 index 00000000..cacced65 --- /dev/null +++ b/tools/test_reporting.py @@ -0,0 +1,83 @@ +import unittest + +from reporting import Issue, Report + + +class ReportingTests(unittest.TestCase): + def test_markdown_report_gives_robohelp_authors_paths_evidence_and_guidance(self): + report = Report([ + Issue( + "source_missing_link", + "missing | target\nsecond line", + "Using_Tools/topic.htm", + detail=["Using_Tools/topic.htm", "missing.htm"], + ), + ], metadata={ + "source_ref": "abc1234", + "chm_count": 2, + "topic_count": 1630, + "image_count": 583, + "pdf_count": 13, + }) + + markdown = report.to_markdown() + + self.assertIn("# Author quality report", markdown) + self.assertIn("`abc1234`", markdown) + self.assertIn("| CHMs | 2 |", markdown) + self.assertIn("## Missing local link (`source_missing_link`)", markdown) + self.assertIn("**How to fix in RoboHelp:**", markdown) + self.assertIn("Open the source topic", markdown) + self.assertIn("`Using_Tools/topic.htm`", markdown) + self.assertIn("missing \\| target
second line", markdown) + self.assertIn('["Using_Tools/topic.htm", "missing.htm"]', markdown) + self.assertIn("[author-report.json](author-report.json)", markdown) + + def test_markdown_report_neutralizes_source_markdown_and_backslashes(self): + report = Report([ + Issue( + "source_missing_link", + "[click](javascript:alert(1)) \\| raw `text`", + "topic[1].htm", + detail={"target": "[bad](missing.htm)"}, + ), + ]) + + markdown = report.to_markdown() + + self.assertNotIn("[click](javascript:alert(1))", markdown) + self.assertNotIn("[bad](missing.htm)", markdown) + self.assertIn("[click](javascript:alert(1))", markdown) + self.assertIn("\\\| raw `text`", markdown) + self.assertIn("`topic[1].htm`", markdown) + + def test_same_issue_catalog_renders_json_readme_and_console(self): + report = Report() + report.add(Issue("source_missing_link", "Missing link", "page.md")) + report.add(Issue("missing_image", "Missing image", "page.md", fatal=True)) + data = report.as_dict() + self.assertEqual(2, data["summary"]["total"]) + self.assertEqual(1, data["summary"]["fatal"]) + self.assertIn("Missing local image", report.to_readme()) + self.assertIn("FATAL", report.to_console()) + self.assertIn('"fatal": 1', report.to_json()) + self.assertIn('"label": "Missing local image"', report.to_json()) + self.assertIn("Missing local image", report.to_console()) + + def test_console_keeps_all_fatals_and_bounds_advisory_detail(self): + issues = [Issue("source_missing_link", f"advisory-{index}", "source.md") + for index in range(500)] + issues.extend(Issue("replacement_character", f"fatal-{index}", f"page-{index}.md", True) + for index in range(4)) + report = Report(issues) + console = report.to_console() + self.assertLess(len(console), 4000) + for index in range(4): + self.assertIn(f"fatal-{index}", console) + self.assertIn("advisories: 500 issue(s) in 1 kind(s)", console) + self.assertNotIn("advisory-499", console) + self.assertIn("author-report.json", console) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_source_safety.py b/tools/test_source_safety.py new file mode 100644 index 00000000..e270f85f --- /dev/null +++ b/tools/test_source_safety.py @@ -0,0 +1,173 @@ +import os +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +from source_safety import SourceSafetyError, discover_source_files, validate_source_tree + + +class SourceSafetyTests(unittest.TestCase): + def test_discovers_regular_files_in_deterministic_casefolded_order(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "z.PDF").write_bytes(b"") + (root / "A.pdf").write_bytes(b"") + nested = root / "nested" + nested.mkdir() + (nested / "b.Pdf").write_bytes(b"") + + self.assertEqual( + ["A.pdf", "nested/b.Pdf", "z.PDF"], + [path.relative_to(root).as_posix() for path in discover_source_files( + root, suffixes={".pdf"}, recursive=True + )], + ) + + def test_root_level_discovery_does_not_include_nested_files(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "root.CHM").write_bytes(b"") + (root / "nested").mkdir() + (root / "nested" / "nested.chm").write_bytes(b"") + + self.assertEqual( + [root / "root.CHM"], + discover_source_files(root, suffixes={".chm"}, recursive=False), + ) + + def test_excluded_directory_is_not_traversed(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + excluded = root / ".git" + excluded.mkdir() + (excluded / "hidden.pdf").write_bytes(b"") + + self.assertEqual([], discover_source_files( + root, suffixes={".pdf"}, recursive=True + )) + + def test_ignored_nonmatching_symlink_does_not_abort_discovery(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + target = root / "notes.txt" + target.write_bytes(b"") + link = root / "notes.link" + try: + link.symlink_to(target) + except (OSError, NotImplementedError): + self.skipTest("symlinks unavailable") + + self.assertEqual([], discover_source_files( + root, suffixes={".pdf"}, recursive=True + )) + + def test_relative_root_returns_relative_paths(self): + with tempfile.TemporaryDirectory() as raw: + absolute_root = Path(raw) + (absolute_root / "file.PDF").write_bytes(b"") + relative_root = Path(os.path.relpath(absolute_root, Path.cwd())) + + found = discover_source_files( + relative_root, suffixes={".pdf"}, recursive=False + ) + + self.assertFalse(found[0].is_absolute()) + self.assertEqual(relative_root, found[0].parent) + + def test_rejects_matching_symlink_instead_of_following_it(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + outside = root.parent / (root.name + "-outside") + outside.mkdir() + target = outside / "real.pdf" + target.write_bytes(b"") + link = root / "linked.pdf" + try: + link.symlink_to(target) + except (OSError, NotImplementedError): + target.unlink() + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + with self.assertRaises(SourceSafetyError): + discover_source_files(root, suffixes={".pdf"}, recursive=True) + finally: + target.unlink() + outside.rmdir() + + def test_ignores_symlink_directory_without_source_suffix(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + outside = Path(raw).parent / (Path(raw).name + "-outside") + outside.mkdir() + (outside / "outside.pdf").write_bytes(b"") + link = root / "linked" + try: + link.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + (outside / "outside.pdf").unlink() + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + self.assertEqual([], discover_source_files( + root, suffixes={".pdf"}, recursive=True + )) + finally: + (outside / "outside.pdf").unlink() + outside.rmdir() + + def test_symlinked_directory_with_source_suffix_is_not_a_regular_candidate(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + outside = root.parent / (root.name + "-directory") + outside.mkdir() + link = root / "linked.pdf" + try: + link.symlink_to(outside, target_is_directory=True) + except (OSError, NotImplementedError): + outside.rmdir() + self.skipTest("symlinks unavailable") + try: + self.assertEqual([], discover_source_files( + root, suffixes={".pdf"}, recursive=True + )) + finally: + outside.rmdir() + + def test_relative_root_with_dotdot_is_normalized_for_consumers(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + nested = root / "nested" + nested.mkdir() + (root / "guide.pdf").write_bytes(b"") + caller_root = nested / ".." + found = discover_source_files( + caller_root, suffixes={".pdf"}, recursive=True + ) + self.assertEqual([root / "guide.pdf"], found) + + def test_source_path_chain_helper_detects_ancestor_links(self): + from source_safety import first_link_in_path + + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + self.assertIsNone(first_link_in_path(root / "child" / "file.pdf")) + + def test_tree_validator_rejects_descendant_link_without_following_it(self): + with tempfile.TemporaryDirectory() as raw: + root = Path(raw) + (root / "topic.htm").write_text("topic", encoding="utf-8") + descendant = root / "linked.htm" + descendant.write_text("linked", encoding="utf-8") + + def fake_first_link(path): + return descendant if Path(path) == descendant else None + + with mock.patch("source_safety.first_link_in_path", side_effect=fake_first_link), \ + self.assertRaises(SourceSafetyError): + validate_source_tree(root) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/test_workflow.py b/tools/test_workflow.py new file mode 100644 index 00000000..f8454c30 --- /dev/null +++ b/tools/test_workflow.py @@ -0,0 +1,117 @@ +"""Static guardrails for the publication workflow's safety contract.""" + +import unittest +from pathlib import Path + +import yaml + +WORKFLOW = Path(__file__).parents[1] / ".github" / "workflows" / "markdown-export.yml" + +# Contexts GitHub refuses to evaluate in jobs..env. Referencing one there is +# a load-time error ("Unrecognized named-value"), which fails the whole run +# before any job starts and produces no logs to diagnose it from. +JOB_ENV_FORBIDDEN_CONTEXTS = ("runner", "steps", "job", "env", "secrets") + + +class WorkflowGuardrailTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.text = WORKFLOW.read_text(encoding="utf-8") + cls.workflow = yaml.safe_load(cls.text) + + def test_job_env_avoids_contexts_github_rejects_at_load_time(self): + for job_id, job in self.workflow["jobs"].items(): + for name, value in (job.get("env") or {}).items(): + for context in JOB_ENV_FORBIDDEN_CONTEXTS: + self.assertNotRegex( + str(value), + r"\$\{\{\s*" + context + r"\.", + f"{job_id}.env.{name} uses the {context} context, " + "which GitHub rejects in job-level env", + ) + + def test_convert_repo_is_relative_to_the_directory_uv_runs_in(self): + steps = self.workflow["jobs"]["validate"]["steps"] + convert = next(step for step in steps if step.get("name") == "Convert") + # "--directory src" changes uv's working directory, so a "--repo src" + # here would resolve to src/src and fail before a single topic converts. + self.assertIn("--directory src", convert["run"]) + self.assertIn("--repo .", convert["run"]) + self.assertNotIn("--repo src", convert["run"]) + + def test_build_paths_are_resolved_from_runner_temp_in_a_step(self): + steps = self.workflow["jobs"]["validate"]["steps"] + resolve = steps[0] + self.assertEqual("Resolve build paths", resolve["name"]) + for name in ("EXPORT_DIR", "WORK_DIR", "DIAGNOSTICS"): + self.assertIn(f"{name}=${{RUNNER_TEMP}}", resolve["run"]) + self.assertIn('>> "$GITHUB_ENV"', resolve["run"]) + + def test_summary_consumes_canonical_report_schema(self): + self.assertIn('r.get("corpus", {})', self.text) + self.assertIn('r.get("summary", {})', self.text) + self.assertIn('r.get("issues", [])', self.text) + self.assertNotIn("r.get('topics',0)", self.text) + + def test_publication_removes_tracked_stale_files(self): + self.assertIn("git -C \"${PUBLISH_DIR}\" rm -r -q --ignore-unmatch -- .", self.text) + self.assertIn('git -C "${PUBLISH_DIR}" clean -fdx -q', self.text) + + def test_publication_keeps_history_and_never_overwrites(self): + self.assertIn("refs/heads/${EXPORT_BRANCH}:refs/remotes/origin/${EXPORT_BRANCH}", self.text) + self.assertNotIn("push --force", self.text) + self.assertNotIn("push --force-with-lease", self.text) + self.assertIn('git push origin "refs/tags/${TAG}"', self.text) + + def test_dry_run_does_not_attempt_release_tagging(self): + self.assertIn("github.event_name != 'workflow_dispatch' || !inputs.dry_run", self.text) + + def test_validation_trigger_and_read_only_permissions(self): + self.assertIn("pull_request:", self.text) + self.assertIn("permissions:\n contents: read", self.text) + self.assertIn("validate:", self.text) + self.assertLess(self.text.index("uv lock --check"), self.text.index("uv sync --frozen")) + self.assertIn("uv sync --frozen", self.text) + + def test_actions_are_immutable_pins(self): + for pin in ( + "actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09", + "astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e", + "actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02", + "actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0", + ): + self.assertIn(pin, self.text) + self.assertNotRegex(self.text, r"uses:\s+[^\s]+@v\d+") + + def test_pandoc_digest_is_checked_before_install(self): + digest = "ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567" + self.assertIn(digest, self.text) + self.assertIn("sha256sum --check --strict", self.text) + self.assertLess(self.text.index("sha256sum --check --strict"), self.text.index("dpkg -i")) + + def test_publish_is_separate_write_job_after_validation(self): + self.assertRegex(self.text, r"publish:\s*[\s\S]+?permissions:\s*\n\s+contents: write") + self.assertIn("needs: validate", self.text) + self.assertIn("actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0", self.text) + publish = self.text[self.text.index("publish:"):] + self.assertNotIn("github.event_name == 'pull_request'", publish) + + def test_diagnostics_are_always_uploaded_and_summary_reads_external_report(self): + self.assertIn("if: always()", self.text) + self.assertIn("markdown-export-diagnostics.json", self.text) + self.assertIn("if-no-files-found: ignore", self.text) + self.assertIn("GITHUB_STEP_SUMMARY", self.text) + + def test_completed_export_preserves_hidden_files(self): + completed = self.text[self.text.index("- name: Upload completed export"):] + self.assertIn("include-hidden-files: true", completed) + self.assertEqual(1, self.text.count("include-hidden-files: true")) + + def test_summary_uses_canonical_codes_and_issue_labels(self): + self.assertIn('"source_missing_link"', self.text) + self.assertIn('"source_missing_image"', self.text) + self.assertIn('item.get("label",', self.text) + + +if __name__ == "__main__": + unittest.main() diff --git a/uv.lock b/uv.lock new file mode 100644 index 00000000..c20ac19c --- /dev/null +++ b/uv.lock @@ -0,0 +1,231 @@ +version = 1 +revision = 3 +requires-python = ">=3.13.5, <3.14" + +[[package]] +name = "flatbuffers" +version = "25.12.19" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/e8/2d/d2a548598be01649e2d46231d151a6c56d10b964d94043a335ae56ea2d92/flatbuffers-25.12.19-py2.py3-none-any.whl", hash = "sha256:7634f50c427838bb021c2d66a3d1168e9d199b0607e6329399f04846d42e20b4", size = 26661, upload-time = "2025-12-19T23:16:13.622Z" }, +] + +[[package]] +name = "fwhelps" +version = "0.0.0" +source = { virtual = "." } +dependencies = [ + { name = "pymupdf" }, + { name = "pymupdf4llm" }, +] + +[package.dev-dependencies] +dev = [ + { name = "ruff" }, +] + +[package.metadata] +requires-dist = [ + { name = "pymupdf", specifier = "==1.28.2" }, + { name = "pymupdf4llm", specifier = "==1.28.2" }, +] + +[package.metadata.requires-dev] +dev = [{ name = "ruff", specifier = "==0.16.4" }] + +[[package]] +name = "networkx" +version = "3.6.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/6a/51/63fe664f3908c97be9d2e4f1158eb633317598cfa6e1fc14af5383f17512/networkx-3.6.1.tar.gz", hash = "sha256:26b7c357accc0c8cde558ad486283728b65b6a95d85ee1cd66bafab4c8168509", size = 2517025, upload-time = "2025-12-08T17:02:39.908Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/9e/c9/b2622292ea83fbb4ec318f5b9ab867d0a28ab43c5717bb85b0a5f6b3b0a4/networkx-3.6.1-py3-none-any.whl", hash = "sha256:d47fbf302e7d9cbbb9e2555a0d267983d2aa476bac30e90dfbe5669bd57f3762", size = 2068504, upload-time = "2025-12-08T17:02:38.159Z" }, +] + +[[package]] +name = "numpy" +version = "2.5.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9a/80/db0b4559e57ec36362bedbb05530a87fafbcb6067708c946967a41d449e7/numpy-2.5.2.tar.gz", hash = "sha256:d482d171c406ae88c5b19cad3b6a1c4c5209f886ab74bc44c2c865c23f52d860", size = 20773161, upload-time = "2026-08-09T13:48:27.962Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f5/d2/6b24738a0ef4557d189b150046cd07823c50e4273e8aebd651222e24306f/numpy-2.5.2-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:8e4cb9a754c8a0c62eaa88273a5fba3391f4a610d1dee893c0755da31c083f15", size = 16886595, upload-time = "2026-08-09T13:45:27.323Z" }, + { url = "https://files.pythonhosted.org/packages/65/60/f2d208d366f263f39c6e69ed309290717aab41078b6d04c9be2a84fa2a07/numpy-2.5.2-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:52c808f96484f5571a5cc863775ce50247c17dfb3b0361f8ed6b4b0456f80080", size = 11896845, upload-time = "2026-08-09T13:45:31.638Z" }, + { url = "https://files.pythonhosted.org/packages/3c/79/81e0bf24f4d020a2b1d5cd297a9f60c3f24eeb116f9bba5870443f7b6a4a/numpy-2.5.2-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:29d81e97f668489cba8ebfd796b9bdd453525d35dd9e162e2daec94bf3fc7740", size = 5343880, upload-time = "2026-08-09T13:45:34.373Z" }, + { url = "https://files.pythonhosted.org/packages/ba/cc/e3141cf06d1a8a2c7e107543fe1269c1d1af760d4d683c0794a4ee1127c2/numpy-2.5.2-cp313-cp313-macosx_14_0_x86_64.whl", hash = "sha256:afb3f0632d6b2e3ba04dbce8d1e48d321b369138b73830b5ca371a0e8d479d56", size = 6682264, upload-time = "2026-08-09T13:45:36.7Z" }, + { url = "https://files.pythonhosted.org/packages/29/f1/2a64a307d92c5d98f5255a4014eb43bb6103ee477087b61ecae44a3aa9b9/numpy-2.5.2-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:0aadf13b60048d501e05fa699efaf7734e2494f3498a4c2a5521d822640324f3", size = 15609566, upload-time = "2026-08-09T13:45:39.518Z" }, + { url = "https://files.pythonhosted.org/packages/7b/44/59a1eb68e773c4098d107ef34a0dbdeca501d72ffcfbff9a7707343921ce/numpy-2.5.2-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:29b86ff8a6cc556b47ec6b64b194815cc80e6bf5eedcc6cddfd65318cb0b4eee", size = 16709995, upload-time = "2026-08-09T13:45:43.661Z" }, + { url = "https://files.pythonhosted.org/packages/8a/4c/3e54d4ddbc359a1295f8b633e8106bcd4d7d4a206e82df051bdfb3058755/numpy-2.5.2-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:6950c4b7dd562453090548ba7f5da7e59f57f85663f15d5dcc60e249192f7e59", size = 16972511, upload-time = "2026-08-09T13:45:47.094Z" }, + { url = "https://files.pythonhosted.org/packages/f2/9f/02e371638ebf19b66d46231e4be52999e87f32d1961b113bc45656608b22/numpy-2.5.2-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:b9727f472d2f3888053b8a75ab0cb94745a9de224bb5846dbadc0092101bc71d", size = 18465609, upload-time = "2026-08-09T13:45:50.808Z" }, + { url = "https://files.pythonhosted.org/packages/eb/ae/ad6645abc7a3510fe48e8ea1ab4598166f500057ef4ebf38bfad4f1577de/numpy-2.5.2-cp313-cp313-win32.whl", hash = "sha256:4f9744f9fbdcea0bc552e8f19e1f141f811a3f9bc2be2cc6e86d982cab23e3f4", size = 6070204, upload-time = "2026-08-09T13:45:54.111Z" }, + { url = "https://files.pythonhosted.org/packages/15/20/f3489f86d81ea460b2bcdceaed094142ca6579f6be0ec527b781d39afe68/numpy-2.5.2-cp313-cp313-win_amd64.whl", hash = "sha256:85aaccb24182c25df891ad0ec333585967e115269d5f1b17f2c9ae005bc96657", size = 12460532, upload-time = "2026-08-09T13:45:57.167Z" }, + { url = "https://files.pythonhosted.org/packages/d5/21/35b31dde1b283b79de828b80f876afd8c94e28fe1e9c375f89e261cc4c0d/numpy-2.5.2-cp313-cp313-win_arm64.whl", hash = "sha256:bd68ece1553d2023c09a4226d9e41c586ad2d20594d1a456186c33513d2cb3f2", size = 10396725, upload-time = "2026-08-09T13:46:00.478Z" }, +] + +[[package]] +name = "onnxruntime" +version = "1.29.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "flatbuffers" }, + { name = "numpy" }, + { name = "packaging" }, + { name = "protobuf" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/41/f8/d375facf60edaf41f5732f9f689c98a800fcc52df5cf6ddfb406703eb5a1/onnxruntime-1.29.0-cp313-cp313-macosx_14_0_arm64.whl", hash = "sha256:be0f8ed688cfb1d4d5765a137193b7bfab0c8ea214eed99260b380bb525a3a7f", size = 21429708, upload-time = "2026-08-17T22:54:01.44Z" }, + { url = "https://files.pythonhosted.org/packages/c9/17/b9ad04051a8c4f504852ce0e8e10f9a6b2f1a331eedcdcc503df776dd0ea/onnxruntime-1.29.0-cp313-cp313-manylinux_2_28_aarch64.whl", hash = "sha256:d67673c5367727860922c5262d724472f1b5539fb7ccf4c81a638f9b71719803", size = 20816263, upload-time = "2026-08-17T22:54:04.088Z" }, + { url = "https://files.pythonhosted.org/packages/83/2c/d8eb945d2a372149df9705a8d5c8d7c6c46c987c5446dbcea9e1ea7f6556/onnxruntime-1.29.0-cp313-cp313-manylinux_2_28_x86_64.whl", hash = "sha256:e2128f31f449e922c62dbe5d8b6b7b079f0bcaf2d56a102fa203cb6e5bb5ab19", size = 23136817, upload-time = "2026-08-17T22:54:06.714Z" }, + { url = "https://files.pythonhosted.org/packages/e1/3b/66b424c63fa92dfaa48d1719efaae66fc8c256b9426a832eda51d8dfe1e9/onnxruntime-1.29.0-cp313-cp313-win_amd64.whl", hash = "sha256:2945e1f82f81f27e88decea88c7861f45baea23818950d467bf3909aa303119e", size = 14001310, upload-time = "2026-08-17T22:54:09.13Z" }, + { url = "https://files.pythonhosted.org/packages/83/22/d6a700e3a6322fa3d56fbe7cee9ffc53f35e77ffcd6b7e97f4b7722a27ab/onnxruntime-1.29.0-cp313-cp313-win_arm64.whl", hash = "sha256:4b940b0d777590c7e20bf298f5c16af1ea6ad1b400a1c822a6be192f64f4d954", size = 13747112, upload-time = "2026-08-17T22:54:11.608Z" }, + { url = "https://files.pythonhosted.org/packages/4a/89/c4af146de3d60a32c89fea48d5d34bfd044faaf8957270043a03bd1b462b/onnxruntime-1.29.0-cp313-cp313t-manylinux_2_28_aarch64.whl", hash = "sha256:533f8370ce124304e5cb08ab961836cf755631e3dd77adc5f3bbdab70c2b7d99", size = 20826136, upload-time = "2026-08-17T22:54:14.315Z" }, + { url = "https://files.pythonhosted.org/packages/9d/f2/e6bbacd11dfe8d070613261a758795ea128b9fc9bea391a2a7da2e4c7a08/onnxruntime-1.29.0-cp313-cp313t-manylinux_2_28_x86_64.whl", hash = "sha256:c1ad3f437153fe77f9d01a08fbaac0beb030e09b8a80ace1603bcf69b6c95481", size = 23138951, upload-time = "2026-08-17T22:54:17.154Z" }, +] + +[[package]] +name = "packaging" +version = "26.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", hash = "sha256:94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79", size = 313412, upload-time = "2026-08-04T18:15:28.737Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", hash = "sha256:d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c", size = 129956, upload-time = "2026-08-04T18:15:27.159Z" }, +] + +[[package]] +name = "protobuf" +version = "7.36.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a7/e7/0553e21d25ca4d9f573135775348a372c3ec34a93a71d5f297c3bac38341/protobuf-7.36.0.tar.gz", hash = "sha256:e8e09cb0d794c6687926fa558a8a6e72aa10edb997d5ca61da0765f12a3e00ea", size = 510034, upload-time = "2026-08-20T16:34:01.071Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/8f/ae/58e3ca96cb2e118cc546b677359b3c6659f79a140935c08dec94c7998585/protobuf-7.36.0-cp310-abi3-macosx_10_9_universal2.whl", hash = "sha256:9103532dffd80c6fab7e50c65a31007680a06eb57537d437bb1b35812c138a37", size = 453256, upload-time = "2026-08-20T16:33:53.945Z" }, + { url = "https://files.pythonhosted.org/packages/f0/15/5162230af4912697f0fe406f6800f80760945babcff0e2c2fe6c84ef2d5d/protobuf-7.36.0-cp310-abi3-manylinux2014_aarch64.whl", hash = "sha256:bf94a5917c71058262de683669bc0a797a7669d3de71f0b36d058e3194f47b44", size = 341436, upload-time = "2026-08-20T16:33:55.134Z" }, + { url = "https://files.pythonhosted.org/packages/d7/09/1670b2bfc9a45e807e520c3e9be36524db9ccc7dc05ea17af7681cabdc61/protobuf-7.36.0-cp310-abi3-manylinux2014_s390x.whl", hash = "sha256:3297e60abdff301e5f74393d87f6cc59dacab5f024a89548a6e8de1d26576b16", size = 354440, upload-time = "2026-08-20T16:33:56.077Z" }, + { url = "https://files.pythonhosted.org/packages/c7/f8/bd5804695ba400e423c33fd4d9f58c28d86633d5ba1945c36ff3967d98cb/protobuf-7.36.0-cp310-abi3-manylinux2014_x86_64.whl", hash = "sha256:70f5ec8eb0da81a44360c0dc0beac99a0d78071d21956a7076bae8bd2051841b", size = 340439, upload-time = "2026-08-20T16:33:56.992Z" }, + { url = "https://files.pythonhosted.org/packages/ef/9f/acd02338235a3e7d03168c4303478347b7624fc8189ff4e7f0d2654bbe86/protobuf-7.36.0-cp310-abi3-win32.whl", hash = "sha256:7326fd717bdc419162a735938d89d4032332bcc3408804012b24ff3a37086071", size = 440216, upload-time = "2026-08-20T16:33:57.99Z" }, + { url = "https://files.pythonhosted.org/packages/0e/4e/12cb93270967a2affff5b3f720694700d4d87712a67afd05c8cb3f6fa52c/protobuf-7.36.0-cp310-abi3-win_amd64.whl", hash = "sha256:1781cc1de61249b750848029bca452c0a8b7e990080316b9bbc2518b2117b488", size = 453731, upload-time = "2026-08-20T16:33:58.951Z" }, + { url = "https://files.pythonhosted.org/packages/01/c3/629999e78d46c1115c11886d51c6bd68c17ce4a944f1ea3e153a91316a33/protobuf-7.36.0-py3-none-any.whl", hash = "sha256:53374d53fc29a67f7dbbf0ade47d7526a0f0137bf0f9c90e48d8a60790ef748c", size = 177024, upload-time = "2026-08-20T16:34:00.053Z" }, +] + +[[package]] +name = "psutil" +version = "7.2.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/aa/c6/d1ddf4abb55e93cebc4f2ed8b5d6dbad109ecb8d63748dd2b20ab5e57ebe/psutil-7.2.2.tar.gz", hash = "sha256:0746f5f8d406af344fd547f1c8daa5f5c33dbc293bb8d6a16d80b4bb88f59372", size = 493740, upload-time = "2026-01-28T18:14:54.428Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/51/08/510cbdb69c25a96f4ae523f733cdc963ae654904e8db864c07585ef99875/psutil-7.2.2-cp313-cp313t-macosx_10_13_x86_64.whl", hash = "sha256:2edccc433cbfa046b980b0df0171cd25bcaeb3a68fe9022db0979e7aa74a826b", size = 130595, upload-time = "2026-01-28T18:14:57.293Z" }, + { url = "https://files.pythonhosted.org/packages/d6/f5/97baea3fe7a5a9af7436301f85490905379b1c6f2dd51fe3ecf24b4c5fbf/psutil-7.2.2-cp313-cp313t-macosx_11_0_arm64.whl", hash = "sha256:e78c8603dcd9a04c7364f1a3e670cea95d51ee865e4efb3556a3a63adef958ea", size = 131082, upload-time = "2026-01-28T18:14:59.732Z" }, + { url = "https://files.pythonhosted.org/packages/37/d6/246513fbf9fa174af531f28412297dd05241d97a75911ac8febefa1a53c6/psutil-7.2.2-cp313-cp313t-manylinux2010_x86_64.manylinux_2_12_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1a571f2330c966c62aeda00dd24620425d4b0cc86881c89861fbc04549e5dc63", size = 181476, upload-time = "2026-01-28T18:15:01.884Z" }, + { url = "https://files.pythonhosted.org/packages/b8/b5/9182c9af3836cca61696dabe4fd1304e17bc56cb62f17439e1154f225dd3/psutil-7.2.2-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:917e891983ca3c1887b4ef36447b1e0873e70c933afc831c6b6da078ba474312", size = 184062, upload-time = "2026-01-28T18:15:04.436Z" }, + { url = "https://files.pythonhosted.org/packages/16/ba/0756dca669f5a9300d0cbcbfae9a4c30e446dfc7440ffe43ded5724bfd93/psutil-7.2.2-cp313-cp313t-win_amd64.whl", hash = "sha256:ab486563df44c17f5173621c7b198955bd6b613fb87c71c161f827d3fb149a9b", size = 139893, upload-time = "2026-01-28T18:15:06.378Z" }, + { url = "https://files.pythonhosted.org/packages/1c/61/8fa0e26f33623b49949346de05ec1ddaad02ed8ba64af45f40a147dbfa97/psutil-7.2.2-cp313-cp313t-win_arm64.whl", hash = "sha256:ae0aefdd8796a7737eccea863f80f81e468a1e4cf14d926bd9b6f5f2d5f90ca9", size = 135589, upload-time = "2026-01-28T18:15:08.03Z" }, + { url = "https://files.pythonhosted.org/packages/e7/36/5ee6e05c9bd427237b11b3937ad82bb8ad2752d72c6969314590dd0c2f6e/psutil-7.2.2-cp36-abi3-macosx_10_9_x86_64.whl", hash = "sha256:ed0cace939114f62738d808fdcecd4c869222507e266e574799e9c0faa17d486", size = 129090, upload-time = "2026-01-28T18:15:22.168Z" }, + { url = "https://files.pythonhosted.org/packages/80/c4/f5af4c1ca8c1eeb2e92ccca14ce8effdeec651d5ab6053c589b074eda6e1/psutil-7.2.2-cp36-abi3-macosx_11_0_arm64.whl", hash = "sha256:1a7b04c10f32cc88ab39cbf606e117fd74721c831c98a27dc04578deb0c16979", size = 129859, upload-time = "2026-01-28T18:15:23.795Z" }, + { url = "https://files.pythonhosted.org/packages/b5/70/5d8df3b09e25bce090399cf48e452d25c935ab72dad19406c77f4e828045/psutil-7.2.2-cp36-abi3-manylinux2010_x86_64.manylinux_2_12_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:076a2d2f923fd4821644f5ba89f059523da90dc9014e85f8e45a5774ca5bc6f9", size = 155560, upload-time = "2026-01-28T18:15:25.976Z" }, + { url = "https://files.pythonhosted.org/packages/63/65/37648c0c158dc222aba51c089eb3bdfa238e621674dc42d48706e639204f/psutil-7.2.2-cp36-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b0726cecd84f9474419d67252add4ac0cd9811b04d61123054b9fb6f57df6e9e", size = 156997, upload-time = "2026-01-28T18:15:27.794Z" }, + { url = "https://files.pythonhosted.org/packages/8e/13/125093eadae863ce03c6ffdbae9929430d116a246ef69866dad94da3bfbc/psutil-7.2.2-cp36-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:fd04ef36b4a6d599bbdb225dd1d3f51e00105f6d48a28f006da7f9822f2606d8", size = 148972, upload-time = "2026-01-28T18:15:29.342Z" }, + { url = "https://files.pythonhosted.org/packages/04/78/0acd37ca84ce3ddffaa92ef0f571e073faa6d8ff1f0559ab1272188ea2be/psutil-7.2.2-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc", size = 148266, upload-time = "2026-01-28T18:15:31.597Z" }, + { url = "https://files.pythonhosted.org/packages/b4/90/e2159492b5426be0c1fef7acba807a03511f97c5f86b3caeda6ad92351a7/psutil-7.2.2-cp37-abi3-win_amd64.whl", hash = "sha256:eb7e81434c8d223ec4a219b5fc1c47d0417b12be7ea866e24fb5ad6e84b3d988", size = 137737, upload-time = "2026-01-28T18:15:33.849Z" }, + { url = "https://files.pythonhosted.org/packages/8c/c7/7bb2e321574b10df20cbde462a94e2b71d05f9bbda251ef27d104668306a/psutil-7.2.2-cp37-abi3-win_arm64.whl", hash = "sha256:8c233660f575a5a89e6d4cb65d9f938126312bca76d8fe087b947b3a1aaac9ee", size = 134617, upload-time = "2026-01-28T18:15:36.514Z" }, +] + +[[package]] +name = "pymupdf" +version = "1.28.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/a3/fb/b6761fa2d5266f2cdb24c3b91f4023070ab7848381417678e7a289a1d52a/pymupdf-1.28.2.tar.gz", hash = "sha256:5e0be7908a715aa20333caddd73f1d6f01e4cd0c26e869fa2dd0b7f344da2249", size = 87903557, upload-time = "2026-08-06T21:43:23.321Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b4/51/550c9a75c4ff3245cb4ecb7bb95cbe2ab7374230b8e2b7a1f7259444150b/pymupdf-1.28.2-cp310-abi3-macosx_10_15_x86_64.whl", hash = "sha256:5fc315b425ff1f7afdd1ea2f348205cb19b806767daae7ce4d64115799c2bae1", size = 24645079, upload-time = "2026-08-06T21:37:25.001Z" }, + { url = "https://files.pythonhosted.org/packages/fa/01/3591f781b417b382a8487a2356e927acfe858b1043bab0ec47f6805bb109/pymupdf-1.28.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:7113846b35dbf0a033f088e4f4fb543dabeb4b0b12c112966a1ca1ee2d5eacae", size = 23875605, upload-time = "2026-08-06T21:37:40.369Z" }, + { url = "https://files.pythonhosted.org/packages/d2/86/4a68f080b71b46802178346af46486e1697508e760855ff5f3b218a6dff7/pymupdf-1.28.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:3050a233dde1211efe89ada74e2add6238436434159f46097a1423aad2842545", size = 25095554, upload-time = "2026-08-06T21:37:58.485Z" }, + { url = "https://files.pythonhosted.org/packages/c7/06/dace3e27af26690cb20bead80dbac42941b0841eb689b8aabbd67dde16f0/pymupdf-1.28.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:397d6715c1f0df7548a92d0afd8ce370fc48fa47aeefac16be2bc04a16a8227f", size = 25762500, upload-time = "2026-08-06T21:38:17.438Z" }, + { url = "https://files.pythonhosted.org/packages/e5/61/4146dfa1d8172a1ce8d59f0eed94896ddefb8deb2274534d0522fbb8abf5/pymupdf-1.28.2-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:f89fb2d86d07d643a269f17a093105057e20c79c1d06c103b53600067b6d2b01", size = 25986309, upload-time = "2026-08-06T21:38:35.472Z" }, + { url = "https://files.pythonhosted.org/packages/52/60/1fb6e64676f7500ebe89054b9e5bbbe14d3101c92d5f1a40ac9a35227673/pymupdf-1.28.2-cp310-abi3-win32.whl", hash = "sha256:530ef543a3885b3b81cb72a854e7c5a625a9233201221132bb6c31698c6a2bdb", size = 18525353, upload-time = "2026-08-06T21:38:47.697Z" }, + { url = "https://files.pythonhosted.org/packages/4a/61/d563bbccba262f9dd6d2d35ccb72593648184d886188efb12d9ce8f34dd6/pymupdf-1.28.2-cp310-abi3-win_amd64.whl", hash = "sha256:ebd244918798502d7b4504c90410d1711a4d7675a32584ca30f1bab419ecbffe", size = 19826532, upload-time = "2026-08-06T21:39:00.213Z" }, + { url = "https://files.pythonhosted.org/packages/e2/93/08f404a1f0155fe24137cf2d3aabd3e2b4b08c62053ed89c60f2611be3e9/pymupdf-1.28.2-cp310-abi3-win_arm64.whl", hash = "sha256:ffe91a24edc75c80da2a4b62f50fc0f54632d34fc8fe4cbc48e5c7ff07cf8fb4", size = 19759252, upload-time = "2026-08-06T21:39:12.937Z" }, + { url = "https://files.pythonhosted.org/packages/58/8c/d897dcd32a25b58186c968b15ce4324ca029e9d96460de12325314e390be/pymupdf-1.28.2-cp313-abi3-pyemscripten_2025_0_wasm32.whl", hash = "sha256:2e1b574c0fd2cb238021033fd3c0f9c4388816638df064e4bfb56d9d81736dc8", size = 18399403, upload-time = "2026-08-06T21:39:25.008Z" }, +] + +[[package]] +name = "pymupdf-layout" +version = "1.28.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "networkx" }, + { name = "numpy" }, + { name = "onnxruntime" }, + { name = "pymupdf" }, + { name = "pyyaml" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/e4/84/cf9030e29adbfa9fb555d925af1e0f212d54810e50711bd059e6e71c7cb1/pymupdf_layout-1.28.2-cp310-abi3-macosx_10_9_x86_64.whl", hash = "sha256:2a7368b5b0d75acb0835fae40a4d6e6c30ba457b388fedc75f06619a581ba17f", size = 42928696, upload-time = "2026-08-06T21:40:10.372Z" }, + { url = "https://files.pythonhosted.org/packages/16/1f/f03250cb18d4942d16f335d90a7eef2411b29097ba52531e0062edf16186/pymupdf_layout-1.28.2-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d55e9b9150e1e9f182063b93092f0ba2c2475b51ea6da656e9d6edad0dd4054d", size = 42927631, upload-time = "2026-08-06T21:40:38.474Z" }, + { url = "https://files.pythonhosted.org/packages/75/82/6cbf0331e148db48bf609c165dbe900cf3c1158546c5d09d4ad7fd4d6b17/pymupdf_layout-1.28.2-cp310-abi3-manylinux_2_28_aarch64.whl", hash = "sha256:fc7716682bfde26c002a7309cda0c520f3d2466dce368054d4fc2b653a74e8bc", size = 42937884, upload-time = "2026-08-06T21:41:03.992Z" }, + { url = "https://files.pythonhosted.org/packages/03/65/6b92d25678c64839fb2066ee98d6d1f164d820ba045d83c77e79021cda98/pymupdf_layout-1.28.2-cp310-abi3-manylinux_2_28_x86_64.whl", hash = "sha256:4b44a1d8ebf897b0e862ee2d73e7df73099f1c047fc024d2dd24ef0632d2cb5f", size = 42939266, upload-time = "2026-08-06T21:41:31.835Z" }, + { url = "https://files.pythonhosted.org/packages/db/a0/f429e4c398db6f88071e305434d67f8c6455498f4c7ebb372b730553f073/pymupdf_layout-1.28.2-cp310-abi3-win_amd64.whl", hash = "sha256:a765d3ba0b1622e8d5794772ad4303d824386f04b382a5d31c999a828ed4a089", size = 42938028, upload-time = "2026-08-06T21:41:59.15Z" }, +] + +[[package]] +name = "pymupdf4llm" +version = "1.28.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "psutil" }, + { name = "pymupdf" }, + { name = "pymupdf-layout" }, + { name = "tabulate" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/31/be/6f78a076947529c13bd0660cbe9e61f0155c903aa0c17fd9751733dccc36/pymupdf4llm-1.28.2.tar.gz", hash = "sha256:02681698ef67bda9a2acd5d9e1115d4e88be0eb9c5d68926e5e3cd98cd351874", size = 2091728, upload-time = "2026-08-06T21:43:27.654Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/7d/93/0ec4c33150f127d19b306d876b969755f02ed721f3a9337fd1f4fe4a1c85/pymupdf4llm-1.28.2-py3-none-any.whl", hash = "sha256:55c06c07d128f94c4d9271bd427d16ee219779e7d796a7b9da4e110be3004d96", size = 176403, upload-time = "2026-08-06T21:39:44.118Z" }, +] + +[[package]] +name = "pyyaml" +version = "6.0.3" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/05/8e/961c0007c59b8dd7729d542c61a4d537767a59645b82a0b521206e1e25c2/pyyaml-6.0.3.tar.gz", hash = "sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f", size = 130960, upload-time = "2025-09-25T21:33:16.546Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d1/11/0fd08f8192109f7169db964b5707a2f1e8b745d4e239b784a5a1dd80d1db/pyyaml-6.0.3-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8", size = 181669, upload-time = "2025-09-25T21:32:23.673Z" }, + { url = "https://files.pythonhosted.org/packages/b1/16/95309993f1d3748cd644e02e38b75d50cbc0d9561d21f390a76242ce073f/pyyaml-6.0.3-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1", size = 173252, upload-time = "2025-09-25T21:32:25.149Z" }, + { url = "https://files.pythonhosted.org/packages/50/31/b20f376d3f810b9b2371e72ef5adb33879b25edb7a6d072cb7ca0c486398/pyyaml-6.0.3-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c", size = 767081, upload-time = "2025-09-25T21:32:26.575Z" }, + { url = "https://files.pythonhosted.org/packages/49/1e/a55ca81e949270d5d4432fbbd19dfea5321eda7c41a849d443dc92fd1ff7/pyyaml-6.0.3-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5", size = 841159, upload-time = "2025-09-25T21:32:27.727Z" }, + { url = "https://files.pythonhosted.org/packages/74/27/e5b8f34d02d9995b80abcef563ea1f8b56d20134d8f4e5e81733b1feceb2/pyyaml-6.0.3-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6", size = 801626, upload-time = "2025-09-25T21:32:28.878Z" }, + { url = "https://files.pythonhosted.org/packages/f9/11/ba845c23988798f40e52ba45f34849aa8a1f2d4af4b798588010792ebad6/pyyaml-6.0.3-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6", size = 753613, upload-time = "2025-09-25T21:32:30.178Z" }, + { url = "https://files.pythonhosted.org/packages/3d/e0/7966e1a7bfc0a45bf0a7fb6b98ea03fc9b8d84fa7f2229e9659680b69ee3/pyyaml-6.0.3-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be", size = 794115, upload-time = "2025-09-25T21:32:31.353Z" }, + { url = "https://files.pythonhosted.org/packages/de/94/980b50a6531b3019e45ddeada0626d45fa85cbe22300844a7983285bed3b/pyyaml-6.0.3-cp313-cp313-win32.whl", hash = "sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26", size = 137427, upload-time = "2025-09-25T21:32:32.58Z" }, + { url = "https://files.pythonhosted.org/packages/97/c9/39d5b874e8b28845e4ec2202b5da735d0199dbe5b8fb85f91398814a9a46/pyyaml-6.0.3-cp313-cp313-win_amd64.whl", hash = "sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c", size = 154090, upload-time = "2025-09-25T21:32:33.659Z" }, + { url = "https://files.pythonhosted.org/packages/73/e8/2bdf3ca2090f68bb3d75b44da7bbc71843b19c9f2b9cb9b0f4ab7a5a4329/pyyaml-6.0.3-cp313-cp313-win_arm64.whl", hash = "sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb", size = 140246, upload-time = "2025-09-25T21:32:34.663Z" }, +] + +[[package]] +name = "ruff" +version = "0.16.4" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", hash = "sha256:13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc", size = 4899731, upload-time = "2026-08-20T17:43:59.196Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", hash = "sha256:df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7", size = 10006909, upload-time = "2026-08-20T17:43:16.888Z" }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604", size = 10240201, upload-time = "2026-08-20T17:43:19.337Z" }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255", size = 9835122, upload-time = "2026-08-20T17:43:21.708Z" }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade", size = 9977162, upload-time = "2026-08-20T17:43:24.236Z" }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff", size = 9829789, upload-time = "2026-08-20T17:43:26.966Z" }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80", size = 10527949, upload-time = "2026-08-20T17:43:29.384Z" }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07", size = 11333695, upload-time = "2026-08-20T17:43:31.872Z" }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7", size = 10727741, upload-time = "2026-08-20T17:43:34.596Z" }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3", size = 10286522, upload-time = "2026-08-20T17:43:37.288Z" }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454", size = 10584182, upload-time = "2026-08-20T17:43:39.984Z" }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b", size = 10134195, upload-time = "2026-08-20T17:43:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d", size = 9825821, upload-time = "2026-08-20T17:43:44.532Z" }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", hash = "sha256:8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0", size = 10267658, upload-time = "2026-08-20T17:43:46.989Z" }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d", size = 10697071, upload-time = "2026-08-20T17:43:49.891Z" }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", hash = "sha256:312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e", size = 10021687, upload-time = "2026-08-20T17:43:52.281Z" }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", hash = "sha256:05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c", size = 10567657, upload-time = "2026-08-20T17:43:54.78Z" }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", hash = "sha256:a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21", size = 10451579, upload-time = "2026-08-20T17:43:57.135Z" }, +] + +[[package]] +name = "tabulate" +version = "0.10.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/46/58/8c37dea7bbf769b20d58e7ace7e5edfe65b849442b00ffcdd56be88697c6/tabulate-0.10.0.tar.gz", hash = "sha256:e2cfde8f79420f6deeffdeda9aaec3b6bc5abce947655d17ac662b126e48a60d", size = 91754, upload-time = "2026-03-04T18:55:34.402Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/99/55/db07de81b5c630da5cbf5c7df646580ca26dfaefa593667fc6f2fe016d2e/tabulate-0.10.0-py3-none-any.whl", hash = "sha256:f0b0622e567335c8fabaaa659f1b33bcb6ddfe2e496071b743aa113f8774f2d3", size = 39814, upload-time = "2026-03-04T18:55:31.284Z" }, +]