diff --git a/.github/workflows/markdown-export.yml b/.github/workflows/markdown-export.yml
new file mode 100644
index 00000000..cfc0c742
--- /dev/null
+++ b/.github/workflows/markdown-export.yml
@@ -0,0 +1,221 @@
+name: Markdown Export
+
+on:
+ pull_request:
+ push:
+ branches: [develop]
+ tags: ['FieldWorks*']
+ workflow_dispatch:
+ inputs:
+ dry_run:
+ description: 'Build and report, but do not publish'
+ type: boolean
+ default: false
+
+concurrency:
+ group: markdown-export
+ cancel-in-progress: false
+
+permissions:
+ contents: read
+
+env:
+ PANDOC_VERSION: '3.9.0.2'
+ PANDOC_SHA256: ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567
+ EXPORT_BRANCH: markdown-export
+
+jobs:
+ validate:
+ name: Validate markdown export
+ runs-on: ubuntu-latest
+ permissions:
+ contents: read
+ steps:
+ # The runner context is not available to job-level env, so the build
+ # paths are resolved here from $RUNNER_TEMP for every later step.
+ - name: Resolve build paths
+ run: |
+ set -euo pipefail
+ {
+ echo "EXPORT_DIR=${RUNNER_TEMP}/markdown-export"
+ echo "WORK_DIR=${RUNNER_TEMP}/markdown-export-work"
+ echo "DIAGNOSTICS=${RUNNER_TEMP}/markdown-export-diagnostics.json"
+ } >> "$GITHUB_ENV"
+
+ - name: Checkout source
+ uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09
+ with:
+ path: src
+ fetch-depth: 1
+
+ - name: Install Pandoc and CHM extractor
+ run: |
+ set -euo pipefail
+ curl -fsSL -o /tmp/pandoc.deb \
+ "https://github.com/jgm/pandoc/releases/download/${PANDOC_VERSION}/pandoc-${PANDOC_VERSION}-1-amd64.deb"
+ echo "${PANDOC_SHA256} /tmp/pandoc.deb" | sha256sum --check --strict
+ sudo dpkg -i /tmp/pandoc.deb
+ # p7zip-full remains a runner-image package dependency for CHM extraction.
+ sudo apt-get update -qq
+ sudo apt-get install -y -qq p7zip-full
+ pandoc --version | head -2
+ 7z i > /dev/null && echo "7z ok"
+
+ - name: Install uv
+ uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e
+ with:
+ version: '0.9.4'
+
+ - name: Provision locked Python and dependencies
+ working-directory: src
+ run: |
+ set -euo pipefail
+ uv python install
+ uv lock --check
+ uv sync --frozen
+
+ - name: Lint converters
+ working-directory: src
+ run: uv run --frozen ruff check tools
+
+ - name: Test converters
+ working-directory: src
+ run: uv run --frozen python -m unittest discover -s tools -p 'test_*.py' -v
+
+ - name: Convert
+ run: |
+ set -euo pipefail
+ rm -rf -- "${EXPORT_DIR}" "${WORK_DIR}"
+ # --directory already puts uv in the checkout, so --repo is relative to it.
+ uv run --frozen --directory src python tools/convert.py \
+ --repo . \
+ --out "${EXPORT_DIR}" \
+ --work "${WORK_DIR}" \
+ --diagnostics "${DIAGNOSTICS}" \
+ --source-ref "${GITHUB_SHA::7}"
+
+ - name: Summarise quality report
+ if: always()
+ run: |
+ uv run --frozen --directory src python -c '
+ import json, os
+ from pathlib import Path
+ diagnostics = Path(os.environ["DIAGNOSTICS"])
+ if not diagnostics.exists():
+ print("## Markdown Export\n\nNo diagnostics produced — validation failed early.")
+ raise SystemExit(0)
+ r = json.loads(diagnostics.read_text(encoding="utf-8"))
+ corpus = r.get("corpus", {})
+ summary = r.get("summary", {})
+ source_ref = corpus.get("source_ref", "?")
+ topic_count = corpus.get("topic_count", 0)
+ pdf_count = corpus.get("pdf_count", 0)
+ fatal_count = summary.get("fatal", 0)
+ advisory_count = summary.get("advisory", 0)
+ print(f"## Markdown Export — ref {source_ref}\n")
+ print(f"**{topic_count:,} topics and {pdf_count:,} PDFs converted**\n")
+ print("| Check | Count |\n| --- | ---: |")
+ print(f"| fatal | {fatal_count} |")
+ print(f"| advisory | {advisory_count} |")
+ for code, count in sorted(summary.get("by_code", {}).items()):
+ label = code.replace("_", " ")
+ print(f"| {label} | {count} |")
+ for code in ("source_missing_link", "source_missing_image", "missing_link", "missing_image", "not_in_toc"):
+ items = [item for item in r.get("issues", []) if item.get("code") == code]
+ if items:
+ label = items[0].get("label", code.replace("_", " "))
+ print(f"\n{label} ({len(items)})\n")
+ for item in items[:50]:
+ item_label = item.get("label", label)
+ item_path = item.get("path", "")
+ item_message = item.get("message", "")
+ print(f"- **{item_label}** `{item_path}`: {item_message}")
+ print("\n")
+ ' >> "$GITHUB_STEP_SUMMARY"
+
+ - name: Upload diagnostics
+ if: always()
+ uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02
+ with:
+ name: markdown-export-diagnostics
+ path: ${{ runner.temp }}/markdown-export-diagnostics.json
+ if-no-files-found: ignore
+ retention-days: 14
+
+ - name: Upload completed export
+ if: success()
+ uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02
+ with:
+ name: markdown-export
+ path: ${{ runner.temp }}/markdown-export
+ if-no-files-found: error
+ include-hidden-files: true
+ retention-days: 14
+
+ publish:
+ name: Publish markdown export
+ needs: validate
+ if: >-
+ needs.validate.result == 'success' &&
+ (github.event_name == 'push' ||
+ (github.event_name == 'workflow_dispatch' && !inputs.dry_run))
+ runs-on: ubuntu-latest
+ permissions:
+ contents: write
+ steps:
+ - name: Download completed export
+ uses: actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0
+ with:
+ name: markdown-export
+ path: ${{ runner.temp }}/markdown-export
+
+ - name: Publish incremental markdown-export history
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ EXPORT_DIR: ${{ runner.temp }}/markdown-export
+ PUBLISH_DIR: ${{ runner.temp }}/markdown-export-publish
+ run: |
+ set -euo pipefail
+ rm -rf -- "${PUBLISH_DIR}"
+ mkdir -p "${PUBLISH_DIR}"
+ git -C "${PUBLISH_DIR}" init -q -b "${EXPORT_BRANCH}"
+ git -C "${PUBLISH_DIR}" config user.name "github-actions[bot]"
+ git -C "${PUBLISH_DIR}" config user.email "41898282+github-actions[bot]@users.noreply.github.com"
+ git -C "${PUBLISH_DIR}" remote add origin \
+ "https://x-access-token:${GH_TOKEN}@github.com/${GITHUB_REPOSITORY}.git"
+ if git -C "${PUBLISH_DIR}" fetch --no-tags origin \
+ "refs/heads/${EXPORT_BRANCH}:refs/remotes/origin/${EXPORT_BRANCH}"; then
+ git -C "${PUBLISH_DIR}" checkout -q -B "${EXPORT_BRANCH}" \
+ "refs/remotes/origin/${EXPORT_BRANCH}"
+ else
+ git -C "${PUBLISH_DIR}" checkout -q -B "${EXPORT_BRANCH}"
+ fi
+ git -C "${PUBLISH_DIR}" rm -r -q --ignore-unmatch -- .
+ git -C "${PUBLISH_DIR}" clean -fdx -q
+ cp -a "${EXPORT_DIR}/." "${PUBLISH_DIR}/"
+ git -C "${PUBLISH_DIR}" add -A
+ if git -C "${PUBLISH_DIR}" diff --cached --quiet; then
+ echo "${EXPORT_BRANCH} is already up to date"
+ else
+ git -C "${PUBLISH_DIR}" commit -q -m "Markdown export of help and PDFs ${GITHUB_SHA::7}"
+ git -C "${PUBLISH_DIR}" push origin "HEAD:${EXPORT_BRANCH}"
+ echo "published to ${EXPORT_BRANCH}"
+ fi
+ echo "PUBLISH_DIR=${PUBLISH_DIR}" >> "$GITHUB_ENV"
+
+ - name: Tag the export to match the release
+ if: startsWith(github.ref, 'refs/tags/FieldWorks') &&
+ (github.event_name != 'workflow_dispatch' || !inputs.dry_run)
+ env:
+ GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+ run: |
+ set -euo pipefail
+ cd "${PUBLISH_DIR}"
+ TAG="${EXPORT_BRANCH}/${GITHUB_REF_NAME}"
+ if git ls-remote --exit-code --tags origin "refs/tags/${TAG}" > /dev/null 2>&1; then
+ echo "tag already exists: ${TAG}"
+ else
+ git tag -a "${TAG}" -m "Markdown export for ${GITHUB_REF_NAME}"
+ git push origin "refs/tags/${TAG}"
+ echo "tagged ${TAG}"
+ fi
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 00000000..ab990338
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,5 @@
+.chm-work/
+*.pyc
+__pycache__/
+tools/out/
+.fwhelps-export-*.lock
diff --git a/.python-version b/.python-version
new file mode 100644
index 00000000..86f8c02e
--- /dev/null
+++ b/.python-version
@@ -0,0 +1 @@
+3.13.5
diff --git a/README.md b/README.md
new file mode 100644
index 00000000..09dee37f
--- /dev/null
+++ b/README.md
@@ -0,0 +1,131 @@
+# FwHelps
+
+Documentation for [FieldWorks Language Explorer](https://software.sil.org/fieldworks/)
+(FLEx). Help content is authored in Adobe RoboHelp and committed here as a
+compiled CHM, alongside training and technical-note PDFs.
+
+| File | Contents |
+| --- | --- |
+| `FieldWorks_Language_Explorer_Help.chm` | The main help system — 1,599 topics |
+| `Language Explorer/Training/` | Technical notes: Send-Receive, imports, Word export |
+| `Language Explorer/Utilities/` | AlloGen, PcPatr, ToneParsFLEx, VarGen documentation |
+| `WW-ConceptualIntro/` | Conceptual Introduction to FLEx |
+
+The FieldWorks installer consumes this repo directly: `patch-installer-cd.yml`
+in [sillsdev/FieldWorks](https://github.com/sillsdev/FieldWorks) checks it out
+via a `helps_ref` input, and `Build/releaseTagger.py` tags it `FieldWorks`
+at release time.
+
+## Markdown export
+
+The CHM is also published as markdown on the
+[**`markdown-export`**](../../tree/markdown-export) branch — one file per help
+topic, with images, YAML frontmatter, and a full table of contents.
+
+It exists for two reasons:
+
+- **AI retrieval.** The FieldWorks AI bot previously ingested raw RoboHelp
+ HTML, where roughly two thirds of every topic is markup rather than
+ documentation. The markdown corpus is about 65% smaller in tokens
+ (~2.14M → ~759K) with the prose intact.
+- **Reviewable diffs.** A help change is otherwise a 5 MB opaque binary.
+ On the export branch it is a readable text diff, one changed file per
+ edited topic.
+
+Built automatically by
+[`markdown-export.yml`](.github/workflows/markdown-export.yml) on every push to
+`develop`. Nothing there is hand-edited — edit the help in RoboHelp and commit
+the CHM.
+
+Each build also produces `author-report.md` for RoboHelp/PDF authors and
+`author-report.json` for automation. Both cover broken links, topics missing
+from the table of contents, and other source/export quality findings. The
+Markdown report preserves the exact source path and evidence for every finding
+and gives issue-specific repair guidance; JSON retains the stable
+`{corpus, summary, issues}` schema.
+
+Exporter mutation boundaries use a non-blocking native advisory lock on a
+deterministic sibling lockfile, so cooperating invocations targeting the same
+destination fail clearly instead of racing. The lock is a coordination aid,
+not protection against a local process that deliberately ignores file locks;
+stale lockfiles are harmless because ownership is held by the OS handle.
+
+### Versions
+
+Each FieldWorks release is tagged here by `releaseTagger.py`; the matching
+export is tagged `markdown-export/`.
+
+| FieldWorks | Released | Markdown export |
+| --- | --- | --- |
+| 9.3.7-beta | 2026-02-25 | [`markdown-export/FieldWorks9.3.7-beta`](../../tree/markdown-export/FieldWorks9.3.7-beta) |
+| 9.3.6-beta | 2026-01-29 | [`markdown-export/FieldWorks9.3.6-beta`](../../tree/markdown-export/FieldWorks9.3.6-beta) |
+| 9.3.4 | 2025-10-30 | [`markdown-export/FieldWorks9.3.4`](../../tree/markdown-export/FieldWorks9.3.4) |
+| 9.3.1 | 2025-07-25 | [`markdown-export/FieldWorks9.3.1`](../../tree/markdown-export/FieldWorks9.3.1) |
+| 9.3.0 | 2025-06-17 | [`markdown-export/FieldWorks9.3.0`](../../tree/markdown-export/FieldWorks9.3.0) |
+
+> [!NOTE]
+> Export tags are created going forward, as each release is tagged. Rows above
+> that have no corresponding export tag yet can be backfilled by re-running the
+> workflow against that tag.
+
+## Tools
+
+| Script | Purpose |
+| --- | --- |
+| [`tools/convert.py`](tools/convert.py) | CHM → markdown corpus (the build) |
+| [`tools/fwhelp.lua`](tools/fwhelp.lua) | Pandoc filter: RoboHelp semantics → clean GFM |
+| [`tools/chm_extract.py`](tools/chm_extract.py) | Cross-platform CHM extraction, with validation |
+| [`tools/pdf_convert.py`](tools/pdf_convert.py) | PDF → markdown (bookmarks or font inference) |
+| [`tools/pdf_outlines.json`](tools/pdf_outlines.json) | Pinned PDF outlines; drift fails the build |
+| [`tools/survey.py`](tools/survey.py) | Read-only census of the corpus |
+
+Local build (needs `pandoc` 3.x, and `7z` or Windows' built-in `hh.exe`):
+
+```sh
+uv run --frozen tools/convert.py --repo . --out export
+```
+
+### Reproducible local setup
+
+The converter uses uv with the exact Python version in
+[`.python-version`](.python-version). `uv.lock` is the authoritative dependency
+file; `requirements.txt` and `requirements-dev.txt` are deterministic
+lock-derived compatibility exports for older tooling and should not be edited
+independently. The CI workflow installs only from the frozen lock:
+
+```sh
+uv python install
+uv lock --check
+uv sync --frozen
+uv run --frozen \
+ -m unittest discover -s tools -p 'test_*.py' -v
+uv run --frozen ruff check tools
+uv run --frozen \
+ tools/convert.py --repo . --out export
+uv run --frozen \
+ tools/pdf_convert.py --repo . --out export --update-outlines
+```
+
+To regenerate the compatibility exports after changing project dependencies:
+
+```sh
+uv export --frozen --no-dev --no-hashes --format requirements.txt \
+ --output-file requirements.txt
+uv export --frozen --only-group dev --no-hashes --format requirements.txt \
+ --output-file requirements-dev.txt
+```
+
+On PowerShell, the same `uv run --frozen` commands work unchanged. Updating
+outline locks is intentional and should be reviewed with the resulting
+`tools/pdf_outlines.json` change.
+
+The workflow still installs the runner-image package `p7zip-full` for CHM
+extraction; it is intentionally outside the Python lock. Pandoc is downloaded
+as the pinned 3.9.0.2 amd64 package and its SHA-256 is checked before install.
+
+> [!IMPORTANT]
+> `hh.exe -decompile` silently truncates filenames when the output path exceeds
+> Windows' 260-character limit — no error, no non-zero exit. `chm_extract.py`
+> refuses to run it into a too-long path and validates every extraction against
+> the CHM's own table of contents. Use a short `--work` directory on Windows,
+> or install 7-Zip.
diff --git a/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md b/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md
new file mode 100644
index 00000000..47caf5df
--- /dev/null
+++ b/docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md
@@ -0,0 +1,349 @@
+# Portable Markdown Export Final Hardening Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Close every confirmed safety, correctness, portability, reporting, and CI gap from the final cross-cutting review without changing valid current-corpus content.
+
+**Architecture:** Harden source discovery and extraction at their input boundaries, centralize issue policy in one module, and split CI validation from privileged publication. Converter-owned manifests authenticate reusable or removable state; validation treats ambiguous or unsafe output as fatal.
+
+**Tech Stack:** Python 3.13.5, uv/`uv.lock`, `unittest`, Pandoc 3.9.0.2, Lua filters, GitHub Actions, PowerShell/Linux shell verification.
+
+---
+
+## File responsibilities
+
+- `tools/source_safety.py`: repository-containment and non-symlink source discovery helpers shared by CHM and PDF tracks.
+- `tools/frontmatter.py`: shared JSON-compatible YAML scalar serialization for CHM and PDF metadata.
+- `tools/issue_catalog.py`: canonical issue code, label, severity, and provenance policy.
+- `tools/output_fs.py`: staged promotion plus cooperative cross-process destination locking.
+- `tools/chm_extract.py`: isolated extraction, parsed reference validation, and destructive-target rejection.
+- `tools/chm_convert.py`: authenticated extraction reuse, destination collision checks, safe frontmatter, and unsafe-link reporting.
+- `tools/pdf_convert.py`: safe PDF discovery and authenticated converter-manifest cleanup.
+- `tools/corpus_validation.py`: emitted-corpus path and URI policy.
+- `tools/fwhelp.lua`: AST transformations that inventory all authored classes and neutralize unsafe URI schemes.
+- `tools/convert.py`: thin orchestration and diagnostic-report persistence.
+- `.github/workflows/markdown-export.yml`: read-only validation job and separately privileged publication job.
+- `pyproject.toml`, `uv.lock`, `.python-version`: exact Python and complete dependency lock.
+
+### Task 1: Source identity, symlinks, and extraction safety
+
+**Files:**
+- Create: `tools/source_safety.py`
+- Create: `tools/test_source_safety.py`
+- Modify: `tools/chm_extract.py`
+- Modify: `tools/chm_convert.py`
+- Modify: `tools/pdf_convert.py`
+- Test: `tools/test_chm_extract.py`
+- Test: `tools/test_convert.py`
+- Test: `tools/test_pdf_convert.py`
+
+- [ ] **Step 1: Write failing source-boundary tests**
+
+Add tests with this behavior:
+
+```python
+def test_source_file_rejects_symlink_outside_repository(self):
+ outside = self.root.parent / "outside.pdf"
+ outside.write_bytes(b"pdf")
+ link = self.root / "linked.pdf"
+ link.symlink_to(outside)
+ with self.assertRaises(SourceSafetyError):
+ discover_source_files(self.root, suffixes={".pdf"}, recursive=True)
+
+def test_extract_rejects_source_directory_as_destination(self):
+ chm = self.root / "Help.chm"
+ chm.write_bytes(b"chm")
+ with self.assertRaises(OutputPathError):
+ extract(chm, self.root)
+```
+
+- [ ] **Step 2: Run the focused tests and verify RED**
+
+Run:
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+```
+
+Expected: failures for missing `SourceSafetyError`, symlink acceptance, source-directory extraction, and stale reuse.
+
+- [ ] **Step 3: Implement shared source discovery and authenticated reuse**
+
+Create a helper with this contract:
+
+```python
+class SourceSafetyError(ValueError):
+ pass
+
+def discover_source_files(
+ root: Path, *, suffixes: set[str], recursive: bool
+) -> list[Path]:
+ """Return stable, regular, non-symlink files resolving beneath root."""
+```
+
+Write `.chm-extraction-manifest.json` only after validated extraction promotion:
+
+```json
+{
+ "schema": 1,
+ "source_name": "FieldWorks_Language_Explorer_Help.chm",
+ "source_sha256": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
+}
+```
+
+Reuse only when schema, name, and hash match; otherwise perform fresh isolated extraction. Reject `outdir.resolve() == chm.parent.resolve()` and any destination that contains the CHM path.
+
+- [ ] **Step 4: Verify GREEN**
+
+Run the focused command from Step 2. Expected: all focused tests pass.
+
+### Task 2: CHM parsing, collisions, frontmatter, and URI safety
+
+**Files:**
+- Modify: `tools/chm_extract.py`
+- Modify: `tools/chm_convert.py`
+- Modify: `tools/chm_metadata.py`
+- Modify: `tools/corpus_validation.py`
+- Modify: `tools/fwhelp.lua`
+- Test: `tools/test_chm_extract.py`
+- Test: `tools/test_chm_metadata.py`
+- Test: `tools/test_convert.py`
+- Test: `tools/test_corpus_validation.py`
+
+- [ ] **Step 1: Write failing conversion-boundary tests**
+
+Cover these exact cases:
+
+```python
+def test_absolute_local_target_is_fatal(self):
+ (self.root / "topic.md").write_text("# Topic\n\n[bad](/outside.md)\n", encoding="utf-8")
+ issues = validate_corpus(self.root)
+ self.assertTrue(any(i.code == "path_escape" and i.fatal for i in issues))
+
+def test_mixed_known_and_unknown_span_class_reports_unknown(self):
+ markdown, unmapped = run_pandoc('text')
+ self.assertIn("NewSemantic", unmapped)
+
+def test_casefolded_image_collision_is_fatal(self):
+ # Icon.png and icon.PNG must claim one normalized destination.
+ self.assertIn("destination_collisions", result["report"])
+```
+
+Also add HHC fixtures using single quotes, reversed `value`/`name` attributes, escaped paths, `.woff`, `.webp`, `.json`, and `.map`; YAML values with newlines/control characters; and `javascript:`, `file:`, `data:`, `https:`, and `mailto:` links.
+
+- [ ] **Step 2: Verify RED**
+
+Run:
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+```
+
+Expected: failures for each new boundary behavior.
+
+- [ ] **Step 3: Implement parsed references and allowlisted output**
+
+Use `html.parser.HTMLParser` to collect `param` elements whose case-folded `name` is `local`, independent of attribute ordering or quote style. Validate local targets referenced by TOC and HTML `href`/`src`; permit unrelated asset extensions. Detect truncation only when an expected target is missing and a prefix sibling exists.
+
+Build a shared case-folded claim table for every topic and image destination before writing. Inventory every span class before selecting the first supported transformation. Permit only `http`, `https`, and `mailto` external schemes; fragments and relative paths remain local. Neutralize source-authored unsafe targets and report them with source provenance; emitted unsafe targets remain fatal.
+
+Serialize frontmatter scalars through one JSON-compatible YAML quoting function:
+
+```python
+def yaml_scalar(value: str) -> str:
+ return json.dumps(value, ensure_ascii=False)
+```
+
+JSON quoting escapes newlines and C0 controls, preventing YAML injection while preserving adversarial source metadata as data rather than rejecting an otherwise convertible help file.
+
+- [ ] **Step 4: Verify GREEN**
+
+Run the focused command from Step 2. Expected: all focused tests pass and ordinary images/links remain unchanged.
+
+### Task 3: PDF manifest ownership and canonical issue policy
+
+**Files:**
+- Create: `tools/issue_catalog.py`
+- Create: `tools/test_issue_catalog.py`
+- Modify: `tools/pdf_convert.py`
+- Modify: `tools/reporting.py`
+- Modify: `tools/convert.py`
+- Modify: `tools/corpus_validation.py`
+- Test: `tools/test_pdf_convert.py`
+- Test: `tools/test_reporting.py`
+- Test: `tools/test_convert.py`
+
+- [ ] **Step 1: Write failing manifest and catalog tests**
+
+```python
+def test_corrupt_manifest_cannot_delete_unrelated_output(self):
+ unrelated = self.out / "keep.md"
+ unrelated.write_text("keep", encoding="utf-8")
+ previous = {"source.pdf": {"markdown": "keep.md", "images": None}}
+ with self.assertRaises(ManifestError):
+ promote_pdf_outputs(self.out, self.stage, previous, {})
+ self.assertEqual(unrelated.read_text(encoding="utf-8"), "keep")
+
+def test_every_emitted_issue_code_has_one_policy(self):
+ for code in emitted_issue_codes():
+ self.assertIn(code, ISSUE_CATALOG)
+```
+
+- [ ] **Step 2: Verify RED**
+
+Run:
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+```
+
+Expected: corrupt-manifest and missing-central-policy failures.
+
+- [ ] **Step 3: Authenticate removable PDF paths and centralize policy**
+
+Version the PDF manifest and record source plus expected derived destinations:
+
+```json
+{
+ "schema": 2,
+ "files": {
+ "Language Explorer/Training/source.pdf": {
+ "markdown": "Language_Explorer/Training/source.md",
+ "images": "Language_Explorer/Training/source_images"
+ }
+ }
+}
+```
+
+Before backup or deletion, recompute `slug_path(source)` and require exact equality with both manifest destinations. Reject schema 1 or corrupt entries before mutation; the next successful run may replace a valid schema-2 set.
+
+Define `IssuePolicy(label, fatal, provenance)` once. Make report creation use `make_issue(code, ...)`; unknown codes at integration boundaries become fatal exporter issues. Mark retained PDF HTML as source provenance.
+
+- [ ] **Step 4: Verify GREEN**
+
+Run the focused command from Step 2. Expected: all tests pass, including rollback tests.
+
+### Task 4: Reproducible, observable, least-privilege CI
+
+**Files:**
+- Create: `pyproject.toml`
+- Create: `uv.lock`
+- Modify: `.python-version`
+- Modify: `requirements.txt`
+- Modify: `requirements-dev.txt`
+- Modify: `.github/workflows/markdown-export.yml`
+- Modify: `tools/convert.py`
+- Modify: `README.md`
+- Test: `tools/test_workflow.py`
+- Test: `tools/test_convert.py`
+
+- [ ] **Step 1: Write failing workflow and diagnostics tests**
+
+Assert the workflow contains:
+
+```python
+self.assertIn("pull_request:", workflow)
+self.assertIn("permissions:\n contents: read", workflow)
+self.assertIn("uv sync --frozen", workflow)
+self.assertIn("ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567", workflow)
+self.assertNotRegex(workflow, r"uses:\s+[^\s]+@v\d+")
+self.assertRegex(workflow, r"publish:\s*[\s\S]+permissions:\s*\n\s+contents: write")
+```
+
+Add a converter test proving `--diagnostics audit-diagnostics.json` writes canonical JSON even when corpus promotion is rejected for a fatal issue.
+
+- [ ] **Step 2: Verify RED**
+
+Run:
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+```
+
+Expected: failures for missing PR trigger, mutable actions, missing digest/frozen lock, broad write permission, and absent failure diagnostics.
+
+- [ ] **Step 3: Lock dependencies and split workflow privileges**
+
+Set `.python-version` to `3.13.5`. Declare runtime dependencies and Ruff in `pyproject.toml`, then run:
+
+```powershell
+uv lock --python 3.13.5
+uv sync --frozen
+```
+
+Keep `requirements.txt` and `requirements-dev.txt` as compatibility exports generated from the lock and document that `uv.lock` is authoritative.
+
+Use these immutable pins:
+
+```yaml
+actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09
+astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e
+actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02
+actions/download-artifact@634f93cb2916e3fdff6788551b99b062d0335ce0
+```
+
+Verify Pandoc before installation:
+
+```bash
+echo "ce4ac48f48aa7eadc1f5dbdf3449a1739f188ecb8c5421c5adc070fe7479e567 /tmp/pandoc.deb" | sha256sum --check --strict
+```
+
+The `validate` job uses `contents: read`, runs on `pull_request`, trusted pushes/tags, and dispatch, writes diagnostics, and always uploads them. A separate `publish` job has `contents: write`, downloads the successful export artifact, and runs only for trusted push/tag or non-dry-run dispatch events. Document `p7zip-full` as the remaining runner-image package dependency.
+
+- [ ] **Step 4: Verify GREEN**
+
+Run:
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+.review-venv\Scripts\ruff.exe check tools
+uv lock --check
+uv sync --frozen
+```
+
+Expected: all commands exit zero.
+
+### Task 5: Full verification and independent review
+
+**Files:**
+- Modify only files required by verified review findings.
+
+- [ ] **Step 1: Run the complete local gates**
+
+```powershell
+.review-venv\Scripts\python.exe -m unittest discover -s tools -p 'test_*.py'
+.review-venv\Scripts\ruff.exe check tools
+git diff --check
+uv lock --check
+```
+
+Expected: all tests pass, Ruff reports `All checks passed!`, and both remaining commands exit zero.
+
+- [ ] **Step 2: Run a real corpus export**
+
+Use fresh verified temporary directories and run:
+
+```powershell
+$sourceRef = git rev-parse --short HEAD
+.review-venv\Scripts\python.exe tools\convert.py --repo . --out $auditOut --work $auditWork --source-ref $sourceRef --diagnostics $auditDiagnostics
+```
+
+Expected: exit zero; report has two CHMs, 1,630 topics, 13 PDFs, and zero fatal issues.
+
+- [ ] **Step 3: Dispatch independent Luna spec and quality reviews**
+
+Review `git diff 1ecb705...HEAD` against the final-hardening design. Resolve every Critical or Important finding and repeat the relevant test/review gate.
+
+- [ ] **Step 4: Commit and push one implementation commit**
+
+```powershell
+git add -- .github/workflows/markdown-export.yml .python-version README.md requirements.txt requirements-dev.txt pyproject.toml uv.lock tools docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md docs/superpowers/plans/2026-08-21-portable-markdown-export-final-hardening.md
+git commit -m "Close final Markdown export hardening gaps"
+git push fork tools/markdown-export
+```
+
+Expected: ordinary fast-forward push; PR #3 updates without force.
+
+- [ ] **Step 5: Regenerate and publish Markdown incrementally**
+
+Generate from the committed SHA, update the existing `johnml1135/FwHelps:markdown-export` tree using tracked-file deletion plus ordinary commit/push, and verify the remote report has zero fatal issues and 2,356 files unless intentional output changes alter that count.
diff --git a/docs/superpowers/plans/2026-08-21-portable-markdown-export.md b/docs/superpowers/plans/2026-08-21-portable-markdown-export.md
new file mode 100644
index 00000000..01ad41a5
--- /dev/null
+++ b/docs/superpowers/plans/2026-08-21-portable-markdown-export.md
@@ -0,0 +1,81 @@
+# Portable Markdown Export Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Harden the CHM/PDF exporter into a safe, reproducible, portable converter with clean and validated generated Markdown.
+
+**Architecture:** Repository policy remains in a thin orchestration module while extraction, PDF conversion, output staging, and corpus validation expose small reusable interfaces. Every build is staged and validated before replacement, and CI publishes an ordinary incremental branch history.
+
+**Tech Stack:** Python 3.13 managed by uv, Pandoc 3.9.0.2 with Lua, PyMuPDF, pymupdf4llm, unittest, Ruff, GitHub Actions.
+
+---
+
+### Task 1: Safe CHM extraction and output ownership
+
+**Files:**
+- Modify: `tools/chm_extract.py`
+- Create: `tools/output_fs.py`
+- Create: `tools/test_chm_extract.py`
+- Create: `tools/test_output_fs.py`
+
+- [ ] Write failing tests proving each extraction backend receives an empty private directory, failed/invalid attempts do not contaminate later attempts, and only a validated extraction is promoted.
+- [ ] Write failing tests proving filesystem roots, repository roots, sources, and overlapping work/output paths are rejected before removal.
+- [ ] Implement `extract(chm, destination)` with per-backend staging and validated promotion.
+- [ ] Implement a small output-staging interface that owns its temporary directory and atomically promotes a successful tree.
+- [ ] Run the new tests and the complete test suite under uv.
+
+### Task 2: Complete and reproducible PDF conversion
+
+**Files:**
+- Modify: `tools/pdf_convert.py`
+- Modify: `tools/pdf_outlines.json`
+- Modify: `tools/test_pdf_convert.py`
+
+- [ ] Write failing tests for complete normalized `(level, text)` outline comparison, fatal unpinned PDFs, traceability frontmatter, destination collisions, and removal of stale PDF outputs.
+- [ ] Change outline locking to compare the complete post-normalization outline and make missing locks fatal outside `--update-outlines`.
+- [ ] Add source hash, source URL, available metadata, and conversion identity to PDF frontmatter.
+- [ ] Stage PDF output as a complete subtree or register all destinations before writing so stale outputs and collisions cannot survive.
+- [ ] Regenerate outline locks explicitly and run all PDF tests.
+
+### Task 3: uv-controlled CI and incremental publication
+
+**Files:**
+- Create: `.python-version`
+- Create: `requirements.txt`
+- Create: `requirements-dev.txt`
+- Modify: `.github/workflows/markdown-export.yml`
+- Modify: `README.md`
+
+- [ ] Pin Python 3.13 and exact converter dependencies, including Ruff for development.
+- [ ] Replace setup-python/pip installation with uv interpreter provisioning and locked requirement installation.
+- [ ] Make CI tests, lint, and conversion run through the uv-managed environment.
+- [ ] Replace orphan initialization and force-push with checkout/update of ordinary `markdown-export` history, including first-publication handling.
+- [ ] Document the exact local uv setup, test, conversion, and outline-update commands.
+
+### Task 4: Portable multi-input orchestration and emitted-corpus validation
+
+**Files:**
+- Modify: `tools/convert.py`
+- Modify: `tools/fwhelp.lua`
+- Create: `tools/corpus_validation.py`
+- Create: `tools/reporting.py`
+- Create: `tools/test_convert.py`
+- Create: `tools/test_corpus_validation.py`
+
+- [ ] Write failing tests for discovering both CHMs, collision rejection, disambiguating display titles, related frontmatter, proper nested lists, and validation of emitted Markdown links/images.
+- [ ] Replace the hard-coded single-CHM flow with deterministic input discovery and format namespaces or explicit collision rejection.
+- [ ] Move report definitions and fatality into one catalog consumed by JSON, README, console, and workflow summary.
+- [ ] Fix list conversion and title selection without losing source content.
+- [ ] Run post-generation validation over the staged corpus, fail on exporter-caused defects, then promote output.
+
+### Task 5: Full-corpus verification and review
+
+**Files:**
+- Modify only files required by confirmed review defects.
+
+- [ ] Run all tests and Ruff through uv from a clean process.
+- [ ] Build the complete corpus from both CHMs and all PDFs into a new temporary destination.
+- [ ] Audit Markdown count, H1 contract, duplicate titles, PDF hierarchy, local links, images, malformed lists, raw HTML, replacement characters, outline locks, and stale files.
+- [ ] Run the same build twice with one controlled source/output change to verify deterministic and incremental behavior.
+- [ ] Perform independent specification and architecture/clean-code reviews; fix every Critical or Important issue and rerun verification.
+
diff --git a/docs/superpowers/plans/2026-08-22-author-report-markdown.md b/docs/superpowers/plans/2026-08-22-author-report-markdown.md
new file mode 100644
index 00000000..17ed1a38
--- /dev/null
+++ b/docs/superpowers/plans/2026-08-22-author-report-markdown.md
@@ -0,0 +1,106 @@
+# Author Report Markdown Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Publish a repair-oriented Markdown author report beside the canonical JSON report.
+
+**Architecture:** Extend the canonical issue catalog with repair guidance and add a pure `Report.to_markdown()` renderer. The corpus orchestrator writes and links both formats inside its existing atomic staging boundary.
+
+**Tech Stack:** Python 3.13, standard-library `json`, `unittest`, uv, Ruff.
+
+---
+
+### Task 1: Canonical Markdown renderer
+
+**Files:**
+- Modify: `tools/issue_catalog.py`
+- Modify: `tools/reporting.py`
+- Test: `tools/test_reporting.py`
+- Test: `tools/test_issue_catalog.py`
+
+- [ ] **Step 1: Write failing renderer and catalog tests**
+
+Add representative source findings with a `.htm` path, broken target, pipe,
+newline, and structured detail. Assert that `to_markdown()` includes corpus
+counts, code, severity, provenance, canonical repair guidance, exact evidence,
+safe table escaping, and the JSON link. Assert every issue policy has nonempty
+repair guidance.
+
+- [ ] **Step 2: Run the focused tests and verify RED**
+
+Run: `uv run --frozen python -m unittest discover -s tools -p 'test_reporting.py'`
+
+Expected: failure because `IssuePolicy.guidance` and `Report.to_markdown()` do
+not exist.
+
+- [ ] **Step 3: Implement the catalog guidance and Markdown renderer**
+
+Add `guidance: str` to `IssuePolicy`, populate it for every canonical issue,
+and implement deterministic grouping plus safe cell formatting in
+`Report.to_markdown()`.
+
+- [ ] **Step 4: Run focused tests and verify GREEN**
+
+Run: `uv run --frozen python -m unittest discover -s tools -p 'test_reporting.py'`
+
+Expected: all reporting tests pass.
+
+### Task 2: Emit and link the report
+
+**Files:**
+- Modify: `tools/convert.py`
+- Modify: `tools/test_convert.py`
+- Modify: `README.md`
+
+- [ ] **Step 1: Write a failing orchestration test**
+
+Assert that a successful staged build contains `author-report.md` and
+`author-report.json`, and that the generated README links both files.
+
+- [ ] **Step 2: Run the focused orchestration test and verify RED**
+
+Run: `uv run --frozen python -m unittest discover -s tools -p 'test_convert.py'`
+
+Expected: failure because `author-report.md` is absent.
+
+- [ ] **Step 3: Write both reports within the atomic stage**
+
+Seed both linked files before corpus link validation, rewrite both after final
+validation, update the generated README links, and document both formats in
+the repository README.
+
+- [ ] **Step 4: Run focused tests and verify GREEN**
+
+Run: `uv run --frozen python -m unittest discover -s tools -p 'test_convert.py'`
+
+Expected: all orchestration tests pass.
+
+### Task 3: Verify, publish, and hand off
+
+**Files:**
+- Generated: `author-report.md` on branch `markdown-export`
+
+- [ ] **Step 1: Run all local gates**
+
+Run: `uv run --frozen python -m unittest discover -s tools -p 'test_*.py'`
+Run: `uv run --frozen ruff check tools`
+Run: `uv lock --check`
+Run: `git diff --check`
+
+Expected: all commands exit zero.
+
+- [ ] **Step 2: Commit and push the tool branch**
+
+Commit only the feature, tests, and approved design/plan. Keep the imported
+conversation transcript untracked. Push `tools/markdown-export` to the fork.
+
+- [ ] **Step 3: Regenerate and verify the real corpus**
+
+Run the exporter from the committed SHA using the authenticated short-path
+work directory. Verify 0 fatal issues, expected corpus counts, both report
+formats, valid README links, and no internal lock artifacts.
+
+- [ ] **Step 4: Publish and verify the generated branch**
+
+Replace the contents of the fork's `markdown-export` branch in a temporary
+clone, commit, push normally, and verify the remote file URL.
diff --git a/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md b/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md
new file mode 100644
index 00000000..e698a73e
--- /dev/null
+++ b/docs/superpowers/specs/2026-08-21-portable-markdown-export-design.md
@@ -0,0 +1,49 @@
+# Portable Markdown Export Hardening Design
+
+## Goal
+
+Turn the current FwHelps CHM/PDF converter into a safe, reproducible, portable tool that produces reviewable Markdown and can later move to a shared repository without redesigning its core interfaces.
+
+## Scope and ownership
+
+The implementation remains in `FwHelps/tools` for this pass. Repository-specific policy—input discovery, source URLs, workflow triggers, and publication branch—stays at the outer orchestration seam. Extraction, document conversion, output staging, and corpus validation must not depend on the FwHelps repository name or on one hard-coded CHM.
+
+Both CHM files and every repository PDF are inputs. Generated destinations must be registered before writing so two inputs cannot silently claim the same path. A future extraction into a standalone repository should therefore move the converter modules with minimal changes while leaving a thin FwHelps adapter behind.
+
+## Safety and publication
+
+Conversion builds into a converter-owned staging directory. It rejects repository roots, source directories, filesystem roots, and overlapping work/output paths before any recursive removal. A successful build replaces the destination; a failed build leaves the previous destination intact. CHM extraction backends likewise use isolated temporary directories and promote only a validated result.
+
+The publication branch retains parent history. CI checks out the existing export branch when present, replaces only its generated tree, commits the resulting diff, and pushes normally. It must not use orphan commits or force-pushes, so ordinary branch protection remains compatible.
+
+## Reproducibility
+
+`.python-version` pins Python 3.13 for uv. Exact runtime versions live in `requirements.txt`; development-only tools live in `requirements-dev.txt`. CI installs uv, provisions the pinned interpreter, installs the locked requirements, runs tests and lint, then converts. Pandoc remains pinned.
+
+## Conversion contracts
+
+CHM conversion preserves the authored hierarchy while using disambiguating page headings when titles collide. Related-topic metadata is emitted explicitly. Nested lists must render as real Markdown lists.
+
+PDF conversion records the complete normalized outline `(level, text)`, not only a count. Every discovered PDF must have a lock entry in normal mode; updating locks requires the explicit update command. PDF frontmatter includes source path, stable source URL, source content hash, available PDF metadata, heading strategy, and normalized outline count. Converter-owned PDF outputs are replaced as a set so deleted inputs leave no stale files.
+
+## Validation and reporting
+
+Validation runs against the emitted corpus after all conversion steps. It verifies local links and images, destination collisions, unique output paths, duplicate display titles, malformed list markers, replacement characters, raw HTML inventory, and PDF outline locks. Build-breaking checks and advisory checks are defined once and rendered consistently in the README, JSON report, console output, and workflow summary.
+
+Source-document defects remain visible as advisories with provenance; exporter regressions fail the build. Complex tables may remain raw HTML when GFM cannot represent them without information loss, but every retained table is counted.
+
+## Tests and acceptance
+
+Every behavior change follows red-green-refactor. Unit tests cover path rejection, backend isolation, destination collisions, complete outline locks, unpinned PDFs, stale-output removal, multi-CHM discovery, related metadata, title disambiguation, list conversion, and emitted-corpus link/image checks.
+
+Acceptance requires:
+
+- all unit tests and focused lint passing under uv;
+- a complete build of both CHMs and all PDFs;
+- no exporter-caused unresolved links or images;
+- no silent destination overwrites or stale generated files;
+- no malformed `- -` nested-list output;
+- all PDFs pinned by complete normalized outlines;
+- a clean incremental publication design with no force-push;
+- an audited `author-report.json` whose counts match the generated tree.
+
diff --git a/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md b/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md
new file mode 100644
index 00000000..6465ed92
--- /dev/null
+++ b/docs/superpowers/specs/2026-08-21-portable-markdown-export-final-hardening-design.md
@@ -0,0 +1,63 @@
+# Portable Markdown Export: Final Hardening Design
+
+**Date:** 2026-08-21
+**Status:** Approved for planning
+**Scope:** Resolve every confirmed finding from the final cross-cutting review of PR #3.
+
+## Objective
+
+Make the exporter safe to reuse outside FwHelps, reproducible in CI, observable on failure, and resistant to malformed or adversarial source files without changing the validated Markdown produced by the current corpus.
+
+## Safety and source identity
+
+- CHM extraction reuse will require a converter-owned manifest whose source SHA-256 matches the current CHM. Missing or mismatched manifests force a fresh extraction.
+- Direct extraction will reject a destination equal to the source directory or containing the source CHM. A separate descendant work directory remains valid because replacing it cannot remove the source CHM.
+- CHM and PDF discovery will reject symlinks and any resolved input outside the repository root.
+- Absolute local Markdown and image targets will be fatal validation errors rather than silently ignored.
+- PDF cleanup will accept prior manifest entries only when they match paths derivable from the manifest's recorded source PDF and remain under the PDF output root. A corrupt manifest will fail before deletion.
+- Export mutation boundaries will use a non-blocking native advisory lock on a deterministic sibling lockfile. This serializes cooperating exporter invocations for the same normalized destination (and allows different destinations to proceed), but is not a defense against a hostile local process that ignores OS locks. Stale lockfiles do not block because ownership is the held OS handle, not file contents.
+
+## Conversion correctness
+
+- CHM topic and image destinations will share case-insensitive collision detection before any output is written.
+- Lua span conversion will inventory every class before applying the first supported semantic transformation, so mixed known/unknown classes remain build-breaking.
+- CHM TOC parsing will use an HTML parser and support attribute order, quoting, case, and escaped targets.
+- Extraction validation will validate TOC targets and known truncation patterns without rejecting legitimate asset extensions solely because they are new.
+- Frontmatter will use one safe serializer that rejects or escapes control characters and multiline scalar injection.
+- Emitted links will use an allowlist of safe schemes. Source-authored unsafe schemes such as `javascript:` and `file:` are neutralized and reported as source advisories; any unsafe target that survives into the emitted corpus remains fatal.
+
+## Reporting policy
+
+A single issue catalog will define canonical code, label, severity, and default provenance. CHM conversion, PDF conversion, corpus validation, console rendering, README rendering, and workflow summaries will consume this catalog. Source-retained PDF HTML will be reported with source provenance. Unknown issue codes remain fatal by default at integration boundaries.
+
+## CI and publication
+
+- Pull requests will run lint, unit/integration tests, and a dry-run corpus conversion with read-only permissions. PR jobs will never publish or tag.
+- Publication remains limited to trusted pushes, release tags, and non-dry-run manual dispatches.
+- Failure summaries and diagnostic artifacts will upload with `if: always()` when any report or staged diagnostics exist.
+- GitHub Actions will be pinned by immutable commit SHA.
+- Python will be pinned to an exact patch release.
+- Python dependencies, including transitive dependencies, will be captured in a uv lockfile and installed frozen.
+- The downloaded Pandoc package will be verified against a committed expected SHA-256. System packages that cannot be version-pinned reliably on the runner will be explicitly identified as the remaining platform dependency.
+- Workflow write permission will be scoped to the publishing job; validation jobs use read-only contents permission.
+
+## Testing and verification
+
+Each behavioral fix begins with a failing regression test. Coverage will include stale reuse, source/output overlap, symlink escape, corrupt manifests, absolute targets, image collisions, mixed span classes, HHC variants, new asset types, unsafe schemes, YAML control content, issue-catalog consistency, PR non-publication, action/checksum pins, frozen dependency installation, and failed-build artifact behavior.
+
+Completion requires:
+
+1. all unit and integration tests pass;
+2. repository-wide Ruff passes;
+3. workflow static tests pass;
+4. a complete two-CHM/13-PDF export has zero fatal issues;
+5. an independent final review has no unresolved Critical or Important findings;
+6. the code branch is committed and fast-forward pushed;
+7. the `markdown-export` branch is regenerated from that commit and fast-forward pushed.
+
+## Non-goals and retained decisions
+
+- Existing source-quality advisories remain advisories unless they create unsafe or invalid generated output.
+- The exporter remains in FwHelps for this PR, but reusable modules contain no FwHelps-specific policy.
+- GitHub repository branch-protection settings are documented and verified where visible, but are not changed by this code patch without separate authorization.
+- No force pushes or history replacement are allowed.
diff --git a/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md b/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md
new file mode 100644
index 00000000..137b93aa
--- /dev/null
+++ b/docs/superpowers/specs/2026-08-22-author-report-markdown-design.md
@@ -0,0 +1,46 @@
+# Author Report Markdown Design
+
+## Goal
+
+Generate a human-readable `author-report.md` beside `author-report.json`. A
+RoboHelp author must be able to identify the affected source topic or PDF,
+understand the evidence, and know the appropriate repair action without
+reading exporter code.
+
+## Design
+
+`Report` remains the single reporting model and JSON remains the stable
+machine-readable format. The issue catalog gains canonical repair guidance so
+labels, severity, provenance, and advice cannot drift between renderers.
+`Report.to_markdown()` renders:
+
+- corpus source ref and CHM/PDF/topic/image counts;
+- fatal and advisory totals plus counts by issue type;
+- one section per issue code with severity, provenance, and repair guidance;
+- every finding with its exact source/output path, message, and structured
+ evidence, escaped for deterministic Markdown tables;
+- a link to the JSON report for automation and complete structured data.
+
+Source issues use the authored `.htm` or PDF path already retained in the
+canonical issue. Generated-tree validation findings retain their emitted
+`chm/.../*.md` or `pdf/.../*.md` path, making exporter failures reproducible.
+Repair guidance explains whether the correction belongs in RoboHelp/PDF
+source or exporter code.
+
+The corpus README links to both report formats. Both reports are written only
+inside the private output stage and are promoted atomically with the corpus.
+
+## Error Handling and Safety
+
+Markdown rendering is pure and deterministic. Table cells escape pipes and
+line breaks; structured evidence is serialized as compact Unicode JSON. An
+unknown issue uses the existing fatal fallback policy and guidance directing
+maintainers to add it to the canonical catalog.
+
+## Tests
+
+Tests verify the Markdown report contains corpus identity, summary totals,
+canonical guidance, exact RoboHelp source paths, problematic targets,
+structured evidence, escaping, and the JSON link. An orchestration test
+verifies successful builds emit both report files and README links. The full
+unit suite, Ruff, uv lock check, and a real corpus build remain release gates.
diff --git a/pyproject.toml b/pyproject.toml
new file mode 100644
index 00000000..eee0c1a3
--- /dev/null
+++ b/pyproject.toml
@@ -0,0 +1,17 @@
+[project]
+name = "fwhelps"
+version = "0.0.0"
+description = "FieldWorks help corpus conversion tools"
+requires-python = ">=3.13.5,<3.14"
+dependencies = [
+ "pymupdf==1.28.2",
+ "pymupdf4llm==1.28.2",
+]
+
+[dependency-groups]
+dev = [
+ "ruff==0.16.4",
+]
+
+[tool.uv]
+package = false
diff --git a/requirements-dev.txt b/requirements-dev.txt
new file mode 100644
index 00000000..c6ddef7d
--- /dev/null
+++ b/requirements-dev.txt
@@ -0,0 +1,3 @@
+# This file was autogenerated by uv via the following command:
+# uv export --frozen --only-group dev --no-hashes --format requirements.txt --output-file requirements-dev.txt
+ruff==0.16.4
diff --git a/requirements.txt b/requirements.txt
new file mode 100644
index 00000000..9ec79c10
--- /dev/null
+++ b/requirements.txt
@@ -0,0 +1,31 @@
+# This file was autogenerated by uv via the following command:
+# uv export --frozen --no-dev --no-hashes --format requirements.txt --output-file requirements.txt
+flatbuffers==25.12.19
+ # via onnxruntime
+networkx==3.6.1
+ # via pymupdf-layout
+numpy==2.5.2
+ # via
+ # onnxruntime
+ # pymupdf-layout
+onnxruntime==1.29.0
+ # via pymupdf-layout
+packaging==26.3
+ # via onnxruntime
+protobuf==7.36.0
+ # via onnxruntime
+psutil==7.2.2
+ # via pymupdf4llm
+pymupdf==1.28.2
+ # via
+ # fwhelps
+ # pymupdf-layout
+ # pymupdf4llm
+pymupdf-layout==1.28.2
+ # via pymupdf4llm
+pymupdf4llm==1.28.2
+ # via fwhelps
+pyyaml==6.0.3
+ # via pymupdf-layout
+tabulate==0.10.0
+ # via pymupdf4llm
diff --git a/tools/chm_convert.py b/tools/chm_convert.py
new file mode 100644
index 00000000..c5c07dec
--- /dev/null
+++ b/tools/chm_convert.py
@@ -0,0 +1,633 @@
+"""Reusable conversion of one extracted CHM into a namespaced Markdown tree."""
+
+from __future__ import annotations
+
+import hashlib
+import html
+import json
+import re
+import subprocess
+import uuid
+from collections import Counter, defaultdict
+from pathlib import Path
+from urllib.parse import quote, unquote, urldefrag
+
+from chm_extract import _extract_already_locked, extract, validate
+from chm_metadata import TopicMeta, parse_toc, safe_stem
+from frontmatter import yaml_scalar
+from output_fs import export_locks
+from source_safety import (
+ SourceSafetyError,
+ first_link_in_path,
+ validate_source_tree,
+)
+
+LUA = Path(__file__).with_name("fwhelp.lua")
+IMAGE_EXTS = {".gif", ".png", ".jpg", ".jpeg", ".bmp", ".svg", ".ico", ".webp"}
+EXTRACTION_MANIFEST = ".chm-extraction-manifest.json"
+
+
+def _normalize_source(source: str) -> str:
+ """Normalize RoboHelp NBSP spacing at the HTML-to-Markdown boundary.
+
+ RoboHelp uses CP1252 byte 0xA0 and HTML NBSP entities for both layout
+ spacing and empty table cells. Markdown/Pandoc can emit U+FFFD for these
+ values, so ordinary spaces preserve word separation without retaining an
+ unsafe format-specific spacing character.
+ """
+ source = source.replace("\u00a0", " ")
+ return re.sub(r"(?i) | ", " ", source)
+
+
+def _norm(path: str) -> str:
+ parts: list[str] = []
+ for segment in path.replace("\\", "/").split("/"):
+ if segment in ("", "."):
+ continue
+ if segment == "..":
+ if parts:
+ parts.pop()
+ else:
+ parts.append(segment)
+ return "/".join(parts)
+
+
+def _relative(source: str, target: str) -> str:
+ base = Path(source).parent.parts
+ parts = Path(target).with_suffix(".md").parts
+ common = 0
+ while common < len(base) and common < len(parts) and base[common] == parts[common]:
+ common += 1
+ return "/".join(("..",) * (len(base) - common) + parts[common:])
+
+
+def frontmatter(fields: dict) -> str:
+
+ lines = ["---"]
+ for key, value in fields.items():
+ if value in (None, "", [], {}):
+ continue
+ if isinstance(value, list):
+ lines.append(f"{key}:")
+ lines.extend(f" - {yaml_scalar(item)}" for item in value)
+ else:
+ lines.append(f"{key}: {yaml_scalar(value)}")
+ lines.append("---")
+ return "\n".join(lines)
+
+
+def run_pandoc(html_text: str, tmp: Path) -> tuple[str, list[str]]:
+ tmp.write_text(html_text, encoding="utf-8")
+ proc = subprocess.run(
+ ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none",
+ f"--lua-filter={LUA}", str(tmp)],
+ capture_output=True, text=True, encoding="utf-8",
+ check=False,
+ )
+ if proc.returncode:
+ raise RuntimeError(f"pandoc failed: {(proc.stderr or '').strip()[:400]}")
+ unmapped: list[str] = []
+ for line in (proc.stderr or "").splitlines():
+ if line.startswith("FWHELP_UNMAPPED_SPAN"):
+ unmapped.extend(part.split("=", 1)[0] for part in line.split(" ", 1)[1].split(","))
+ return _normalize_nested_lists(proc.stdout or ""), unmapped
+
+
+def _normalize_nested_lists(markdown: str) -> str:
+ """Repair Pandoc's literal ``- -`` spelling without dropping list items."""
+ return re.sub(r"^(\s*)-\s+-\s+", lambda m: m.group(1) + " - ", markdown, flags=re.MULTILINE)
+
+
+_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
+_DRIVE = re.compile(r"^[A-Za-z]:[\\/]")
+_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]")
+
+
+def _uri_kind(raw: str) -> tuple[str, str]:
+ value = unquote(urldefrag(raw.strip())[0]).replace("\\", "/")
+ canonical = _URI_ASCII_SEPARATORS.sub("", value)
+ if not canonical:
+ return "fragment", value
+ if canonical.startswith("/") or _DRIVE.match(canonical):
+ return "path_escape", value
+ scheme = _SCHEME.match(canonical)
+ if scheme:
+ return ("external", value) if scheme.group(0)[:-1].casefold() in {
+ "http", "https", "mailto"
+ } else ("unsafe_uri", value)
+ return "local", value
+
+
+def _check_links(links: list[str], topic_rel: str, known: set[str]) -> list[str]:
+ broken = []
+ for href in links:
+ kind, value = _uri_kind(href)
+ if kind in {"external", "fragment", "unsafe_uri", "path_escape"}:
+ continue
+ path = value
+ if not path or Path(path).suffix.lower() in IMAGE_EXTS:
+ continue
+ target = _norm((Path(topic_rel).parent / path).as_posix())
+ if target.casefold() not in known:
+ broken.append(href)
+ return broken
+
+
+def _record_unsafe_uri(report: dict[str, list], rel: str, raw: str) -> None:
+ kind, _ = _uri_kind(raw)
+ if kind in {"unsafe_uri", "path_escape"}:
+ code = f"source_{kind}"
+ if not any(
+ str(item[0]).casefold() == rel.casefold()
+ and str(item[1]).casefold() == raw.casefold()
+ for item in report[code]
+ ):
+ report[code].append([rel, raw])
+
+
+def _append_reference(report: dict[str, list], code: str, rel: str, raw: str) -> None:
+ if not any(
+ str(item[0]).casefold() == rel.casefold()
+ and str(item[1]).casefold() == raw.casefold()
+ for item in report[code]
+ ):
+ report[code].append([rel, raw])
+
+
+def _sanitize_markdown_targets(markdown: str, report: dict[str, list], rel: str) -> str:
+ """Neutralize unsafe Markdown destinations while preserving safe syntax."""
+ opener = re.compile(r"(?!)?\[[^\]]*\]\(")
+ output: list[str] = []
+ cursor = 0
+ for match in opener.finditer(markdown):
+ output.append(markdown[cursor:match.end()])
+ pos, depth = match.end(), 1
+ angle = pos < len(markdown) and markdown[pos] == "<"
+ while pos < len(markdown):
+ char = markdown[pos]
+ if angle and char == ">":
+ angle = False
+ elif not angle and char == "(":
+ depth += 1
+ elif not angle and char == ")":
+ depth -= 1
+ if depth == 0:
+ break
+ pos += 1
+ if depth:
+ continue
+ inner = markdown[match.end():pos]
+ if inner.startswith("<") and ">" in inner:
+ target = inner[1:inner.index(">")]
+ suffix = inner[inner.index(">") + 1:]
+ replacement_target = ""
+ else:
+ pieces = inner.split(None, 1)
+ target = pieces[0] if pieces else ""
+ suffix = (" " + pieces[1]) if len(pieces) == 2 else ""
+ replacement_target = "TARGET"
+ kind, _ = _uri_kind(target)
+ if kind in {"unsafe_uri", "path_escape"}:
+ report.setdefault(kind, []).append([rel, target])
+ replacement = replacement_target.replace("TARGET", "#") + suffix
+ output.append(replacement + ")")
+ else:
+ output.append(inner + ")")
+ cursor = pos + 1
+ output.append(markdown[cursor:])
+ return "".join(output)
+
+
+_ATTR_TARGET = re.compile(
+ r"(?P\b(?:href|src)\s*=\s*)"
+ r"(?:(?P[\"'])(?P.*?)(?P=quote)|(?P[^\s>]+))",
+ re.IGNORECASE,
+)
+
+
+def _canonical_case(target: str, rel: str, canonical: dict[str, str]) -> str | None:
+ """Return target restated in its file's real case, or None if it already is."""
+ kind, value = _uri_kind(target)
+ if kind != "local" or not value:
+ return None
+ resolved = _norm((Path(rel).parent / value).as_posix())
+ actual = canonical.get(resolved.casefold())
+ if actual is None or actual == resolved:
+ return None
+ prefix = _norm(Path(rel).parent.as_posix())
+ shared = 0
+ actual_parts, prefix_parts = actual.split("/"), prefix.split("/") if prefix else []
+ while (shared < len(prefix_parts) and shared < len(actual_parts) - 1
+ and prefix_parts[shared] == actual_parts[shared]):
+ shared += 1
+ hops = [".."] * (len(prefix_parts) - shared)
+ fragment = urldefrag(target.strip())[1]
+ return "/".join(hops + actual_parts[shared:]) + (f"#{fragment}" if fragment else "")
+
+
+def _canonicalize_link_case(source: str, rel: str, canonical: dict[str, str],
+ report: dict[str, list]) -> str:
+ """Restate authored links in their target's real case before conversion.
+
+ RoboHelp resolves hrefs case-insensitively, so authored case drifts from the
+ topic's real path. Such a link still opens inside a CHM and 404s once the
+ corpus is published to a case-sensitive host, so emit the case the file
+ actually has and report the drift back to the author.
+ """
+ def replace(match: re.Match[str]) -> str:
+ target = match.group("quoted") if match.group("quote") else match.group("bare")
+ # Source attributes carry HTML entities ("&") over path characters
+ # that the extracted filenames spell literally, so compare decoded and
+ # re-encode the corrected path on the way out.
+ fixed = _canonical_case(html.unescape(target), rel, canonical)
+ if fixed is None:
+ return match.group(0)
+ fixed = quote(fixed, safe="/#")
+ _append_reference(report, "link_case_mismatches", rel, target)
+ if match.group("quote"):
+ return match.group("prefix") + match.group("quote") + fixed + match.group("quote")
+ return match.group("prefix") + fixed
+
+ return _ATTR_TARGET.sub(replace, source)
+
+
+def _sanitize_raw_targets(markdown: str, report: dict[str, list], rel: str) -> str:
+ def replace(match: re.Match[str]) -> str:
+ target = match.group("quoted") if match.group("quote") else match.group("bare")
+ kind, _ = _uri_kind(target)
+ if kind in {"unsafe_uri", "path_escape"}:
+ report.setdefault(kind, []).append([rel, target])
+ if match.group("quote"):
+ return match.group("prefix") + match.group("quote") + "#" + match.group("quote")
+ return match.group("prefix") + "#"
+ return match.group(0)
+
+ return _ATTR_TARGET.sub(replace, markdown)
+
+
+def _version(extraction: Path) -> str:
+ for path in sorted(extraction.rglob("*"), key=lambda item: item.as_posix().casefold()):
+ if not path.is_file() or path.suffix.lower() != ".hhk":
+ continue
+ match = re.search(r"_(\d+\.\d+)\.hhk$", path.name)
+ if match:
+ return match.group(1)
+ return ""
+
+
+def _sha256(path: Path) -> str:
+ digest = hashlib.sha256()
+ with Path(path).open("rb") as stream:
+ for chunk in iter(lambda: stream.read(1024 * 1024), b""):
+ digest.update(chunk)
+ return digest.hexdigest()
+
+
+def _authenticated_manifest(path: Path, chm: Path, source_hash: str) -> bool:
+ """Return whether an extraction manifest authenticates this exact CHM."""
+ try:
+ loaded = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError):
+ return False
+ if not isinstance(loaded, dict):
+ return False
+ schema = loaded.get("schema")
+ return (
+ isinstance(schema, int) and not isinstance(schema, bool) and schema == 1
+ and loaded.get("source_name") == chm.name
+ and loaded.get("source_sha256") == source_hash
+ and isinstance(loaded.get("source_sha256"), str)
+ and loaded["source_sha256"] == loaded["source_sha256"].lower()
+ and len(loaded["source_sha256"]) == 64
+ and all(char in "0123456789abcdef" for char in loaded["source_sha256"])
+ )
+
+
+def _write_extraction_manifest(extraction: Path, chm: Path, source_hash: str) -> None:
+ extraction.mkdir(parents=True, exist_ok=True)
+ (extraction / EXTRACTION_MANIFEST).write_text(
+ json.dumps({
+ "schema": 1,
+ "source_name": chm.name,
+ "source_sha256": source_hash,
+ }, indent=2) + "\n",
+ encoding="utf-8",
+ )
+
+
+def _topic_files(extraction: Path) -> list[Path]:
+ return sorted(
+ (path for path in extraction.rglob("*")
+ if path.is_file() and path.suffix.lower() in {".htm", ".html"}),
+ key=lambda path: (
+ path.relative_to(extraction).as_posix().casefold(),
+ path.relative_to(extraction).as_posix(),
+ ),
+ )
+
+
+def _validate_conversion_destination(chm: Path, extraction: Path, destination: Path) -> None:
+ """Reject linked or source-overlapping destinations before any write."""
+ if (link := first_link_in_path(destination)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction conversion destination: {link}")
+ if Path(destination).exists():
+ validate_source_tree(destination)
+ destination_abs = Path(destination).resolve(strict=False)
+ chm_abs = Path(chm).resolve(strict=False)
+ extraction_abs = Path(extraction).resolve(strict=False)
+ for label, protected in (("CHM", chm_abs), ("extraction", extraction_abs)):
+ if destination_abs == protected or destination_abs in protected.parents:
+ raise SourceSafetyError(
+ f"refusing conversion destination overlapping {label}: {destination}"
+ )
+
+
+def _convert_chm_locked(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop", source_url_base: str | None = None,
+ extractor=extract, extract_fn=None) -> dict:
+ """Extract and convert one CHM, returning report facts and TOC nodes."""
+ chm = Path(chm)
+ if extract_fn is not None:
+ extractor = extract_fn
+ extraction = Path(work_root) / safe_stem(chm.name)
+ destination = Path(destination)
+ if (link := first_link_in_path(chm)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}")
+ if (link := first_link_in_path(extraction)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}")
+ _validate_conversion_destination(chm, extraction, destination)
+ source_hash = _sha256(chm)
+ advisory: list[str] = []
+ manifest = extraction / EXTRACTION_MANIFEST
+ if reuse and extraction.exists():
+ validate_source_tree(extraction)
+ if reuse and _topic_files(extraction) and _authenticated_manifest(manifest, chm, source_hash):
+ fatal, advisory = validate(extraction)
+ if fatal:
+ raise RuntimeError("reused extraction failed validation: " + "; ".join(fatal))
+ else:
+ default_extractor = extractor is extract
+ if extractor is extract:
+ # ``convert_chm`` already owns extraction's lock for its entire
+ # read/convert lifetime; taking it again would deadlock.
+ extractor = _extract_already_locked
+ extractor(chm, extraction)
+ validate_source_tree(extraction)
+ if default_extractor:
+ fatal, advisory = validate(extraction)
+ if fatal:
+ raise RuntimeError(
+ "fresh extraction failed validation: " + "; ".join(fatal)
+ )
+ else:
+ advisory = list(getattr(extractor, "advisory", []))
+ # ``extract`` promotes only after its staged extraction passes its
+ # checks. Record identity only after that call has returned.
+ _write_extraction_manifest(extraction, chm, source_hash)
+ _validate_conversion_destination(chm, extraction, destination)
+ hhc = next((path for path in sorted(extraction.rglob("*"), key=lambda item: item.as_posix().casefold())
+ if path.is_file() and path.suffix.lower() == ".hhc"), None)
+ toc = parse_toc(hhc) if hhc else []
+ crumbs = {
+ node["href"].casefold(): node["breadcrumb"]
+ for node in toc if node["href"]
+ }
+ version = _version(extraction)
+ source_hash = "sha256:" + source_hash
+ all_topics = [path.relative_to(extraction).as_posix() for path in _topic_files(extraction)]
+ topics = list(all_topics)
+ known = {topic.casefold() for topic in topics}
+ extraction_files = {
+ path.relative_to(extraction).as_posix().casefold()
+ for path in extraction.rglob("*") if path.is_file()
+ }
+ if limit:
+ topics = topics[:limit]
+ claimed: dict[str, list[str]] = defaultdict(list)
+ # Links are canonicalized against the source paths, before the .htm to .md
+ # rewrite, so authored case is corrected once at the conversion boundary.
+ canonical_sources: dict[str, str] = {}
+ for rel in all_topics:
+ claimed[Path(rel).with_suffix(".md").as_posix().casefold()].append(f"topic:{rel}")
+ canonical_sources[rel.casefold()] = rel
+ for image in extraction.rglob("*"):
+ if image.is_file() and image.suffix.lower() in IMAGE_EXTS:
+ rel = image.relative_to(extraction).as_posix()
+ claimed[rel.casefold()].append(f"asset:{rel}")
+ canonical_sources[rel.casefold()] = rel
+ collisions = [(dest, paths) for dest, paths in sorted(claimed.items()) if len(paths) > 1]
+ if collisions:
+ report: dict[str, list] = {"destination_collisions": collisions}
+ return {"chm": chm.name, "stem": safe_stem(chm.name), "version": _version(extraction),
+ "toc": toc, "topics": 0, "images": 0, "topics_paths": [], "report": report}
+ destination.mkdir(parents=True, exist_ok=True)
+ # Keep each invocation's scratch file unique even when sibling
+ # destinations share a work directory. The finally block below removes
+ # only this converter-owned path.
+ tmp = destination.parent / (
+ f".{safe_stem(chm.name)}-pandoc-{uuid.uuid4().hex}.html"
+ )
+ report: dict[str, list] = defaultdict(list)
+ unmapped: Counter[str] = Counter()
+ unmapped_topics: dict[str, set[str]] = defaultdict(set)
+ source_replacement_paths: list[str] = []
+ titles: dict[str, list[str]] = defaultdict(list)
+ records: dict[str, tuple[str, TopicMeta]] = {}
+ for rel in topics:
+ meta = TopicMeta()
+ meta.feed((extraction / rel).read_bytes().decode("cp1252", errors="replace"))
+ original = html.unescape(meta.title).strip() or Path(rel).stem.replace("_", " ")
+ records[rel] = (original, meta)
+ titles[original.casefold()].append(rel)
+ display_titles: dict[str, str] = {}
+ for paths in titles.values():
+ if len(paths) == 1:
+ display_titles[paths[0]] = records[paths[0]][0]
+ continue
+ original = records[paths[0]][0]
+ used: set[str] = set()
+ for rel in paths:
+ _, meta = records[rel]
+ heading = html.unescape(meta.page_heading).strip()
+ candidate = heading if heading and heading.casefold() != original.casefold() else f"{original} ({Path(rel).stem.replace('_', ' ')})"
+ display_titles[rel] = candidate
+ used.add(candidate.casefold())
+ if len(used) != len(paths):
+ report["duplicate_titles"].append([original, paths])
+ written = 0
+ try:
+ for rel in topics:
+ raw = (extraction / rel).read_bytes()
+ source = _normalize_source(raw.decode("cp1252", errors="replace"))
+ source = _canonicalize_link_case(source, rel, canonical_sources, report)
+ if "\ufffd" in source:
+ source_replacement_paths.append(rel)
+ original_title, meta = records[rel]
+ title = display_titles[rel]
+ for href in meta.links:
+ _record_unsafe_uri(report, rel, href)
+ for href in _check_links(meta.links, rel, known):
+ _append_reference(report, "broken_links", rel, href)
+ for image_href in meta.images:
+ _record_unsafe_uri(report, rel, image_href)
+ image_kind, image_path = _uri_kind(image_href)
+ if image_kind in {"external", "fragment", "unsafe_uri", "path_escape"}:
+ continue
+ image_target = _norm((Path(rel).parent / image_path).as_posix())
+ if image_path and image_target.casefold() not in extraction_files:
+ _append_reference(report, "broken_images", rel, image_href)
+ try:
+ markdown, unknown = run_pandoc(source, tmp)
+ except RuntimeError as exc:
+ report["pandoc_failures"].append([rel, str(exc)])
+ continue
+ unmapped.update(unknown)
+ for class_name in unknown:
+ unmapped_topics[class_name].add(rel)
+ markdown = re.sub(r"^\s*#\s+.*?\n+", "", markdown, count=1)
+ markdown = re.sub(r"^# ", "## ", markdown, flags=re.MULTILINE)
+ markdown = _normalize_nested_lists(markdown)
+ markdown = _sanitize_markdown_targets(markdown, report, rel)
+ markdown = _sanitize_raw_targets(markdown, report, rel)
+ breadcrumb = crumbs.get(rel.casefold()) or [part.replace("_", " ") for part in Path(rel).parent.parts]
+ if any("chm::" in item or item.endswith(".hhc") for item in breadcrumb):
+ breadcrumb = []
+ if not crumbs.get(rel.casefold()):
+ report["not_in_toc"].append(rel)
+ related = []
+ for label, href in meta.related:
+ target = _norm((Path(rel).parent / urldefrag(unquote(href))[0]).as_posix())
+ if target.casefold() in known:
+ related.append(f"{label} -> {_relative(rel, target)}")
+ page_heading = html.unescape(meta.page_heading).strip()
+ if re.sub(r"[^a-z0-9]", "", page_heading.lower()) in {
+ re.sub(r"[^a-z0-9]", "", title.lower()),
+ re.sub(r"[^a-z0-9]", "", original_title.lower()),
+ }:
+ page_heading = ""
+ fields = {
+ "title": title,
+ "source_title": original_title,
+ "breadcrumb": breadcrumb,
+ "source": rel,
+ "source_url": (
+ f"{source_url_base.rstrip('/')}/index.htm#t={quote(rel)}"
+ if source_url_base else None
+ ),
+ "source_hash": source_hash,
+ "keywords": list(dict.fromkeys(k.strip() for k in meta.meta.get("rh-index-keywords", "").split(",") if k.strip())),
+ "related": related,
+ "fw_help_version": version,
+ "page_heading": page_heading,
+ "type": "index" if "overview" in Path(rel).stem.lower() else "topic",
+ "content_hash": "sha256:" + hashlib.sha256(markdown.encode()).hexdigest()[:16],
+ }
+ safe_title = yaml_scalar(title)[1:-1]
+ trail = " › ".join(breadcrumb[:-1]) if len(breadcrumb) > 1 else ""
+ safe_trail = yaml_scalar(trail)[1:-1] if trail else ""
+ header = f"{frontmatter(fields)}\n\n# {safe_title}\n"
+ if trail:
+ header += f"\n*{safe_trail}*\n"
+ markdown = re.sub(r"^#{1,6} Related Topics\s*$", "## Related topics", markdown, flags=re.MULTILINE)
+ markdown = re.sub(r"^#{1,6} Related Internet Sites\s*$", "## Related links", markdown, flags=re.MULTILINE)
+ target = destination / Path(rel).with_suffix(".md")
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_text(header + "\n" + markdown.strip() + "\n", encoding="utf-8")
+ written += 1
+ images = 0
+ for image in extraction.rglob("*"):
+ if image.is_file() and image.suffix.lower() in IMAGE_EXTS:
+ target = destination / image.relative_to(extraction)
+ target.parent.mkdir(parents=True, exist_ok=True)
+ target.write_bytes(image.read_bytes())
+ images += 1
+ finally:
+ tmp.unlink(missing_ok=True)
+ for item in advisory:
+ if item.startswith(("source_unsafe_uri:", "source_path_escape:")):
+ code, source, raw = item.split(": ", 2)
+ if not any(
+ str(item[0]).casefold() == source.casefold()
+ and str(item[1]).casefold() == raw.casefold()
+ for item in report[code]
+ ):
+ report[code].append([source, raw])
+ elif item.startswith("HTML "):
+ # HTML missing-target advisories are rechecked above so the
+ # converter emits one source_missing_link/image report only.
+ continue
+ else:
+ report["stale_toc_entries"].append(item)
+ if unmapped:
+ report["unmapped_span_classes"] = [
+ [name, count, sorted(unmapped_topics[name], key=str.casefold)]
+ for name, count in unmapped.most_common()
+ ]
+ return {"chm": chm.name, "stem": safe_stem(chm.name), "version": version,
+ "toc": toc, "topics": written, "images": images, "report": dict(report),
+ "source_replacement_paths": source_replacement_paths,
+ "topics_paths": topics}
+
+
+def convert_chm(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop", source_url_base: str | None = None,
+ extractor=extract, extract_fn=None) -> dict:
+ """Convert one CHM with deterministic extraction-then-destination locks.
+
+ The extraction/work-root lock is acquired first and held through all
+ extraction, validation, source reads, and Pandoc conversion. The
+ destination lock is acquired second. This order avoids lock inversion;
+ the internal already-locked extraction entry point prevents recursive
+ acquisition.
+ """
+ chm = Path(chm)
+ extraction = Path(work_root) / safe_stem(chm.name)
+ destination = Path(destination)
+ # Perform pure validation first so equal/overlapping targets report the
+ # intended source-safety error rather than a duplicate-lock busy error.
+ if (link := first_link_in_path(chm)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}")
+ if (link := first_link_in_path(extraction)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}")
+ _validate_conversion_destination(chm, extraction, destination)
+ with export_locks(extraction, destination):
+ # The implementation repeats source and destination validation after
+ # lock acquisition, immediately before its first destination write.
+ return _convert_chm_locked(
+ chm, work_root, destination, reuse=reuse, limit=limit,
+ source_ref=source_ref, source_url_base=source_url_base,
+ extractor=extractor, extract_fn=extract_fn,
+ )
+
+
+def run_in_private_stage(chm: Path, work_root: Path, destination: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop",
+ source_url_base: str | None = None,
+ extractor=extract, extract_fn=None) -> dict:
+ """Convert into a caller-owned private stage while locking extraction.
+
+ The caller must guarantee that ``destination`` is an unshared staging
+ path. Such a path needs no destination lock; creating one beside it would
+ turn the lockfile into generated content when the enclosing stage is
+ promoted.
+ """
+ chm = Path(chm)
+ extraction = Path(work_root) / safe_stem(chm.name)
+ destination = Path(destination)
+ if (link := first_link_in_path(chm)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction CHM path: {link}")
+ if (link := first_link_in_path(extraction)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction extraction path: {link}")
+ _validate_conversion_destination(chm, extraction, destination)
+ with export_locks(extraction):
+ _validate_conversion_destination(chm, extraction, destination)
+ return _convert_chm_locked(
+ chm, work_root, destination, reuse=reuse, limit=limit,
+ source_ref=source_ref, source_url_base=source_url_base,
+ extractor=extractor, extract_fn=extract_fn,
+ )
+
+
+__all__ = [
+ "convert_chm", "frontmatter", "run_in_private_stage", "run_pandoc", "yaml_scalar",
+]
diff --git a/tools/chm_extract.py b/tools/chm_extract.py
new file mode 100644
index 00000000..8fc11e37
--- /dev/null
+++ b/tools/chm_extract.py
@@ -0,0 +1,328 @@
+"""Cross-platform CHM extraction.
+
+Tries, in order:
+ 1. 7z / 7za / 7zz (Linux + Windows, best choice for CI)
+ 2. extract_chmLib (Linux, chmlib package)
+ 3. hh.exe -decompile (Windows only, ships with the OS)
+"""
+
+from __future__ import annotations
+
+import os
+import re
+import shutil
+import subprocess
+import sys
+import time
+from html.parser import HTMLParser
+from pathlib import Path
+from urllib.parse import unquote, urldefrag
+
+from output_fs import ExportLock, OutputPathError, OutputStaging
+from source_safety import first_link_in_path, validate_source_tree
+
+
+class ExtractError(RuntimeError):
+ pass
+
+
+def _validate_extract_paths(chm: Path, outdir: Path) -> tuple[Path, Path]:
+ """Reject extraction destinations that could consume their source."""
+ chm = Path(os.path.abspath(os.fspath(Path(chm).expanduser())))
+ outdir = Path(os.path.abspath(os.fspath(Path(outdir).expanduser())))
+ if (link := first_link_in_path(chm)) is not None:
+ raise ExtractError(f"refusing symlink/junction CHM path component: {link}")
+ if (link := first_link_in_path(outdir)) is not None:
+ raise OutputPathError(f"refusing symlink/junction extraction destination: {link}")
+ try:
+ source = chm.resolve(strict=True)
+ except OSError as exc:
+ raise ExtractError(f"no such CHM: {chm}") from exc
+ destination = outdir.resolve(strict=False)
+ if not source.is_file():
+ raise ExtractError(f"no such CHM: {chm}")
+ if destination == source or destination in source.parents:
+ raise OutputPathError(
+ "refusing extraction destination that contains the source CHM: "
+ f"source={source}, destination={destination}"
+ )
+ # Preserve the lexical destination so OutputStaging can independently
+ # enforce its own path-chain ownership checks before creating staging.
+ return chm, outdir
+
+
+def _sevenzip(chm: Path, outdir: Path) -> str | None:
+ for exe in ("7z", "7za", "7zz"):
+ found = shutil.which(exe)
+ if not found:
+ continue
+ subprocess.run(
+ [found, "x", "-y", f"-o{outdir}", str(chm)],
+ check=True,
+ stdout=subprocess.DEVNULL,
+ stderr=subprocess.PIPE,
+ )
+ return exe
+ return None
+
+
+def _chmlib(chm: Path, outdir: Path) -> str | None:
+ found = shutil.which("extract_chmLib")
+ if not found:
+ return None
+ subprocess.run(
+ [found, str(chm), str(outdir)],
+ check=True,
+ stdout=subprocess.DEVNULL,
+ stderr=subprocess.PIPE,
+ )
+ return "extract_chmLib"
+
+
+# The deepest path inside FieldWorks_Language_Explorer_Help.chm is ~140 chars.
+# hh.exe silently TRUNCATES any output path that exceeds Windows MAX_PATH (260)
+# -- no error, no non-zero exit, just a file named "Foo_field_(Extended_Note)"
+# with the ".htm" chopped off. Refuse to run rather than corrupt the corpus.
+MAX_PATH = 260
+ASSUMED_MAX_INTERNAL = 160
+
+
+def _hh(chm: Path, outdir: Path) -> str | None:
+ if os.name != "nt":
+ return None
+ found = shutil.which("hh") or r"C:\Windows\hh.exe"
+ if not Path(found).exists():
+ return None
+
+ budget = MAX_PATH - ASSUMED_MAX_INTERNAL
+ if len(str(outdir.resolve())) > budget:
+ raise ExtractError(
+ "hh.exe would silently truncate filenames: output path is "
+ f"{len(str(outdir.resolve()))} chars, must be <= {budget}.\n"
+ f" {outdir.resolve()}\n"
+ " Use a shorter --work directory (e.g. C:\\fwhelps-work), or "
+ "install 7-Zip, which has no such limit."
+ )
+ # hh.exe detaches immediately and wants native separators + absolute paths.
+ subprocess.run(
+ [str(found), "-decompile", str(outdir.resolve()), str(chm.resolve())],
+ check=True,
+ )
+ # It returns before the write finishes; wait for the file count to settle.
+ stable, last = 0, -1
+ for _ in range(60):
+ time.sleep(0.5)
+ count = sum(1 for _ in outdir.rglob("*") if _.is_file())
+ stable = stable + 1 if count == last and count > 0 else 0
+ if stable >= 4:
+ break
+ last = count
+ return "hh.exe"
+
+
+EXTRACTION_MANIFEST_NAME = ".chm-extraction-manifest.json"
+
+
+class _ReferenceParser(HTMLParser):
+ """Collect CHM sitemap locals and HTML href/src references."""
+
+ def __init__(self) -> None:
+ super().__init__(convert_charrefs=True)
+ self.local: list[str] = []
+ self.html: list[tuple[str, str]] = []
+
+ def handle_starttag(self, tag: str, attrs) -> None:
+ values = {key.casefold(): value or "" for key, value in attrs}
+ if (tag.casefold() == "param"
+ and values.get("name", "").casefold() == "local"
+ and values.get("value")):
+ self.local.append(values["value"])
+ for attr in ("href", "src"):
+ if values.get(attr):
+ self.html.append((attr, values[attr]))
+
+
+_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
+_DRIVE = re.compile(r"^[A-Za-z]:[\\/]")
+_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]")
+
+
+def _reference_kind(raw: str) -> tuple[str, str]:
+ """Return (kind, path), classifying URI safety before filesystem access."""
+ value = unquote(urldefrag(raw.strip())[0]).replace("\\", "/")
+ canonical = _URI_ASCII_SEPARATORS.sub("", value)
+ if not canonical:
+ return "fragment", value
+ # A leading double slash is a UNC path in CHM content. It is never a
+ # permitted external URL because only explicitly allowed schemes are
+ # accepted by the exporter.
+ if canonical.startswith(("/", "//")) or _DRIVE.match(canonical):
+ return "path_escape", value
+ scheme = _SCHEME.match(canonical)
+ if scheme:
+ return ("external", value) if scheme.group(0)[:-1].casefold() in {
+ "http", "https", "mailto"
+ } else ("unsafe_uri", value)
+ return "local", value
+
+
+def _prefix_sibling(target: Path, expected: str) -> bool:
+ if not target.parent.is_dir():
+ return False
+ expected_folded = expected.casefold()
+ return any(
+ sibling.is_file()
+ and sibling.name.casefold() != expected_folded
+ and expected_folded.startswith(sibling.name.casefold())
+ for sibling in target.parent.iterdir()
+ )
+
+
+def validate(outdir: Path) -> tuple[list[str], list[str]]:
+ """Check an extraction, separating tool failure from content bugs.
+
+ Returns (fatal, advisory).
+
+ `fatal` means the extractor lost data -- a truncated corpus yields a subtly
+ wrong RAG index, which is worse than no index. `advisory` means the CHM
+ itself is inconsistent, e.g. the author renamed a topic and left a stale
+ TOC entry behind. That is real, but it is the doc author's to fix and must
+ not block a build.
+
+ Distinguishing them: hh.exe truncation leaves a sibling whose name is a
+ prefix of the expected one ("Discussion_field_(Extended_Note)" for
+ "...(Extended_Note).htm"). A genuinely absent topic leaves no such trace.
+ """
+ fatal, advisory = [], []
+
+ parsers: list[tuple[Path, _ReferenceParser]] = []
+ for path in sorted(outdir.rglob("*"), key=lambda p: p.relative_to(outdir).as_posix().casefold()):
+ if not path.is_file() or path.name == EXTRACTION_MANIFEST_NAME:
+ continue
+ if path.suffix.casefold() not in {".hhc", ".htm", ".html"}:
+ continue
+ parser = _ReferenceParser()
+ parser.feed(path.read_text(encoding="cp1252", errors="replace"))
+ parsers.append((path, parser))
+
+ if not any(path.suffix.casefold() == ".hhc" for path, _ in parsers):
+ fatal.append("no .hhc table of contents found in extraction")
+ return fatal, advisory
+
+ seen: set[tuple[str, str, str]] = set()
+ for source, parser in parsers:
+ references = [("toc", value) for value in parser.local] + parser.html
+ for kind, raw in references:
+ uri_kind, value = _reference_kind(raw)
+ if uri_kind == "fragment" or uri_kind == "external":
+ continue
+ if uri_kind in {"unsafe_uri", "path_escape"}:
+ key = (uri_kind, source.as_posix(), raw)
+ if key not in seen:
+ advisory.append(
+ f"source_{uri_kind}: {source.relative_to(outdir).as_posix()}: {raw}"
+ )
+ seen.add(key)
+ continue
+ target = (source.parent / value).resolve()
+ if target != outdir.resolve() and outdir.resolve() not in target.parents:
+ advisory.append(
+ f"source_path_escape: {source.relative_to(outdir).as_posix()}: {raw}"
+ )
+ continue
+ if target.exists():
+ continue
+ if _prefix_sibling(target, target.name):
+ message = f"{('TOC target' if kind == 'toc' else 'HTML target')} lost to filename truncation: {value}"
+ fatal.append(message)
+ elif kind == "toc":
+ advisory.append(f"TOC points at a topic that does not exist: {value}")
+ else:
+ advisory.append(f"HTML {kind} points at a target that does not exist: {value}")
+
+ return fatal, advisory
+
+
+def _extract_locked(chm: Path, outdir: Path, clean: bool = True, check: bool = True) -> str:
+ """Extract `chm` into `outdir`. Returns the name of the tool that worked."""
+ chm = Path(chm)
+ outdir = Path(outdir)
+ chm, outdir = _validate_extract_paths(chm, outdir)
+ # Keep failed attempts in private directories beside the destination. The
+ # staging owner guarantees that an existing destination is untouched until
+ # a validated extraction is promoted, and same-directory staging makes the
+ # final rename safe even when the system temporary directory is another
+ # volume.
+ errors: list[str] = []
+ extract.advisory = []
+ for backend in (_sevenzip, _chmlib, _hh):
+ try:
+ with OutputStaging(outdir) as staging:
+ tool = backend(chm, staging.path)
+ if not tool:
+ continue
+ validate_source_tree(staging.path)
+ if not any(
+ path.is_file() and path.suffix.lower() in {".htm", ".html"}
+ for path in staging.rglob("*")
+ ):
+ errors.append(f"{tool}: produced no .htm files")
+ continue
+ advisory: list[str] = []
+ if check:
+ fatal, advisory = validate(staging.path)
+ if fatal:
+ errors.append(
+ f"{tool} produced a corrupt/incomplete extraction "
+ f"({len(fatal)} problems): " + "; ".join(fatal)
+ )
+ continue
+ # Content-level inconsistencies ride along for the author
+ # report rather than failing the build.
+ if clean or not outdir.exists():
+ staging.promote()
+ else:
+ # Retain the historical clean=False merge semantics,
+ # but construct the merged tree privately so a copy
+ # failure cannot partially modify the destination.
+ validate_source_tree(outdir)
+ with OutputStaging(outdir) as merged:
+ shutil.copytree(outdir, merged.path, dirs_exist_ok=True)
+ shutil.copytree(staging.path, merged.path, dirs_exist_ok=True)
+ merged.promote()
+ extract.advisory = advisory
+ return tool
+ except subprocess.CalledProcessError as exc:
+ errors.append(f"{backend.__name__}: exit {exc.returncode}")
+ continue
+
+ raise ExtractError(
+ "no working CHM extractor found.\n"
+ " Linux: apt-get install p7zip-full (or libchm-bin)\n"
+ r" Windows: 7-Zip, or the built-in C:\Windows\hh.exe" "\n"
+ + (" tried: " + "; ".join(errors) if errors else "")
+ )
+
+
+def _extract_already_locked(
+ chm: Path, outdir: Path, clean: bool = True, check: bool = True
+) -> str:
+ """Internal extraction entry point for callers holding ``outdir``'s lock."""
+ return _extract_locked(chm, outdir, clean=clean, check=check)
+
+
+def extract(chm: Path, outdir: Path, clean: bool = True, check: bool = True) -> str:
+ """Extract into ``outdir`` while serializing overlapping exporters."""
+ chm, outdir = _validate_extract_paths(Path(chm), Path(outdir))
+ with ExportLock(outdir):
+ # Revalidate after acquiring the lock to close the preflight-to-write
+ # path/link window for cooperating invocations.
+ chm, outdir = _validate_extract_paths(chm, outdir)
+ return _extract_locked(chm, outdir, clean=clean, check=check)
+
+
+if __name__ == "__main__":
+ tool = extract(Path(sys.argv[1]), Path(sys.argv[2]))
+ print(f"extracted with {tool}")
+ for note in getattr(extract, "advisory", []):
+ print(f" advisory: {note}")
diff --git a/tools/chm_metadata.py b/tools/chm_metadata.py
new file mode 100644
index 00000000..c3d5e839
--- /dev/null
+++ b/tools/chm_metadata.py
@@ -0,0 +1,127 @@
+"""Small, repository-independent CHM metadata helpers."""
+
+from __future__ import annotations
+
+import html
+import re
+from html.parser import HTMLParser
+from pathlib import Path
+from urllib.parse import unquote, urldefrag
+
+
+def safe_stem(name: str) -> str:
+ """Return a deterministic namespace name for a CHM filename."""
+ stem = Path(name).stem
+ value = re.sub(r"[^A-Za-z0-9._-]+", "_", stem).strip("._-")
+ return value or "chm"
+
+
+class TopicMeta(HTMLParser):
+ """Facts needed by conversion and corpus provenance checks."""
+
+ def __init__(self) -> None:
+ super().__init__(convert_charrefs=True)
+ self.title = ""
+ self.meta: dict[str, str] = {}
+ self.links: list[str] = []
+ self.images: list[str] = []
+ self.related: list[tuple[str, str]] = []
+ self.page_heading = ""
+ self._in_title = False
+ self._heading: list[str] | None = None
+ self._section = ""
+ self._anchor: tuple[str, list[str]] | None = None
+
+ def handle_starttag(self, tag: str, attrs) -> None:
+ a = {k.lower(): (v or "") for k, v in attrs}
+ if tag == "title":
+ self._in_title = True
+ elif tag == "meta":
+ key = a.get("name") or a.get("http-equiv")
+ if key:
+ self.meta[key.lower()] = a.get("content", "")
+ elif re.fullmatch(r"h[1-6]", tag):
+ self._heading = []
+ elif tag == "img" and a.get("src"):
+ self.images.append(a["src"])
+ elif tag == "a" and a.get("href"):
+ href = a["href"]
+ self.links.append(href)
+ self._anchor = (href, [])
+
+ def handle_endtag(self, tag: str) -> None:
+ if tag == "title":
+ self._in_title = False
+ elif re.fullmatch(r"h[1-6]", tag) and self._heading is not None:
+ text = "".join(self._heading).strip()
+ if text and not self.page_heading:
+ self.page_heading = text
+ if text in {"Related Topics", "Related Internet Sites"}:
+ self._section = text
+ elif text:
+ self._section = ""
+ self._heading = None
+ elif tag == "a" and self._anchor is not None:
+ href, label = self._anchor
+ if self._section == "Related Topics":
+ self.related.append(("".join(label).strip(), href))
+ self._anchor = None
+
+ def handle_data(self, data: str) -> None:
+ if self._in_title:
+ self.title += data
+ if self._heading is not None:
+ self._heading.append(data)
+ if self._anchor is not None:
+ self._anchor[1].append(data)
+
+
+class _SitemapParser(HTMLParser):
+ def __init__(self) -> None:
+ super().__init__(convert_charrefs=True)
+ self.depth = 0
+ self.entries: list[dict] = []
+ self._current: list[tuple[str, str]] | None = None
+
+ def handle_starttag(self, tag: str, attrs) -> None:
+ a = {k.lower(): (v or "") for k, v in attrs}
+ if tag == "ul":
+ self.depth += 1
+ elif tag == "object":
+ self._current = []
+ elif tag == "param" and self._current is not None:
+ self._current.append((a.get("name", "").lower(), a.get("value", "")))
+
+ def handle_endtag(self, tag: str) -> None:
+ if tag == "ul":
+ self.depth = max(0, self.depth - 1)
+ elif tag == "object" and self._current:
+ self.entries.append({"depth": self.depth, "params": self._current})
+ self._current = None
+
+
+def parse_toc(path: Path) -> list[dict]:
+ parser = _SitemapParser()
+ parser.feed(path.read_text(encoding="cp1252", errors="replace"))
+ nodes, trail = [], {}
+ for entry in parser.entries:
+ params = dict(entry["params"])
+ title = html.unescape(params.get("name", "")).strip()
+ local = unquote(html.unescape(params.get("local", ""))).replace("\\", "/").strip()
+ href = urldefrag(local)[0] if local else ""
+ depth = entry["depth"]
+ trail[depth] = title
+ for d in list(trail):
+ if d > depth:
+ del trail[d]
+ nodes.append({
+ "title": title,
+ "href": href,
+ "depth": depth,
+ "breadcrumb": [trail[d] for d in sorted(trail) if trail[d]],
+ "is_container": not bool(href),
+ })
+ return nodes
+
+
+__all__ = ["TopicMeta", "parse_toc", "safe_stem"]
diff --git a/tools/convert.py b/tools/convert.py
new file mode 100644
index 00000000..8764d600
--- /dev/null
+++ b/tools/convert.py
@@ -0,0 +1,395 @@
+"""Thin orchestration seam for the CHM/PDF Markdown export."""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import re
+import tempfile
+from pathlib import Path
+from urllib.parse import quote
+
+import pdf_convert
+from chm_convert import run_in_private_stage
+from corpus_validation import validate_corpus
+from output_fs import ExportLock, OutputPathError, OutputStaging, validate_output_paths
+from reporting import Issue, Report, make_issue
+from source_safety import discover_source_files
+
+DEFAULT_SOURCE_REPO = "https://github.com/sillsdev/FwHelps"
+DEFAULT_CHM_SOURCE_URL_BASE = "https://downloads.languagetechnology.org/fieldworks/Documentation/en"
+
+
+def discover_chms(repo: Path) -> list[Path]:
+ """Discover all CHMs at the repository root in stable case-folded order."""
+ return discover_source_files(Path(repo), suffixes={".chm"}, recursive=False)
+
+
+def _report_issue(code: str, item: object, default_path: str = ""):
+ if code == "unmapped_span_classes" and isinstance(item, (list, tuple)):
+ class_name = str(item[0]) if item else ""
+ count = item[1] if len(item) > 1 else 0
+ topics = [str(path) for path in item[2]] if len(item) > 2 else []
+ path = topics[0] if topics else default_path
+ detail = {"class": class_name, "count": count, "topics": topics}
+ message = f"span class '{class_name}' reported {count} time(s)"
+ return make_issue(code, message, path, detail)
+ if code == "duplicate_titles" and isinstance(item, (list, tuple)):
+ title = str(item[0]) if item else ""
+ raw_topics = item[1] if len(item) > 1 else []
+ topics = [str(path) for path in raw_topics] if isinstance(raw_topics, list) else [str(raw_topics)]
+ path = topics[0] if topics else default_path
+ detail = {"title": title, "topics": topics}
+ message = f"duplicate title '{title}' appears in {len(topics)} topics"
+ return make_issue(code, message, path, detail)
+ path = item[0] if isinstance(item, (list, tuple)) and item else str(item)
+ message = item[1] if isinstance(item, (list, tuple)) and len(item) > 1 else str(item)
+ return make_issue(code, str(message), str(path or default_path), item)
+
+
+def _write_readme(stage: Path, chms: list[dict], pdf_count: int,
+ source_ref: str, report: Report) -> None:
+ inventory = (
+ "- **Root CHMs (auto-discovered):** "
+ + ", ".join(f"`{item['chm']}`" for item in chms)
+ if chms else "- **Root CHMs (auto-discovered):** none found"
+ )
+ lines = [
+ "# FieldWorks Help — Portable Markdown", "",
+ "Generated documentation corpus. **Do not edit these files**; the tree is replaced as a set.", "",
+ f"- **Source ref:** `{source_ref}`", f"- **CHMs:** {len(chms)} **PDFs:** {pdf_count}", "",
+ inventory,
+ "",
+ report.to_readme(), "",
+ (
+ "Full detail: [author-report.md](author-report.md) for authors; "
+ "[author-report.json](author-report.json) for automation."
+ ), "",
+ "## CHM navigation", "",
+ ]
+ for result in chms:
+ lines.append(f"### {result['chm']}")
+ known_topics = {
+ str(item).replace("\\", "/").casefold()
+ for item in result.get("topics_paths", [])
+ }
+ for node in result.get("toc", []):
+ if not node.get("title"):
+ continue
+ indent = " " * max(0, int(node.get("depth", 1)) - 1)
+ href = node.get("href", "")
+ topic_path = Path(href.split("#", 1)[0]).as_posix()
+ if href and topic_path.casefold() in known_topics:
+ target = quote(f"chm/{result['stem']}/{Path(href).with_suffix('.md').as_posix()}")
+ lines.append(f"{indent}- [{node['title']}]({target})")
+ else:
+ lines.append(f"{indent}- **{node['title']}**")
+ lines.extend(["", "## PDF navigation", ""])
+ pdf_root = stage / "pdf"
+ for path in sorted(pdf_root.rglob("*.md")) if pdf_root.exists() else []:
+ rel = path.relative_to(stage).as_posix()
+ lines.append(f"- [{path.stem.replace('_', ' ')}]({quote(rel)})")
+ (stage / "README.md").write_text("\n".join(lines) + "\n", encoding="utf-8")
+ (stage / ".nojekyll").write_text("", encoding="utf-8")
+
+
+def _build_locked(repo: Path, out: Path, work: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop",
+ source_repo: str = DEFAULT_SOURCE_REPO,
+ chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE) -> dict:
+ """Build a complete corpus in private staging and promote only if valid."""
+ repo, out, work = Path(repo).resolve(), Path(out).resolve(), Path(work).resolve()
+ report = Report()
+ advisory_links: set[tuple[str, str]] = set()
+ advisory_images: set[tuple[str, str]] = set()
+ source_replacement_paths: set[str] = set()
+ chm_results: list[dict] = []
+ chms = discover_chms(repo)
+ if not chms:
+ report.add(make_issue("chm_discovery", "no repository-root CHM files found"))
+
+ # Validate the extraction workspace independently. The publication stage
+ # must live beside the destination so promotion is one same-volume rename
+ # even when --work is on another drive.
+ validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo)
+ with OutputStaging(out, repo_root=repo, source_root=repo) as staging:
+ seen_names: set[str] = set()
+ for chm in chms:
+ from chm_metadata import safe_stem
+ stem = safe_stem(chm.name)
+ if stem.casefold() in seen_names:
+ report.add(make_issue("destination_collision", f"CHM namespace collision: {stem}", chm.name))
+ continue
+ seen_names.add(stem.casefold())
+ try:
+ result = run_in_private_stage(chm, work, staging.path / "chm" / stem,
+ reuse=reuse, limit=limit, source_ref=source_ref,
+ source_url_base=chm_source_url_base)
+ chm_results.append(result)
+ stem = result.get("stem", "")
+ for source_rel, raw_target in result.get("report", {}).get("broken_links", []):
+ output_rel = f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}"
+ advisory_links.add((output_rel, raw_target))
+ if "#" in raw_target:
+ source_target, fragment = raw_target.split("#", 1)
+ else:
+ source_target, fragment = raw_target, ""
+ if re.search(r"\.html?$", source_target, re.IGNORECASE):
+ source_target = re.sub(r"\.html?$", ".md", source_target, flags=re.IGNORECASE)
+ emitted_target = source_target + (f"#{fragment}" if fragment else "")
+ advisory_links.add((output_rel, emitted_target))
+ advisory_links.add((output_rel, quote(emitted_target, safe="/#()%")))
+ for source_rel, raw_target in result.get("report", {}).get("broken_images", []):
+ output_rel = f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}"
+ advisory_images.add((output_rel, raw_target))
+ advisory_images.add((output_rel, quote(raw_target, safe="/#()%")))
+ for source_rel in result.get("source_replacement_paths", []):
+ source_replacement_paths.add(
+ f"chm/{stem}/{Path(source_rel).with_suffix('.md').as_posix()}"
+ )
+ for code, items in result.get("report", {}).items():
+ for item in items:
+ report.add(_report_issue(code, item, chm.name))
+ except Exception as exc: # noqa: BLE001 - isolate one corrupt CHM
+ report.add(make_issue("chm_failure", f"{type(exc).__name__}: {exc}", chm.name))
+
+ pdf_url = f"{source_repo.rstrip('/')}/blob/{source_ref}/{{path}}"
+ try:
+ pdf_result, _ = pdf_convert.run_in_private_stage(
+ repo, staging.path / "pdf", update=False, source_url=pdf_url,
+ )
+ except Exception as exc: # noqa: BLE001 - isolate PDF backend failure
+ pdf_result = {"converted": 0, "report": {
+ "pdf_failures": [["", f"{type(exc).__name__}: {exc}"]]
+ }}
+ pdf_report = pdf_result.get("report", {})
+ pdf_export_paths = {
+ str(item[0]) for item in pdf_report.get("pdf_export_replacements", [])
+ if isinstance(item, (list, tuple)) and item
+ }
+ for item in pdf_report.get("pdf_source_replacements", []):
+ if not isinstance(item, (list, tuple)) or len(item) < 2:
+ continue
+ source_rel, details = item[0], item[1]
+ emitted = Path(pdf_convert.slug_path(str(source_rel))).with_suffix(".md").as_posix()
+ emitted_rel = f"pdf/{emitted}"
+ # A page reporting both source and exporter replacements must not
+ # be allowlisted: validator evidence must keep the exporter part
+ # fatal even though the source risk is also reported.
+ if str(source_rel) not in pdf_export_paths:
+ source_replacement_paths.add(emitted_rel)
+ report.add(make_issue(
+ "source_replacement_character",
+ f"source PDF replacement characters: {details}",
+ str(source_rel), {
+ "source_pdf": str(source_rel),
+ "generated_markdown": emitted_rel,
+ "pages": details,
+ },
+ ))
+ for code, items in pdf_report.items():
+ if code == "pdf_source_replacements":
+ continue
+ for item in items:
+ report.add(_report_issue(code, item))
+ report.metadata = {
+ "source_ref": source_ref,
+ "source_repo": source_repo.rstrip("/"),
+ "chms": [
+ {"name": item.get("chm", ""), "stem": item.get("stem", ""),
+ "version": item.get("version", ""), "topics": item.get("topics", 0),
+ "images": item.get("images", 0)}
+ for item in chm_results
+ ],
+ "chm_count": len(chm_results),
+ "topic_count": sum(item.get("topics", 0) for item in chm_results),
+ "image_count": sum(item.get("images", 0) for item in chm_results),
+ "pdf_count": pdf_result.get("converted", 0),
+ }
+ # README navigation is part of the emitted corpus. Seed the linked
+ # report target, render navigation, then validate all links before the
+ # final report/count rewrite.
+ (staging.path / "author-report.json").write_text("{}\n", encoding="utf-8")
+ (staging.path / "author-report.md").write_text(
+ "# Author quality report\n\nBuild validation is in progress.\n",
+ encoding="utf-8",
+ )
+ _write_readme(staging.path, chm_results, pdf_result.get("converted", 0), source_ref, report)
+ report.extend(validate_corpus(
+ staging.path,
+ advisory_links=advisory_links,
+ advisory_images=advisory_images,
+ source_replacement_paths=source_replacement_paths,
+ ))
+ _write_readme(staging.path, chm_results, pdf_result.get("converted", 0), source_ref, report)
+ (staging.path / "author-report.json").write_text(
+ report.to_json(), encoding="utf-8"
+ )
+ (staging.path / "author-report.md").write_text(
+ report.to_markdown(), encoding="utf-8"
+ )
+ if report.fatal:
+ return {"report": report.as_dict(), "chms": chm_results,
+ "pdfs": pdf_result.get("converted", 0), "promoted": False}
+ staging.promote()
+ return {"report": report.as_dict(), "chms": chm_results,
+ "pdfs": pdf_result.get("converted", 0), "promoted": True}
+
+
+def _build(repo: Path, out: Path, work: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop",
+ source_repo: str = DEFAULT_SOURCE_REPO,
+ chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE) -> dict:
+ """Serialize one complete corpus mutation for cooperating exporters."""
+ repo, out, work = Path(repo).resolve(), Path(out).resolve(), Path(work).resolve()
+ validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo)
+ with ExportLock(out):
+ # The destination/link chain is checked again after lock acquisition,
+ # immediately before staging and eventual promotion begin.
+ validate_output_paths(out, work_dir=work, repo_root=repo, source_root=repo)
+ return _build_locked(
+ repo, out, work, reuse=reuse, limit=limit, source_ref=source_ref,
+ source_repo=source_repo, chm_source_url_base=chm_source_url_base,
+ )
+
+
+def _path_overlaps(left: Path, right: Path) -> bool:
+ return left == right or left in right.parents or right in left.parents
+
+
+def _validate_diagnostics_path(
+ diagnostics: Path | str,
+ *,
+ repo: Path,
+ out: Path,
+ work: Path,
+) -> Path:
+ """Validate an external diagnostics destination before creating anything."""
+ lexical = Path(os.path.abspath(os.fspath(Path(diagnostics).expanduser())))
+ resolved = lexical.resolve(strict=False)
+ if resolved == Path(resolved.anchor):
+ raise OutputPathError(f"refusing filesystem root as diagnostics: {resolved}")
+ if lexical.is_symlink() or (lexical.exists() and not lexical.is_file()):
+ raise OutputPathError(f"refusing non-regular diagnostics path: {lexical}")
+ # Existing symlinked parents can redirect an apparently external path into
+ # a protected tree. Missing parents are created only after this check.
+ current = lexical.parent
+ while current != Path(current.anchor):
+ if current.is_symlink() or (
+ current.exists() and getattr(current, "is_junction", lambda: False)()
+ ):
+ raise OutputPathError(f"refusing symlink/junction in diagnostics parent: {current}")
+ current = current.parent
+ protected = {
+ "repository": Path(repo).resolve(strict=False),
+ "output": Path(out).resolve(strict=False),
+ "work": Path(work).resolve(strict=False),
+ }
+ for label, root in protected.items():
+ if _path_overlaps(resolved, root):
+ raise OutputPathError(f"diagnostics must not overlap {label}: {resolved}")
+ return resolved
+
+
+def _sanitize_diagnostic_value(value: object, roots: dict[str, Path]) -> object:
+ """Remove run-specific absolute paths while preserving report structure."""
+ if isinstance(value, dict):
+ return {key: _sanitize_diagnostic_value(item, roots) for key, item in value.items()}
+ if isinstance(value, list):
+ return [_sanitize_diagnostic_value(item, roots) for item in value]
+ if isinstance(value, tuple):
+ return [_sanitize_diagnostic_value(item, roots) for item in value]
+ if not isinstance(value, str):
+ return value
+ sanitized = value
+ for label, root in sorted(roots.items(), key=lambda item: len(str(item[1])), reverse=True):
+ for spelling in (str(root), str(root).replace("\\", "/")):
+ sanitized = sanitized.replace(spelling, f"<{label}>")
+ sanitized = re.sub(r"\.output-(?:stage|backup)-[0-9A-Za-z-]+", ".output-", sanitized)
+ return sanitized
+
+
+def _write_diagnostics(path: Path, report: dict, *, roots: dict[str, Path]) -> None:
+ """Atomically write a stable report without exposing staging filenames."""
+ path.parent.mkdir(parents=True, exist_ok=True)
+ stable_report = _sanitize_diagnostic_value(report, roots)
+ payload = json.dumps(stable_report, indent=2, ensure_ascii=False, sort_keys=True) + "\n"
+ fd, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent)
+ temporary_path = Path(temporary)
+ try:
+ with os.fdopen(fd, "w", encoding="utf-8", newline="\n") as handle:
+ handle.write(payload)
+ handle.flush()
+ os.fsync(handle.fileno())
+ os.replace(temporary_path, path)
+ finally:
+ if temporary_path.exists():
+ temporary_path.unlink()
+
+
+def build(repo: Path, out: Path, work: Path, *, reuse: bool = False,
+ limit: int = 0, source_ref: str = "develop",
+ source_repo: str = DEFAULT_SOURCE_REPO,
+ chm_source_url_base: str = DEFAULT_CHM_SOURCE_URL_BASE,
+ diagnostics: Path | str | None = None) -> dict:
+ """Build a corpus and optionally persist diagnostics outside generated trees."""
+ repo_path, out_path, work_path = (
+ Path(repo).resolve(), Path(out).resolve(), Path(work).resolve()
+ )
+ diagnostics_path = (
+ _validate_diagnostics_path(
+ diagnostics, repo=repo_path, out=out_path, work=work_path
+ )
+ if diagnostics is not None else None
+ )
+ try:
+ result = _build(
+ repo_path, out_path, work_path, reuse=reuse, limit=limit,
+ source_ref=source_ref, source_repo=source_repo,
+ chm_source_url_base=chm_source_url_base,
+ )
+ except Exception as exc:
+ if diagnostics_path is not None:
+ _write_diagnostics(
+ diagnostics_path,
+ Report([make_issue("unknown_issue", f"{type(exc).__name__}: {exc}")]).as_dict(),
+ roots={"repo": repo_path, "out": out_path, "work": work_path},
+ )
+ raise
+ if diagnostics_path is not None:
+ _write_diagnostics(
+ diagnostics_path,
+ result["report"],
+ roots={"repo": repo_path, "out": out_path, "work": work_path},
+ )
+ return result
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--repo", default=".", type=Path)
+ parser.add_argument("--out", default="out", type=Path)
+ parser.add_argument("--work", default=None, type=Path)
+ parser.add_argument("--reuse", action="store_true")
+ parser.add_argument("--limit", type=int, default=0)
+ parser.add_argument("--source-ref", default="develop")
+ parser.add_argument("--source-repo", default=DEFAULT_SOURCE_REPO)
+ parser.add_argument("--chm-source-url-base", default=DEFAULT_CHM_SOURCE_URL_BASE)
+ parser.add_argument("--diagnostics", default=None, type=Path)
+ args = parser.parse_args()
+ repo = args.repo.resolve()
+ work = (args.work or repo / ".chm-work").resolve()
+ result = build(repo, args.out.resolve(), work, reuse=args.reuse,
+ limit=args.limit, source_ref=args.source_ref,
+ source_repo=args.source_repo,
+ chm_source_url_base=args.chm_source_url_base,
+ diagnostics=args.diagnostics)
+ print(Report(Issue(**{key: value for key, value in item.items()
+ if key in {"code", "message", "path", "fatal", "provenance", "detail"}})
+ for item in result["report"].get("issues", [])).to_console())
+ return 1 if result["report"]["summary"]["fatal"] else 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/tools/corpus_validation.py b/tools/corpus_validation.py
new file mode 100644
index 00000000..b6117eee
--- /dev/null
+++ b/tools/corpus_validation.py
@@ -0,0 +1,260 @@
+"""Validate the files actually emitted by the portable Markdown exporter."""
+
+from __future__ import annotations
+
+import os
+import re
+from html.parser import HTMLParser
+from pathlib import Path
+from urllib.parse import unquote, urldefrag
+
+from reporting import Issue, make_issue
+
+_EXTERNAL = re.compile(r"^(?:[a-z][a-z0-9+.-]*:|//)", re.IGNORECASE)
+_SCHEME = re.compile(r"^[a-z][a-z0-9+.-]*:", re.IGNORECASE)
+_DRIVE = re.compile(r"^[a-z]:[\\/]")
+_URI_ASCII_SEPARATORS = re.compile(r"[\x00-\x20]")
+_FENCE = re.compile(r"^\s*(```|~~~)")
+
+
+class _RawHTML(HTMLParser):
+ def __init__(self) -> None:
+ super().__init__(convert_charrefs=True)
+ self.targets: list[tuple[str, str]] = []
+ self.tags: list[str] = []
+
+ def handle_starttag(self, tag: str, attrs) -> None:
+ self.tags.append(tag.lower())
+ values = {key.lower(): value or "" for key, value in attrs}
+ if tag.lower() == "a" and values.get("href"):
+ self.targets.append(("href", values["href"]))
+ if tag.lower() == "img" and values.get("src"):
+ self.targets.append(("src", values["src"]))
+
+
+def _frontmatter(text: str) -> dict[str, str]:
+ if not text.startswith("---"):
+ return {}
+ end = text.find("\n---", 3)
+ if end < 0:
+ return {}
+ fields = {}
+ for line in text[4:end].splitlines():
+ if ":" in line and not line.startswith(" "):
+ key, value = line.split(":", 1)
+ fields[key.strip()] = value.strip().strip('"')
+ return fields
+
+
+def _target_path(raw: str) -> str:
+ return urldefrag(unquote(raw.replace("\\", "/")))[0]
+
+
+def _is_external(raw: str) -> bool:
+ value = _target_path(raw)
+ if not value or value.startswith("#"):
+ return True
+ scheme = _SCHEME.match(value)
+ return bool(scheme and scheme.group(0)[:-1].casefold() in {"http", "https", "mailto"})
+
+
+def _uri_kind(raw: str) -> str:
+ value = _target_path(raw).replace("\\", "/")
+ canonical = _URI_ASCII_SEPARATORS.sub("", value)
+ if not canonical or canonical.startswith("#"):
+ return "fragment"
+ if canonical.startswith("/") or _DRIVE.match(canonical):
+ return "path_escape"
+ scheme = _SCHEME.match(canonical)
+ if scheme:
+ return "external" if scheme.group(0)[:-1].casefold() in {
+ "http", "https", "mailto"
+ } else "unsafe_uri"
+ return "local"
+
+
+def _resolve(root: Path, source: Path, raw: str) -> Path | None:
+ path = _target_path(raw)
+ if _uri_kind(raw) != "local" or not path:
+ return None
+ # Absolute paths are never local corpus references.
+ if path.startswith(("/", "\\")):
+ return None
+ # Normalize ".." lexically rather than with Path.resolve(): on Windows
+ # resolve() rewrites the path to the real on-disk case, so a link whose
+ # case does not match its target would look valid here and 404 on the
+ # case-sensitive host the corpus is published to.
+ return Path(os.path.normpath(source.parent / path))
+
+
+def _inside_root(root: Path, candidate: Path) -> bool:
+ return candidate == root or root in candidate.parents
+
+
+def _corpus_entries(root: Path) -> set[str]:
+ """Return every corpus path, in the exact case it was emitted with."""
+ return {item.relative_to(root).as_posix() for item in root.rglob("*")}
+
+
+def _exists(root: Path, entries: set[str], candidate: Path) -> bool:
+ """Return whether candidate exists, matching case on every platform."""
+ if candidate == root:
+ return True
+ try:
+ relative = candidate.relative_to(root).as_posix()
+ except ValueError:
+ return False
+ return relative in entries
+
+
+def _markdown_targets(text: str):
+ """Yield (is_image, target) while balancing parentheses in destinations."""
+ start = re.compile(r"(?!)?\[[^\]]*\]\(")
+ for match in start.finditer(text):
+ cursor, depth = match.end(), 1
+ angle = cursor < len(text) and text[cursor] == "<"
+ while cursor < len(text):
+ char = text[cursor]
+ if angle and char == ">":
+ angle = False
+ elif not angle and char == "(":
+ depth += 1
+ elif not angle and char == ")":
+ depth -= 1
+ if depth == 0:
+ break
+ cursor += 1
+ if depth:
+ continue
+ value = text[match.end():cursor].strip()
+ if value.startswith("<") and ">" in value:
+ value = value[1:value.find(">")]
+ else:
+ value = value.split(None, 1)[0] if value else ""
+ yield bool(match.group("image")), value
+
+
+def _without_fenced(text: str) -> str:
+ lines, fenced = [], False
+ for line in text.splitlines():
+ if _FENCE.match(line):
+ fenced = not fenced
+ continue
+ if not fenced:
+ lines.append(line)
+ return "\n".join(lines)
+
+
+def validate_corpus(root: Path, *, advisory_links: set[tuple[str, str]] | None = None,
+ source_replacement_paths: set[str] | None = None,
+ source_links: set[tuple[str, str]] | None = None,
+ advisory_images: set[tuple[str, str]] | None = None) -> list[Issue]:
+ """Return issues for a corpus tree; never mutate the emitted files."""
+ root = Path(root).resolve()
+ advisory_links = advisory_links or source_links or set()
+ advisory_images = advisory_images or set()
+ source_replacement_paths = source_replacement_paths or set()
+ issues: list[Issue] = []
+ entries = _corpus_entries(root)
+ markdown = sorted(root.rglob("*.md"))
+ titles: dict[str, list[str]] = {}
+ for path in markdown:
+ rel = path.relative_to(root).as_posix()
+ text = path.read_text(encoding="utf-8", errors="replace")
+ body = text
+ if text.startswith("---") and "\n---" in text[3:]:
+ body = text[text.find("\n---", 3) + 4:]
+ link_body = _without_fenced(body)
+ h1s = []
+ fenced = False
+ for line in body.splitlines():
+ if _FENCE.match(line):
+ fenced = not fenced
+ continue
+ if fenced:
+ continue
+ heading = re.match(r"^\s*(#{1,6})\s+(.+?)\s*#*\s*$", line)
+ if heading and len(heading.group(1)) == 1:
+ h1s.append(heading.group(2).strip())
+ if len(h1s) != 1:
+ issues.append(make_issue("one_h1", f"expected one H1, found {len(h1s)}", rel))
+ if "\ufffd" in text:
+ source = rel in source_replacement_paths
+ issues.append(make_issue(
+ "source_replacement_character" if source else "replacement_character",
+ "contains U+FFFD", rel,
+ ))
+ if re.search(r"(?m)^\s*-\s+-\s+", link_body):
+ issues.append(make_issue("malformed_list", "literal '- -' list marker", rel))
+ if h1s:
+ titles.setdefault(h1s[0].casefold(), []).append(rel)
+
+ # Parse Markdown links separately so image links do not produce a
+ # second ordinary-link finding.
+ for is_image, raw in _markdown_targets(link_body):
+ uri_kind = _uri_kind(raw)
+ if uri_kind in {"unsafe_uri", "path_escape"}:
+ issues.append(make_issue(
+ uri_kind, f"unsafe target: {raw}", rel
+ ))
+ continue
+ target = _resolve(root, path, raw)
+ if target is not None:
+ inside = _inside_root(root, target)
+ if not inside or not _exists(root, entries, target):
+ # Source-authored missing targets are visible but advisory;
+ # an escape is always an exporter safety error.
+ advisory = inside and (
+ (not is_image and (rel, raw) in advisory_links)
+ or (is_image and (rel, raw) in advisory_images)
+ )
+ code = (("source_missing_image" if is_image else "source_missing_link")
+ if advisory else ("missing_image" if is_image else "missing_link"))
+ issues.append(make_issue(
+ code,
+ f"target {'escapes corpus root' if not inside else 'does not exist'}: {raw}",
+ rel,
+ ))
+
+ raw_html = _RawHTML()
+ raw_html.feed(link_body)
+ if any(tag not in {"a", "img"} for tag in raw_html.tags):
+ issues.append(make_issue(
+ "raw_html", f"raw HTML tags: {', '.join(sorted(set(raw_html.tags)))}",
+ rel,
+ ))
+ for kind, raw in raw_html.targets:
+ uri_kind = _uri_kind(raw)
+ if uri_kind in {"unsafe_uri", "path_escape"}:
+ issues.append(make_issue(
+ uri_kind, f"unsafe target: {raw}", rel
+ ))
+ continue
+ target = _resolve(root, path, raw)
+ if target is not None and (not _exists(root, entries, target)
+ or not _inside_root(root, target)):
+ inside = _inside_root(root, target)
+ advisory = inside and (
+ (kind == "href" and (rel, raw) in advisory_links)
+ or (kind == "src" and (rel, raw) in advisory_images)
+ )
+ code = (("source_missing_image" if kind == "src" else "source_missing_link")
+ if advisory else ("missing_image" if kind == "src" else "missing_link"))
+ issues.append(make_issue(
+ code,
+ f"raw HTML target {'escapes corpus root' if not inside else 'does not exist'}: {raw}",
+ rel,
+ ))
+
+ for title, paths in sorted(titles.items()):
+ if len(paths) > 1:
+ issues.append(make_issue(
+ "duplicate_title", f"display title {title!r}: {', '.join(paths)}",
+ paths[0], paths,
+ ))
+ return issues
+
+
+validate_emitted_corpus = validate_corpus
+
+__all__ = ["validate_corpus", "validate_emitted_corpus"]
diff --git a/tools/frontmatter.py b/tools/frontmatter.py
new file mode 100644
index 00000000..d9d0e157
--- /dev/null
+++ b/tools/frontmatter.py
@@ -0,0 +1,13 @@
+"""Shared helpers for safely rendering YAML front matter values."""
+
+from __future__ import annotations
+
+import json
+
+
+def yaml_scalar(value: object) -> str:
+ """Render a scalar as a JSON-quoted YAML-compatible string."""
+ return json.dumps(str(value), ensure_ascii=False)
+
+
+__all__ = ["yaml_scalar"]
diff --git a/tools/fwhelp.lua b/tools/fwhelp.lua
new file mode 100644
index 00000000..8dc7ee2b
--- /dev/null
+++ b/tools/fwhelp.lua
@@ -0,0 +1,285 @@
+-- Pandoc Lua filter: RoboHelp XHTML -> clean GFM for humans and RAG.
+--
+-- Runs inside pandoc (Lua 5.4 is bundled; no separate install). Everything here
+-- needs the document tree, which is why it lives in Lua rather than Python.
+--
+-- Unmapped span classes are collected and reported so the build can fail loudly
+-- rather than silently dropping a semantic distinction we did not know about.
+
+local unmapped = {}
+
+-- RoboHelp's authored character styles. Anything not listed is a build error.
+local SPAN_MAP = {
+ UserInterface = "strong", -- 18708x: menu/button/field names
+ Strong = "strong",
+ Emphasis = "emph",
+ DefinedWord = "emph",
+ BookTitle = "emph",
+ VernacularWord = "emph", -- example-language data, NOT English prose
+ TypedText = "code", -- literal text the user types
+ Keyboard = "code", -- key names
+ FileName = "code",
+ Filename = "code", -- authoring typo, same intent
+ Placeholder = "emph",
+ -- RoboHelp table presentation classes carry no document semantics.
+ hcp1 = "plain",
+ hcp2 = "plain",
+ hcp3 = "plain",
+ hcp4 = "plain",
+ Superscript = "superscript",
+ nobr = "plain",
+ expandtext = "plain",
+ ["Strong\""] = "strong", -- malformed class attribute in source
+}
+
+-- h4 callout classes -> GitHub alert blocks
+local ALERT_MAP = {
+ Note = "NOTE", Tip = "TIP", Important = "IMPORTANT",
+ Warning = "WARNING", Caution = "CAUTION",
+}
+
+local function uri_kind(target)
+ local path = tostring(target or "")
+ path = path:gsub("%%([0-9A-Fa-f][0-9A-Fa-f])", function(hex)
+ return string.char(tonumber(hex, 16))
+ end)
+ path = path:gsub("\\", "/")
+ path = path:gsub("[%z\001-\032]", "")
+ path = path:gsub("#.*$", "")
+ if path == "" then return "fragment" end
+ if path:sub(1, 1) == "/" or path:match("^[A-Za-z]:/") then return "path_escape" end
+ local scheme = path:match("^([A-Za-z][A-Za-z0-9+%%.-]*):")
+ if scheme then
+ scheme = scheme:lower()
+ if scheme == "http" or scheme == "https" or scheme == "mailto" then return "external" end
+ return "unsafe_uri"
+ end
+ return "local"
+end
+
+local function neutralize_uri(target)
+ local kind = uri_kind(target)
+ if kind == "unsafe_uri" or kind == "path_escape" then
+ io.stderr:write("FWHELP_" .. kind:upper() .. " " .. tostring(target) .. "\n")
+ return "#"
+ end
+ return target
+end
+
+local function text_of(inlines)
+ -- gsub returns (string, count); returning it directly would pass the count
+ -- as pandoc.Code's second argument, which it reads as an Attr.
+ local s = pandoc.utils.stringify(inlines)
+ s = s:gsub("^%s+", "")
+ s = s:gsub("%s+$", "")
+ return s
+end
+
+--- Collapse authored character styles into real markdown emphasis.
+function Span(el)
+ if #el.classes == 0 then return el.content end
+ -- Inventory every authored class before applying a supported transform.
+ -- A span may carry both a supported class and an unknown semantic.
+ for _, cls in ipairs(el.classes) do
+ if SPAN_MAP[cls] == nil then
+ unmapped[cls] = (unmapped[cls] or 0) + 1
+ end
+ end
+ for _, cls in ipairs(el.classes) do
+ local kind = SPAN_MAP[cls]
+ if kind == "strong" then return pandoc.Strong(el.content)
+ elseif kind == "emph" then return pandoc.Emph(el.content)
+ elseif kind == "code" then return pandoc.Code(text_of(el.content))
+ elseif kind == "superscript" then return pandoc.Superscript(el.content)
+ elseif kind == "plain" then return el.content
+ end
+ end
+ return el.content
+end
+
+--- Repoint internal topic links at their .md counterparts.
+--- Done here rather than with a regex over the rendered markdown because
+--- filenames like "Export_full_lexicon_(LIFT).htm" contain parentheses, which
+--- no sane link regex survives. The AST has the target as a plain string.
+function Link(el)
+ local t = el.target
+ local kind = uri_kind(t)
+ if kind == "unsafe_uri" or kind == "path_escape" then
+ el.target = neutralize_uri(t)
+ return el
+ end
+ if kind == "fragment" or kind == "external" then return el end
+ local path, frag = t:match("^([^#]*)(.*)$")
+ local rewritten, n = path:gsub("%.html?$", ".md")
+ if n > 0 then el.target = rewritten .. frag end
+ return el
+end
+
+--- Drop the presentational attributes RoboHelp puts on every image.
+--- GFM cannot express width/height/style, so pandoc falls back to a raw
+--- tag and 2,109 of them were surviving into the markdown across 842 files.
+--- The sizes are RoboHelp's inline-icon dimensions, not information.
+function Image(el)
+ -- Decorative RoboHelp marker on Tip/Note headings. Empty-alt images
+ -- stringify as U+FFFD in alert detection and GFM output.
+ local src = tostring(el.src or "")
+ if src == "" and el.target then src = tostring(el.target) end
+ if src:lower():match("note[_-]?icon%.gif") then
+ return pandoc.List()
+ end
+ el.src = neutralize_uri(src)
+ el.attr = pandoc.Attr()
+ return el
+end
+
+--- Collect every row across head/bodies/foot as a flat list.
+local function all_rows(el)
+ local rows = pandoc.List()
+ for _, r in ipairs(el.head.rows) do rows:insert(r) end
+ for _, b in ipairs(el.bodies) do
+ for _, r in ipairs(b.head) do rows:insert(r) end
+ for _, r in ipairs(b.body) do rows:insert(r) end
+ end
+ for _, r in ipairs(el.foot.rows) do rows:insert(r) end
+ return rows
+end
+
+--- Is this a 2-column "label:/value" layout rather than real tabular data?
+--- 560 of 769 tables in the corpus are these -- "Full name:", "Location:",
+--- "Description:", "Field type:" -- i.e. definition lists that RoboHelp
+--- happened to render with
. 79% of tables carry block content in a
+--- cell, so pandoc must fall back to raw HTML for them; turning them into
+--- prose removes almost all remaining HTML from the output.
+local function is_definition_table(rows)
+ if #rows == 0 then return false end
+ local labelish = 0
+ for _, row in ipairs(rows) do
+ if #row.cells ~= 2 then return false end
+ local label = text_of(row.cells[1].contents)
+ if label:sub(-1) == ":" and #label < 40 then labelish = labelish + 1 end
+ end
+ return labelish / #rows >= 0.8
+end
+
+function Table(el)
+ local rows = all_rows(el)
+
+ if is_definition_table(rows) then
+ local out = pandoc.List()
+ for _, row in ipairs(rows) do
+ local label = text_of(row.cells[1].contents):gsub(":%s*$", "")
+ local value = row.cells[2].contents
+ local head = pandoc.List({ pandoc.Strong(pandoc.Str(label .. ":")) })
+ -- Fold a single-paragraph value onto the label line; otherwise keep the
+ -- label on its own line and let lists/multiple paragraphs follow.
+ if #value == 1 and value[1].t == "Para" then
+ head:insert(pandoc.Space())
+ head:extend(value[1].content)
+ out:insert(pandoc.Para(head))
+ else
+ out:insert(pandoc.Para(head))
+ out:extend(value)
+ end
+ end
+ return out
+ end
+
+ -- Genuine data table: drop RoboHelp's fixed column widths so it emits as a
+ -- GFM pipe table instead of a grid table (11,670 inline width: styles).
+ for i, spec in ipairs(el.colspecs) do
+ el.colspecs[i] = { spec[1], nil }
+ end
+ return el
+end
+
+--- Strip presentational wrappers pandoc lifts from RoboHelp's
/.
+function Div(el)
+ -- el.attributes is an AttributeList, not a plain table, so `next` fails on it.
+ if #el.classes == 0 and el.identifier == "" then return el.content end
+ return el
+end
+
+--- Every topic opens with an
title, which Python promotes to the page's
+--- single h1. Shift body headings up to match, leaving the 4 stray h1s alone.
+function Header(el)
+ if el.level > 1 then el.level = el.level - 1 end
+ return el
+end
+
+local function alert_kind(block)
+ if block.t ~= "Header" then return nil end
+ for _, cls in ipairs(block.classes) do
+ if ALERT_MAP[cls] then return ALERT_MAP[cls] end
+ end
+ -- Some callouts carry the word as heading text with no class.
+ local t = text_of(block.content):gsub("^%s*[^%w]*%s*", "")
+ return ALERT_MAP[t]
+end
+
+--- Wrap Note/Tip/Important/Warning/Caution callouts in GitHub alert
+--- blockquotes. This has to happen here rather than in Header() because the
+--- callout body is the *following* blocks, not the heading's children.
+function Blocks(blocks)
+ local out = pandoc.List()
+ local i = 1
+ while i <= #blocks do
+ local b = blocks[i]
+
+ -- The "Related Topics" / "Related Internet Sites" trailers stay where the
+ -- author put them. Reconstructing them from link labels alone lost the
+ -- prose between the links -- "Lists overview (task helps)" became "Lists
+ -- overview", and "Choose a translation type (in an Example Lexicon Edit)"
+ -- lost both its qualifier and its second link. Python only normalises the
+ -- heading level afterwards.
+ local kind = alert_kind(b)
+ if kind then
+ -- Raw, not Str: pandoc would escape the brackets to "\[!NOTE\]", which
+ -- GitHub no longer recognises as an alert.
+ local marker = pandoc.RawInline("gfm", "[!" .. kind .. "]")
+ local body = pandoc.List({ pandoc.Para({ marker }) })
+ local level = b.level
+ i = i + 1
+ while i <= #blocks and not (blocks[i].t == "Header" and blocks[i].level <= level) do
+ body:insert(blocks[i])
+ i = i + 1
+ end
+ out:insert(pandoc.BlockQuote(body))
+ goto continue
+ end
+
+ -- A handful of RoboHelp list fragments arrive as literal text beginning
+ -- "- -" instead of a nested list node. Reparse only that malformed
+ -- marker through Pandoc's Markdown reader so the item remains content but
+ -- is emitted as a real nested list (never a literal marker).
+ if b.t == "Para" and text_of(b.content):match("^%s*%-%s+%-%s+") then
+ local repaired = pandoc.read(text_of(b.content), "markdown")
+ for _, item in ipairs(repaired.blocks) do out:insert(item) end
+ else
+ out:insert(b)
+ end
+ i = i + 1
+ ::continue::
+ end
+ return out
+end
+
+--- Emit unmapped classes on stderr for the build to pick up.
+function Pandoc(doc)
+ local names = {}
+ for cls, n in pairs(unmapped) do names[#names + 1] = cls .. "=" .. n end
+ if #names > 0 then
+ io.stderr:write("FWHELP_UNMAPPED_SPAN " .. table.concat(names, ",") .. "\n")
+ end
+ return doc
+end
+
+-- Only the functions named here run: returning an explicit filter list opts out
+-- of pandoc's pick-up-every-global behaviour, so a handler left off this table
+-- is silently dead code.
+-- Span/Table/Div/Header run before Blocks so trailers are detected on clean text.
+return {
+ { Span = Span, Table = Table, Div = Div, Header = Header, Link = Link,
+ Image = Image },
+ { Blocks = Blocks },
+ { Pandoc = Pandoc },
+}
diff --git a/tools/issue_catalog.py b/tools/issue_catalog.py
new file mode 100644
index 00000000..620234ce
--- /dev/null
+++ b/tools/issue_catalog.py
@@ -0,0 +1,110 @@
+"""The single issue vocabulary shared by every exporter stage."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+from types import MappingProxyType
+
+
+@dataclass(frozen=True)
+class IssuePolicy:
+ label: str
+ fatal: bool
+ provenance: str
+ guidance: str = ""
+
+
+def _policy(label: str, fatal: bool, provenance: str, guidance: str) -> IssuePolicy:
+ return IssuePolicy(label, fatal, provenance, guidance)
+
+
+ISSUE_CATALOG = MappingProxyType({
+ "missing_link": _policy("Missing local link", True, "exporter",
+ "Do not edit RoboHelp yet. Inspect the generated source path and target, then fix the exporter so a valid authored link remains valid."),
+ "source_missing_link": _policy("Missing local link", False, "source",
+ "Open the source topic in RoboHelp, find the hyperlink named in Evidence, and retarget or remove it; then rebuild the CHM."),
+ "missing_image": _policy("Missing local image", True, "exporter",
+ "Do not edit RoboHelp yet. Verify the source image exists, then fix exporter copying or link rewriting for the reported generated path."),
+ "source_missing_image": _policy("Missing local image", False, "source",
+ "Open the source topic in RoboHelp, find the image reference named in Evidence, and restore, retarget, or remove it; then rebuild the CHM."),
+ "source_link_case": _policy("Link case mismatch", False, "source",
+ "RoboHelp resolves links case-insensitively, so this one opens in the CHM but would 404 on a case-sensitive host. The export publishes the corrected case; open the source topic in RoboHelp and retype the hyperlink to match the target topic's real path so the two stay in step."),
+ "duplicate_title": _policy("Duplicate display title", False, "source",
+ "Open the listed topics in RoboHelp and give each page a distinct, descriptive title or heading so search results identify the correct page."),
+ "malformed_list": _policy("Malformed nested list", True, "exporter",
+ "Inspect the reported generated page and its source topic, then correct list conversion while preserving the authored nesting."),
+ "replacement_character": _policy("Replacement character", True, "exporter",
+ "Compare the generated page with its CHM/PDF source and fix decoding or conversion where the replacement character was introduced."),
+ "source_replacement_character": _policy("Replacement character", False, "source",
+ "Open the reported source topic or PDF at the page in Evidence and replace the invalid or unsupported source character."),
+ "raw_html": _policy("Raw HTML retained", False, "source",
+ "Inspect the reported RoboHelp topic or PDF content and the tag names in Problem. Simplify unsupported source markup when practical; otherwise confirm the retained HTML is intentional."),
+ "one_h1": _policy("Invalid H1 count", True, "exporter",
+ "Compare the generated page headings with the source topic and fix title normalization so the Markdown has exactly one H1."),
+ "destination_collision": _policy("Destination collision", True, "exporter",
+ "Rename colliding source files/topics or adjust deterministic destination naming so every source maps to one unique output path."),
+ "unsafe_uri": _policy("Unsafe URI", True, "exporter",
+ "Inspect the generated path and URI in Evidence, then fix sanitization so unsafe source targets cannot be emitted as active links."),
+ "path_escape": _policy("Local path escape", True, "exporter",
+ "Inspect the generated link or image target and fix path normalization so it cannot resolve outside the published corpus."),
+ "source_unsafe_uri": _policy("Source unsafe URI", False, "source",
+ "Open the source topic in RoboHelp, find the URI in Evidence, and replace malformed, file:, or script-like targets with a valid safe link or remove them."),
+ "source_path_escape": _policy("Source local path escape", False, "source",
+ "Open the source topic in RoboHelp and retarget the local link or image so it stays inside the help project."),
+ "pandoc_failure": _policy("Pandoc conversion failure", True, "exporter",
+ "Reproduce conversion for the reported source topic, inspect the Pandoc error in Problem, and fix the exporter or unsupported source markup."),
+ "unmapped_span": _policy("Unmapped span class", True, "exporter",
+ "Find the reported CSS class in RoboHelp source, decide its intended semantics, and add an explicit exporter mapping before publishing."),
+ "pdf_failure": _policy("PDF conversion failure", True, "exporter",
+ "Open the reported PDF to confirm it is readable, then reproduce the error in Problem and fix the PDF conversion path."),
+ "outline_drift": _policy("PDF outline drift", True, "exporter",
+ "Review the reported PDF headings against the document, then either fix heading inference or deliberately repin the approved outline."),
+ "outline_unpinned": _policy("PDF outline unpinned", True, "exporter",
+ "Review the generated headings for the reported PDF and deliberately add its approved outline to pdf_outlines.json."),
+ "stale_toc_entries": _policy("Stale TOC entry", False, "source",
+ "Open the RoboHelp table of contents, locate the target in Evidence, and retarget or remove the entry before rebuilding the CHM."),
+ "not_in_toc": _policy("Topic missing from TOC", False, "source",
+ "Find the source topic path in RoboHelp and either add it to the appropriate table-of-contents location or remove the orphan topic."),
+ "chm_failure": _policy("CHM conversion failure", True, "exporter",
+ "Verify the named CHM opens normally, then reproduce the extraction/conversion error in Problem and correct the failing backend or source package."),
+ "chm_discovery": _policy("CHM discovery failure", True, "exporter",
+ "Place the intended CHM files at the repository root or correct discovery configuration, then rerun the exporter."),
+ "unknown_issue": _policy("Unknown issue", True, "exporter",
+ "Add the producer code shown in Problem to the canonical issue catalog with explicit severity, provenance, and repair guidance."),
+})
+
+
+# Producer report keys are deliberately aliases, so their severity and
+# provenance are still selected by ISSUE_CATALOG rather than local mappings.
+ISSUE_ALIASES = MappingProxyType({
+ "broken_links": "source_missing_link",
+ "broken_images": "source_missing_image",
+ "duplicate_titles": "duplicate_title",
+ "link_case_mismatches": "source_link_case",
+ "pandoc_failures": "pandoc_failure",
+ "unmapped_span_classes": "unmapped_span",
+ "destination_collisions": "destination_collision",
+ "source_unsafe_uris": "source_unsafe_uri",
+ "source_path_escapes": "source_path_escape",
+ "pdf_failures": "pdf_failure",
+ "html_tables_kept": "raw_html",
+ "pdf_export_replacements": "replacement_character",
+ "pdf_source_replacements": "source_replacement_character",
+})
+
+
+def canonical_code(code: str) -> str:
+ """Return the catalog code for a producer report or issue code."""
+ return ISSUE_ALIASES.get(code, code)
+
+
+def policy_for(code: str) -> tuple[str, IssuePolicy]:
+ """Return a safe catalog code and policy, including unknown integration codes."""
+ canonical = canonical_code(str(code))
+ policy = ISSUE_CATALOG.get(canonical)
+ if policy is None:
+ return "unknown_issue", ISSUE_CATALOG["unknown_issue"]
+ return canonical, policy
+
+
+__all__ = ["ISSUE_ALIASES", "ISSUE_CATALOG", "IssuePolicy", "canonical_code", "policy_for"]
diff --git a/tools/output_fs.py b/tools/output_fs.py
new file mode 100644
index 00000000..6c4ffbd4
--- /dev/null
+++ b/tools/output_fs.py
@@ -0,0 +1,349 @@
+"""Owned staging and conservative filesystem promotion for generated trees.
+
+The public surface deliberately has no general-purpose delete operation. A
+``OutputStaging`` instance may remove only its own temporary tree and the
+destination tree that it validated before promotion.
+"""
+
+from __future__ import annotations
+
+import errno
+import hashlib
+import os
+import shutil
+import tempfile
+import uuid
+from collections.abc import Iterator
+from contextlib import contextmanager
+from pathlib import Path
+from typing import Self
+
+if os.name == "nt":
+ import msvcrt
+else:
+ import fcntl
+
+
+class OutputPathError(ValueError):
+ """A requested generated path is unsafe or internally inconsistent."""
+
+
+class ExportBusyError(OutputPathError):
+ """Another cooperating exporter currently owns the destination lock."""
+
+
+def _lexical_absolute(path: Path | str) -> Path:
+ """Make an absolute, normalized path without following links."""
+ return Path(os.path.abspath(os.fspath(Path(path).expanduser())))
+
+
+def _absolute(path: Path | str) -> Path:
+ return _lexical_absolute(path).resolve(strict=False)
+
+
+def _first_link(path: Path) -> Path | None:
+ """Return the first symlink/junction in a lexical path, if any."""
+ current = Path(path.anchor)
+ for component in path.parts[1:]:
+ current /= component
+ is_junction = getattr(current, "is_junction", lambda: False)
+ if current.is_symlink() or is_junction():
+ return current
+ return None
+
+
+def _is_root(path: Path) -> bool:
+ return path == Path(path.anchor)
+
+
+def _overlaps(left: Path, right: Path) -> bool:
+ return left == right or left in right.parents or right in left.parents
+
+
+def validate_output_paths(
+ destination: Path | str,
+ *,
+ work_dir: Path | str | None = None,
+ repo_root: Path | str | None = None,
+ source_root: Path | str | None = None,
+) -> tuple[Path, Path]:
+ """Validate generated paths before staging or recursive removal.
+
+ Repository and source roots are protected exact paths. Their children are
+ valid outputs (the normal CLI writes ``repo/out``), while trying to replace
+ either root itself is rejected. ``work_dir`` and ``destination`` may not
+ overlap because either relationship could make a promotion consume its
+ own input tree.
+ """
+
+ destination_lexical = _lexical_absolute(destination)
+ destination_path = destination_lexical.resolve(strict=False)
+ explicit_work = work_dir is not None
+ work_lexical = _lexical_absolute(work_dir) if explicit_work else destination_lexical.parent
+ work_path = work_lexical.resolve(strict=False)
+ protected = {
+ label: (_lexical_absolute(value), _absolute(value))
+ for label, value in (("repository", repo_root), ("source", source_root))
+ if value is not None
+ }
+
+ for label, lexical, path in (
+ ("destination", destination_lexical, destination_path),
+ ("work", work_lexical, work_path),
+ ):
+ if _is_root(path):
+ raise OutputPathError(f"refusing filesystem root as {label}: {path}")
+ if (link := _first_link(lexical)) is not None:
+ raise OutputPathError(f"refusing symlink/junction in {label}: {link}")
+
+ for label, (protected_lexical, protected_path) in protected.items():
+ if _is_root(protected_path):
+ raise OutputPathError(f"refusing filesystem root as {label} root: {protected_path}")
+ if (link := _first_link(protected_lexical)) is not None:
+ raise OutputPathError(f"refusing symlink/junction in {label} root: {link}")
+ if destination_path == protected_path:
+ raise OutputPathError(f"destination must not replace {label} root: {destination_path}")
+ if explicit_work and work_path == protected_path:
+ raise OutputPathError(f"work directory must not be {label} root: {work_path}")
+
+ if explicit_work and _overlaps(destination_path, work_path):
+ raise OutputPathError(
+ "work and output paths must not overlap: "
+ f"work={work_path}, output={destination_path}"
+ )
+ return destination_path, work_path
+
+
+class ExportLock:
+ """A non-blocking, cooperative process lock for one output destination.
+
+ The lock is advisory: it serializes exporter invocations that use this
+ class, but cannot stop a hostile process that ignores OS file locks. The
+ lock file is a deterministic sibling derived from the normalized
+ destination path. It is intentionally never deleted, so an abandoned
+ (stale) lock file does not block a later acquisition.
+
+ ``ExportLock`` is non-reentrant. Callers should acquire it once at their
+ mutation boundary and pass through to lower-level helpers without taking
+ another lock for the same destination.
+ """
+
+ _LOCK_PREFIX = ".fwhelps-export-"
+ _LOCK_SUFFIX = ".lock"
+
+ def __init__(self, destination: Path | str) -> None:
+ self.destination = _lexical_absolute(destination)
+ self.lock_path = self._lock_path(self.destination)
+ self._fd: int | None = None
+
+ @classmethod
+ def _lock_path(cls, destination: Path) -> Path:
+ normalized = os.path.normcase(os.path.normpath(os.fspath(destination)))
+ digest = hashlib.sha256(os.fsencode(normalized)).hexdigest()
+ return destination.parent / f"{cls._LOCK_PREFIX}{digest}{cls._LOCK_SUFFIX}"
+
+ @staticmethod
+ def _validate_chain(destination: Path, lock_path: Path) -> None:
+ if _is_root(destination):
+ raise OutputPathError(f"refusing filesystem root as export destination: {destination}")
+ if (link := _first_link(destination)) is not None:
+ raise OutputPathError(
+ f"refusing symlink/junction in export destination: {link}"
+ )
+
+ # Missing destination parents are safe to create only when every
+ # existing ancestor is a real directory and has no link component.
+ current = destination.parent
+ while current != Path(current.anchor):
+ if (link := _first_link(current)) is not None:
+ raise OutputPathError(f"refusing symlink/junction in export lock parent: {link}")
+ if current.exists() and not current.is_dir():
+ raise OutputPathError(f"export lock parent is not a directory: {current}")
+ current = current.parent
+
+ if (link := _first_link(lock_path)) is not None:
+ raise OutputPathError(f"refusing symlink/junction in export lock path: {link}")
+ if lock_path.exists() and not lock_path.is_file():
+ raise OutputPathError(f"export lock path is not a regular file: {lock_path}")
+
+ def acquire(self) -> Self:
+ """Acquire this destination lock without waiting for another owner."""
+ if self._fd is not None:
+ raise RuntimeError("export lock is already held")
+ self._validate_chain(self.destination, self.lock_path)
+ self.lock_path.parent.mkdir(parents=True, exist_ok=True)
+ # Revalidate after creating missing parents, before opening the lock.
+ self._validate_chain(self.destination, self.lock_path)
+ flags = os.O_CREAT | os.O_RDWR
+ if hasattr(os, "O_CLOEXEC"):
+ flags |= os.O_CLOEXEC
+ if hasattr(os, "O_NOFOLLOW"):
+ flags |= os.O_NOFOLLOW
+ try:
+ fd = os.open(self.lock_path, flags, 0o600)
+ except OSError as exc:
+ raise OutputPathError(f"cannot open export lock {self.lock_path}: {exc}") from exc
+
+ try:
+ if os.name == "nt":
+ os.lseek(fd, 0, os.SEEK_SET)
+ msvcrt.locking(fd, msvcrt.LK_NBLCK, 1)
+ else:
+ fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except OSError as exc:
+ os.close(fd)
+ if exc.errno in {errno.EACCES, errno.EAGAIN, errno.EDEADLK}:
+ raise ExportBusyError(
+ f"export destination is busy: {self.destination} "
+ f"(lock: {self.lock_path})"
+ ) from exc
+ raise OutputPathError(
+ f"cannot acquire export lock {self.lock_path}: {exc}"
+ ) from exc
+ self._fd = fd
+ return self
+
+ def release(self) -> None:
+ """Release the OS lock and close this instance's handle."""
+ if self._fd is None:
+ return
+ fd, self._fd = self._fd, None
+ try:
+ try:
+ if os.name == "nt":
+ os.lseek(fd, 0, os.SEEK_SET)
+ msvcrt.locking(fd, msvcrt.LK_UNLCK, 1)
+ else:
+ fcntl.flock(fd, fcntl.LOCK_UN)
+ finally:
+ os.close(fd)
+ except OSError:
+ # The handle is closed and no longer owned even if the platform
+ # reports an unlock error; do not mask the caller's exception.
+ pass
+
+ def __enter__(self) -> Self:
+ return self.acquire()
+
+ def __exit__(self, exc_type, exc_value, traceback) -> bool:
+ self.release()
+ return False
+
+
+@contextmanager
+def export_locks(*destinations: Path | str) -> Iterator[list[ExportLock]]:
+ """Acquire distinct destination locks and always unwind partial success.
+
+ Targets are deduplicated by normalized lexical path and acquired in a
+ stable platform-aware canonical absolute-path order, independent of caller
+ order. If a later lock is busy, all earlier locks are released before the
+ error escapes.
+ """
+ unique: dict[str, ExportLock] = {}
+ for destination in destinations:
+ lock = ExportLock(destination)
+ key = os.path.normcase(os.path.normpath(os.fspath(lock.destination)))
+ unique.setdefault(key, lock)
+ locks = [unique[key] for key in sorted(unique)]
+ acquired: list[ExportLock] = []
+ try:
+ for lock in locks:
+ lock.acquire()
+ acquired.append(lock)
+ yield locks
+ finally:
+ for lock in reversed(acquired):
+ lock.release()
+
+
+class OutputStaging:
+ """Own a temporary generated tree and promote it as one destination.
+
+ ``with OutputStaging(...) as stage`` yields this object; use ``stage.path``
+ or path-like operations to populate it, then call ``stage.promote()`` only
+ after the caller's validation succeeds. Exiting without promotion removes
+ only the temporary tree and leaves an existing destination untouched.
+ ``OutputStaging`` does not acquire an ``ExportLock`` itself; callers own one
+ non-reentrant lock for the complete mutation boundary.
+ """
+
+ _STAGE_PREFIX = ".output-stage-"
+ _BACKUP_PREFIX = ".output-backup-"
+
+ def __init__(
+ self,
+ destination: Path | str,
+ *,
+ work_dir: Path | str | None = None,
+ repo_root: Path | str | None = None,
+ source_root: Path | str | None = None,
+ ) -> None:
+ self.destination, self.work_dir = validate_output_paths(
+ destination,
+ work_dir=work_dir,
+ repo_root=repo_root,
+ source_root=source_root,
+ )
+ self.work_dir.mkdir(parents=True, exist_ok=True)
+ self.path = Path(tempfile.mkdtemp(prefix=self._STAGE_PREFIX, dir=self.work_dir))
+ self._promoted = False
+
+ def __enter__(self) -> Self:
+ return self
+
+ def __exit__(self, exc_type, exc_value, traceback) -> bool:
+ if not self._promoted:
+ self._remove_owned(self.path, self._STAGE_PREFIX)
+ return False
+
+ def __fspath__(self) -> str:
+ return os.fspath(self.path)
+
+ def __truediv__(self, child: str) -> Path:
+ return self.path / child
+
+ def iterdir(self):
+ return self.path.iterdir()
+
+ def rglob(self, pattern: str):
+ return self.path.rglob(pattern)
+
+ def promote(self) -> Path:
+ """Replace the validated destination with the staged tree."""
+ if self._promoted:
+ raise RuntimeError("staging tree has already been promoted")
+ if not self.path.is_dir() or self.path.parent != self.work_dir:
+ raise OutputPathError("staging tree is no longer owned by this instance")
+
+ self.destination.parent.mkdir(parents=True, exist_ok=True)
+ backup: Path | None = None
+ if self.destination.exists() or self.destination.is_symlink():
+ if self.destination.is_symlink():
+ raise OutputPathError(f"refusing symlink destination: {self.destination}")
+ backup = self.destination.parent / f"{self._BACKUP_PREFIX}{uuid.uuid4().hex}"
+ os.replace(self.destination, backup)
+ try:
+ os.replace(self.path, self.destination)
+ except Exception:
+ if backup is not None and not self.destination.exists():
+ os.replace(backup, self.destination)
+ raise
+ if backup is not None:
+ self._remove_owned(backup, self._BACKUP_PREFIX)
+ self._promoted = True
+ return self.destination
+
+ @staticmethod
+ def _remove_owned(path: Path, prefix: str) -> None:
+ """Remove a path only when it is a child with our ownership prefix."""
+ if path.name.startswith(prefix) and path.parent != path:
+ if path.is_dir() and not path.is_symlink():
+ shutil.rmtree(path)
+ elif path.exists() or path.is_symlink():
+ path.unlink()
+
+
+__all__ = [
+ "ExportBusyError", "ExportLock", "OutputPathError", "OutputStaging",
+ "export_locks", "validate_output_paths",
+]
diff --git a/tools/pdf_convert.py b/tools/pdf_convert.py
new file mode 100644
index 00000000..337d7463
--- /dev/null
+++ b/tools/pdf_convert.py
@@ -0,0 +1,1318 @@
+"""Convert the FwHelps PDFs into markdown.
+
+Thirteen PDFs, 349 pages: technical notes, utility documentation, and the
+Conceptual Introduction. All have real text layers, so no OCR is involved.
+
+Heading structure comes from one of two places:
+
+ * 7 PDFs carry bookmarks, which give exact headings and levels.
+ * 6 -- the Word-produced "Technical Notes" family -- carry none, so headings
+ are inferred from font size and weight.
+
+Inference is pinned. The expected outline of every PDF is recorded in
+pdf_outlines.json and re-checked on each build: these files change roughly once
+a decade (9 of 13 have exactly one commit in the repo's history), so drift
+almost always means the inference broke rather than the document changing.
+A mismatch fails the build instead of quietly shipping a mis-structured corpus.
+
+Usage:
+ python tools/pdf_convert.py --repo . --out export [--update-outlines]
+"""
+
+from __future__ import annotations
+
+import argparse
+import collections
+import hashlib
+import json
+import os
+import re
+import shutil
+import subprocess
+import tempfile
+import unicodedata
+from collections.abc import Mapping
+from pathlib import Path
+from urllib.parse import quote, urlsplit, urlunsplit
+
+import pymupdf as fitz # "fitz" name is deprecated; alias keeps call sites short
+import pymupdf4llm
+from frontmatter import yaml_scalar
+from output_fs import export_locks
+from pymupdf4llm.helpers.pymupdf_rag import TocHeaders
+from reporting import Report, make_issue
+from source_safety import discover_source_files, first_link_in_path
+
+OUTLINES = Path(__file__).parent / "pdf_outlines.json"
+_WINDOWS_RESERVED_BASENAMES = {
+ "CON", "PRN", "AUX", "NUL",
+ *(f"COM{i}" for i in range(1, 10)),
+ *(f"LPT{i}" for i in range(1, 10)),
+}
+
+
+def _slug_component(part: str) -> str:
+ value = part.replace(" ", "_")
+ value = re.sub(r'[<>:"|?*\\]', "_", value)
+ while value.endswith((".", " ")):
+ value = value[:-1] + "_"
+ if value.split(".", 1)[0].rstrip(" .").upper() in _WINDOWS_RESERVED_BASENAMES:
+ value = "_" + value
+ return value or "_"
+
+
+def slug_path(rel: str) -> str:
+ """Space-free output path.
+
+ pymupdf4llm rewrites spaces to underscores when it derives image filenames
+ from image_path, so a directory created as "Technical Notes_images" is not
+ the one it writes into. Underscores throughout also keep the URLs clean and
+ match the CHM side of the export, which RoboHelp already names that way.
+ """
+ return "/".join(_slug_component(part) for part in rel.split("/"))
+
+
+def clean(text: str) -> str:
+ """Normalise the non-breaking spaces Word leaves in headings and bookmarks."""
+ text = unicodedata.normalize("NFKC", text)
+ return re.sub(r"\s+", " ", text).strip()
+
+
+class FontHeaders:
+ """Infer heading levels from font size and weight.
+
+ The corpus's bookmark-less PDFs are Word documents with a consistent style:
+ 12pt regular body, 18pt bold H1, 14pt bold H2. Crucially, 12pt *bold* is
+ used for inline emphasis ("Note:", "Tip:") and must not become a heading --
+ so size alone is not enough, and bold alone is not enough either. A span has
+ to be both bold and meaningfully larger than the body text.
+ """
+
+ BOLD = 1 << 4
+
+ def __init__(self, doc: fitz.Document, min_lines: int = 2):
+ sizes: collections.Counter = collections.Counter()
+ bold_sizes: collections.Counter = collections.Counter()
+
+ for page in doc:
+ for block in page.get_text("dict")["blocks"]:
+ for line in block.get("lines", []):
+ spans = line.get("spans", [])
+ if not spans:
+ continue
+ text = "".join(s["text"] for s in spans).strip()
+ if not text:
+ continue
+ span = spans[0]
+ size = round(span["size"], 1)
+ sizes[size] += 1
+ if span["flags"] & self.BOLD:
+ bold_sizes[size] += 1
+
+ self.body = sizes.most_common(1)[0][0] if sizes else 12.0
+ # Candidate heading sizes: bold, bigger than the body, and used often
+ # enough to be a real style rather than a one-off.
+ candidates = sorted(
+ (s for s, n in bold_sizes.items() if s > self.body + 0.4 and n >= min_lines),
+ reverse=True,
+ )
+ self.levels = {size: i + 1 for i, size in enumerate(candidates[:6])}
+
+ def get_header_id(self, span: dict, page=None) -> str:
+ if not (span["flags"] & self.BOLD):
+ return ""
+ level = self.levels.get(round(span["size"], 1))
+ return "#" * level + " " if level else ""
+
+
+def running_margins(doc: fitz.Document, threshold: float = 0.6) -> tuple[float, float]:
+ """Find the top/bottom bands occupied by running headers and footers.
+
+ Every Word-produced PDF here repeats a header ("2 Getting started ... 4")
+ and footer ("Technical Notes on ...doc Edited on 8/13/2026") on each page;
+ left in, they appear in the markdown once per page.
+
+ Measured per document, not globally: silewp2007_002.pdf carries real body
+ text where the others put a footer, so a blanket margin would silently eat
+ content. A band only counts as furniture if it recurs on most pages.
+ """
+ if not doc.page_count:
+ return 0.0, 0.0
+ height = doc[0].rect.height
+ top_ys: list[float] = []
+ bottom_ys: list[float] = []
+ top_pages: set[int] = set()
+ bottom_pages: set[int] = set()
+
+ for i, page in enumerate(doc):
+ for block in page.get_text("dict")["blocks"]:
+ y0, y1 = block["bbox"][1], block["bbox"][3]
+ text = "".join(
+ s["text"] for line in block.get("lines", []) for s in line["spans"]
+ ).strip()
+ if not text:
+ continue
+ if y1 < height * 0.12:
+ top_pages.add(i)
+ top_ys.append(y1)
+ elif y0 > height * 0.90:
+ bottom_pages.add(i)
+ bottom_ys.append(y0)
+
+ def pct(values: list[float], q: float) -> float:
+ vs = sorted(values)
+ return vs[min(len(vs) - 1, int(q * len(vs)))]
+
+ need = threshold * doc.page_count
+ top = pct(top_ys, 0.9) + 2 if len(top_pages) >= need else 0.0
+ bottom = height - pct(bottom_ys, 0.1) + 2 if len(bottom_pages) >= need else 0.0
+ return round(top, 1), round(bottom, 1)
+
+
+# Word runs the dots together ("Introduction ....... 4"); XLingPaper/LaTeX
+# spaces them out (". . . . . . . 2"). Both end in a page number.
+LEADER = re.compile(r"(\.\s*){4,}\s*\d+\s*$")
+CONTENTS_HEAD = re.compile(
+ r"^#{1,6}\s*\**\s*(table of contents|contents|list of figures|list of tables)"
+ r"\s*:?[ ]*\**\s*$", re.IGNORECASE)
+INDEX_HEAD = re.compile(
+ r"^#{1,6}\s*\**\s*(language|subject|topic)?\s*index\s*\**\s*$",
+ re.IGNORECASE,
+)
+HEADING = re.compile(r"^(#{1,6})\s+(.*\S)\s*$")
+EQUATION_LABEL = re.compile(r"^\(\d+\)$")
+
+
+def strip_toc(md: str) -> str:
+ """Drop the document's own table of contents.
+
+ Every Word-produced PDF opens with dotted-leader contents lines, which
+ pymupdf4llm renders as a bogus one-column table -- 16 to 37 lines of
+ "2.1 Starting up a Project .......... 4" per document. The markdown file
+ already has real headings, and this branch has a README index, so the
+ inline copy is pure noise for a reader and for retrieval alike.
+ """
+ out = []
+ lines = md.splitlines()
+ front_limit = max(1, int(len(lines) * 0.25))
+ for i, line in enumerate(lines):
+ bare = line.strip().strip("|").strip()
+ in_front_matter = i < front_limit
+ if in_front_matter and LEADER.search(bare):
+ continue
+ # The table skeleton left behind once its rows are gone. Tested with a
+ # character-set check, not a regex: a nested-quantifier pattern like
+ # (:?-+:?\s*\|?)+ backtracks exponentially on a long separator row.
+ # A separator is only real if an actual table row precedes it.
+ if in_front_matter and bare and set(bare) <= set("|-: "):
+ prev = next((x for x in reversed(out) if x.strip()), "")
+ if not prev.strip().startswith("|"):
+ continue
+ if in_front_matter and re.fullmatch(
+ r"\|\s*Contents\s*\|", line.strip(), re.IGNORECASE
+ ):
+ continue
+ out.append(line)
+ return "\n".join(out)
+
+
+def strip_contents_sections(md: str) -> str:
+ """Drop front-matter contents sections, whatever their rendered shape."""
+ lines = md.splitlines()
+ out, i = [], 0
+ while i < len(lines):
+ match = CONTENTS_HEAD.match(lines[i].strip())
+ if match and i < len(lines) * 0.4:
+ level = len(HEADING.match(lines[i]).group(1))
+ end = i + 1
+ while end < len(lines):
+ # The TonePars paper places edition notes immediately after a
+ # compact contents list. They are prose, not TOC furniture.
+ if re.match(
+ r"^Editor['’]s note\b", lines[end].strip(), re.IGNORECASE
+ ):
+ break
+ heading = HEADING.match(lines[end])
+ if heading and len(heading.group(1)) <= level:
+ break
+ end += 1
+ # Without a verified closing boundary, preserve the section. A
+ # lower-level chapter may be the real body, and deleting to EOF is
+ # worse than retaining a redundant contents list.
+ if end < len(lines):
+ i = end
+ continue
+ out.append(lines[i])
+ i += 1
+ return "\n".join(out)
+
+
+def strip_back_index(md: str) -> str:
+ """Drop a back-of-book index.
+
+ ConceptualIntroFLEx ends with "Language index" and "Subject index": several
+ thousand words of headword-plus-page-number that carry no sentences, cannot
+ be followed without the printed pagination, and would otherwise be the
+ single largest retrievable block in the document.
+
+ Only honoured near the end of a document, so a section legitimately called
+ "Index" mid-text is left alone.
+ """
+ lines = md.splitlines()
+ for i, line in enumerate(lines):
+ if INDEX_HEAD.match(line.strip()) and i > len(lines) * 0.75:
+ return "\n".join(lines[:i]).rstrip() + "\n"
+ return md
+
+
+def demote_headings(md: str) -> str:
+ """Push every heading down one level and strip emphasis markers.
+
+ The page's own H1 is the document title, added by the caller, so the PDF's
+ top-level sections belong at H2. Word also bolds its headings, which pandoc
+ faithfully reproduces as "# **1 Introduction**".
+ """
+ out = []
+ for line in md.splitlines():
+ m = HEADING.match(line)
+ if not m:
+ out.append(line)
+ continue
+ level = min(6, len(m.group(1)) + 1)
+ text = re.sub(r"^[*_]{1,2}|[*_]{1,2}$", "", m.group(2).strip()).strip()
+ out.append(f"{'#' * level} {text}" if text else "")
+ return "\n".join(out)
+
+
+def markdown_label(line: str) -> str:
+ """Plain text from one simple Markdown heading or emphasized line."""
+ text = re.sub(r"^#{1,6}\s+", "", line.strip())
+ text = clean(re.sub(r"[*_`]", "", text))
+ return re.sub(r"\s+([:;,])", r"\1", text)
+
+
+def pick_title(meta_title: str, md: str, stem: str) -> str:
+ """Choose useful metadata, a document heading, or the curated filename."""
+ stem = stem.replace("_", " ").strip()
+ bad = re.compile(
+ r"\.(doc|pdf|rtf)x?\b|^microsoft word\b|\breadme\b", re.IGNORECASE
+ )
+
+ candidate = clean(meta_title)
+ if candidate and not bad.search(candidate) and 4 < len(candidate) < 120:
+ if candidate.lower() in stem.lower() and len(candidate) < len(stem):
+ return stem
+ return candidate
+
+ generic = {"contents", "table of contents", "list of figures", "list of tables"}
+ for line in md.splitlines()[:24]:
+ text = markdown_label(line)
+ if not text or text.rstrip(":").lower() in generic:
+ continue
+ if re.match(r"^\d+(?:\.\d+)*\s+(?=\S)", text) or text.isdigit():
+ continue
+ heading = HEADING.match(line)
+ if heading:
+ if 4 < len(text) < 120 and not bad.search(text):
+ return text
+ # Some title pages have no bookmark or heading style. Accept a title
+ # line before falling through to the generic Contents bookmark.
+ elif 12 < len(text) < 120 and not bad.search(text):
+ return text
+ return stem
+
+
+def drop_repeated_title(md: str, title: str) -> str:
+ """Remove the document's own title heading when it restates the page title.
+
+ Otherwise every PDF opens with the title twice -- once as the H1 this tool
+ adds, once as the heading from the PDF's title page ("Technical Notes on
+ FieldWorks Send-Receive" then "Technical Notes on Fieldworks Send/Receive").
+ Compared on letters and digits alone, so punctuation and casing differences
+ like Send-Receive vs Send/Receive still count as the same title.
+ """
+ key = lambda s: re.sub(r"[^a-z0-9]", "", markdown_label(s).lower())
+ want = key(title)
+ lines = md.splitlines()
+
+ # A title may be a plain emphasized line, a heading, or a heading followed
+ # by a separately styled continuation. Remove every opening copy: the
+ # Conceptual Introduction PDF contains the same split title twice.
+ while True:
+ found = False
+ for start in range(min(len(lines), 24)):
+ if not lines[start].strip():
+ continue
+ joined = ""
+ used = 0
+ for end in range(start, min(len(lines), start + 8)):
+ if not lines[end].strip():
+ continue
+ joined += key(lines[end])
+ used += 1
+ if joined == want:
+ del lines[start:end + 1]
+ while start < len(lines) and not lines[start].strip():
+ del lines[start]
+ found = True
+ break
+ if used >= 3 or not want.startswith(joined):
+ break
+ if found:
+ break
+ if not found:
+ break
+ return "\n".join(lines).strip() + "\n"
+
+
+def normalize_pdf_headings(md: str) -> str:
+ """Remove audited false headings and make the first real level H2.
+
+ Some bookmark trees promote author bylines and equation numbers to
+ headings. The former only occurs as a short title-page heading at level
+ four or deeper; the latter is unambiguously a standalone parenthesized
+ number. After those are removed, shift an otherwise valid tree so its
+ shallowest heading is the PDF body's H2.
+ """
+ lines = md.splitlines()
+ headings: list[tuple[int, int, str]] = []
+ fenced = False
+ for i, line in enumerate(lines):
+ if line.lstrip().startswith("```"):
+ fenced = not fenced
+ continue
+ if fenced:
+ continue
+ match = HEADING.match(line)
+ if match:
+ headings.append((i, len(match.group(1)), clean(match.group(2))))
+
+ remove: set[int] = set()
+ for i, level, text in headings:
+ if EQUATION_LABEL.fullmatch(text):
+ remove.add(i)
+
+ remaining = [(i, level, text) for i, level, text in headings if i not in remove]
+ if remaining:
+ first_i, first_level, first_text = remaining[0]
+ words = first_text.split()
+ looks_like_name = (
+ first_i < 8 and first_level >= 4 and 2 <= len(words) <= 5
+ and all(word[:1].isupper() for word in words if word)
+ and not any(char.isdigit() for char in first_text)
+ )
+ if looks_like_name:
+ remove.add(first_i)
+ remaining = remaining[1:]
+
+ shift = 2 - min((level for _, level, _ in remaining), default=2)
+ out: list[str] = []
+ for i, line in enumerate(lines):
+ if i in remove:
+ continue
+ match = HEADING.match(line)
+ if not match:
+ out.append(line)
+ continue
+ level = max(1, len(match.group(1)) + shift)
+ out.append("#" * level + line[len(match.group(1)):])
+ return "\n".join(out)
+
+
+def outline_of(md: str) -> list[tuple[int, str]]:
+ """Headings in generated markdown, ignoring anything inside fenced code."""
+ out, fenced = [], False
+ for line in md.splitlines():
+ if line.lstrip().startswith("```"):
+ fenced = not fenced
+ continue
+ if fenced:
+ continue
+ m = re.match(r"^(#{1,6})\s+(.*\S)\s*$", line)
+ if m:
+ out.append((len(m.group(1)), clean(m.group(2))))
+ return out
+
+
+def normalize_outline(outline: list[tuple[int, str]] | list[list]) -> list[list]:
+ """Return the ordered, comparable representation used by outline locks."""
+ return [[int(level), clean(str(text))] for level, text in outline]
+
+
+def outline_matches(pin: dict, outline: list[tuple[int, str]] | list[list]) -> bool:
+ """Compare every normalized heading, in order, with a pinned outline."""
+ expected = pin.get("outline")
+ if not isinstance(expected, list):
+ # Legacy count/level1 pins are intentionally not sufficient locks.
+ return False
+ return expected == normalize_outline(outline)
+
+
+def finalize_pdf(meta_title: str, md: str, stem: str) -> tuple[str, str, list]:
+ """Select the title and derive structure from the body that will be emitted."""
+ title = pick_title(meta_title, md, stem)
+ body = normalize_pdf_headings(drop_repeated_title(md, title))
+ return title, body, outline_of(body)
+
+
+PAGE_NUMBER = re.compile(r"^\d+$")
+HTML_TABLE = re.compile(r"
", re.DOTALL | re.IGNORECASE)
+
+
+def strip_furniture(pages: list[str], threshold: float = 0.6) -> list[str]:
+ """Remove running headers and footers from per-page markdown.
+
+ Detection is by repetition rather than by position, which keeps it
+ independent of the producing toolchain -- the corpus spans four Word
+ versions, XLingPaper/LaTeX, and two Acrobat Distiller variants. A line
+ qualifies only if it sits at the very top or bottom of its page and recurs,
+ modulo page numbers, on most pages.
+ """
+ if len(pages) < 3:
+ return pages
+
+ def key(line: str) -> str:
+ line = line.strip()
+ return "#page-number" if PAGE_NUMBER.fullmatch(line) else line
+
+ counts: collections.Counter = collections.Counter()
+ samples: dict[str, str] = {}
+ for page in pages:
+ lines = [ln.strip() for ln in page.splitlines() if ln.strip()]
+ for line in set(lines[:2] + lines[-2:]):
+ normalized = key(line)
+ counts[normalized] += 1
+ samples.setdefault(normalized, line)
+
+ # Word repeats the *current section* heading in the running header, so any
+ # one header text may cover only two pages. Accept two occurrences only for
+ # explicit Markdown headings and page numbers. Plain prose needs both three
+ # occurrences and a majority of pages, preventing repeated instructions or
+ # distinct numbered headings from being collapsed into furniture.
+ need = max(3, int(threshold * len(pages) + 0.999))
+ furniture = {
+ k for k, n in counts.items()
+ if ((k == "#page-number" or HEADING.match(samples[k])) and n >= 2)
+ or n >= need
+ }
+ if not furniture:
+ return pages
+
+ cleaned = []
+ for page in pages:
+ lines = page.splitlines()
+ # Trim from the ends only; an identical sentence mid-page is content.
+ while lines and (not lines[0].strip()
+ or key(lines[0]) in furniture):
+ lines.pop(0)
+ while lines and (not lines[-1].strip()
+ or key(lines[-1]) in furniture):
+ lines.pop()
+ cleaned.append("\n".join(lines))
+ return cleaned
+
+
+LUA = Path(__file__).parent / "fwhelp.lua"
+
+
+def tables_to_gfm(md: str, unconverted: list | None = None) -> str:
+ """Convert the HTML tables pymupdf4llm emits into GFM pipe tables.
+
+ Runs through fwhelp.lua, the same filter the CHM side uses. Without it
+ pandoc emits the table straight back as HTML, because pymupdf4llm attaches
+
widths and fixed column widths force a grid table -- the identical
+ problem RoboHelp's markup causes, and already solved there.
+ """
+ def convert(match: re.Match) -> str:
+ # A GFM pipe cell cannot hold a line break, so pandoc answers any table
+ # containing one with raw HTML instead -- which left 43 tables across
+ # the corpus unconverted. The breaks are where the PDF happened to wrap
+ # the cell text ("the analysis data control file used by XAmple."),
+ # so they encode page width rather than meaning and collapse to a space.
+ html = re.sub(r" ", " ", match.group(0))
+ proc = subprocess.run(
+ ["pandoc", "-f", "html", "-t", "gfm", "--wrap=none",
+ f"--lua-filter={LUA}"],
+ input=html, capture_output=True, text=True, encoding="utf-8",
+ check=False,
+ )
+ if proc.returncode != 0 or not proc.stdout.strip() or "
tuple[str, str, list]:
+ """Convert one PDF and retain source replacement-character provenance."""
+ doc = fitz.open(pdf)
+ try:
+ toc = doc.get_toc()
+ if toc:
+ hdr, strategy = TocHeaders(doc), "bookmarks"
+ else:
+ hdr, strategy = FontHeaders(doc), "font-inference"
+
+ image_dir.mkdir(parents=True, exist_ok=True)
+ chunks = pymupdf4llm.to_markdown(
+ doc,
+ hdr_info=hdr,
+ # Per page, so running headers/footers can be found by repetition.
+ # pymupdf4llm's `margins` argument does not reach this content --
+ # the footer survives at every margin value tested, including one
+ # far larger than the band it occupies -- so strip_furniture()
+ # removes them afterwards instead.
+ page_chunks=True,
+ write_images=True,
+ image_path=str(image_dir),
+ image_format="png",
+ # Skip decorative rules and page-furniture fragments; the corpus
+ # PDFs use them heavily and each would otherwise become a file.
+ image_size_limit=0.08,
+ # "lines_strict" only, never "text": the text strategy reports a
+ # 6-column x 53-row "table" for an ordinary page of prose, and
+ # treats every dotted-leader contents page as tabular. Verified by
+ # rendering the pages -- lines_strict finds 11 genuinely ruled
+ # tables across the corpus, and they are all real.
+ table_strategy="lines_strict",
+ # HTML, then pandoc, because pymupdf4llm's markdown table writer
+ # concatenates spans without separators: "part of Entry by default"
+ # comes out as "part of Entrybydefault", silently corrupting the SFM
+ # marker reference tables. Its HTML output keeps the spacing.
+ table_output="html",
+ show_progress=False,
+ )
+ pages = doc.page_count
+ source_replacements = []
+ for page_number, page in enumerate(doc, 1):
+ details = source_replacement_details(page.get_text())
+ if details:
+ source_replacements.append({"page": page_number, **details})
+ finally:
+ doc.close()
+
+ unconverted: list = []
+ md = tables_to_gfm("\n\n".join(strip_furniture([c["text"] for c in chunks])),
+ unconverted)
+
+ # pymupdf4llm builds image references relative to the process's working
+ # directory rather than to the markdown file that holds them, so each link
+ # arrives carrying the whole export path ("tools/out/pdf/X_images/...") and
+ # resolves only when the file is read from that one directory -- or, when
+ # --out is not under the cwd, as an absolute path that resolves nowhere
+ # else at all. The images are written beside the markdown, so reduce every
+ # reference to the sibling folder it actually sits in.
+ md = re.sub(rf"\(\S*?{re.escape(image_dir.name)}/", f"({image_dir.name}/", md)
+
+ # Remove the document's own contents list and back-of-book index, then push
+ # its headings down one level so the title supplied by the caller is the
+ # page's only H1.
+ md = demote_headings(strip_back_index(strip_contents_sections(strip_toc(md))))
+
+ md = re.sub(r"\n{4,}", "\n\n\n", md).strip() + "\n"
+ if source_replacements:
+ unconverted.append({"kind": "source_replacement", "pages": source_replacements})
+ return md, f"{strategy} ({pages}p)", unconverted
+
+
+def frontmatter(fields: dict) -> str:
+ lines = ["---"]
+
+ def emit(k, v, indent=""):
+ if v in (None, "", [], {}):
+ return
+ if isinstance(v, dict):
+ lines.append(f"{indent}{k}:")
+ for child, value in v.items():
+ emit(child, value, indent + " ")
+ elif isinstance(v, list):
+ lines.append(f"{indent}{k}:")
+ for item in v:
+ lines.append(f"{indent} - {yaml_scalar(item)}")
+ else:
+ lines.append(f"{indent}{k}: {yaml_scalar(v)}")
+
+ for k, v in fields.items():
+ emit(k, v)
+ return "\n".join(lines + ["---"])
+
+
+PDF_MANIFEST = ".pdf-converter-manifest.json"
+PDF_MANIFEST_SCHEMA = 2
+
+
+class ManifestError(ValueError):
+ """A PDF manifest is malformed or does not describe canonical outputs."""
+
+
+def _validate_lock_path(path: Path) -> Path:
+ """Reject linked or parent-traversing lock paths before any lock I/O."""
+ path = Path(path)
+ if ".." in path.parts:
+ raise ManifestError(f"unsafe lock path: {path}")
+ if first_link_in_path(path) is not None:
+ raise ManifestError(f"lock path contains symlink/junction: {path}")
+ return path
+
+
+def _manifest_pairs(pairs):
+ result = {}
+ for key, value in pairs:
+ if key in result:
+ raise ManifestError(f"duplicate manifest key: {key!r}")
+ result[key] = value
+ return result
+
+
+def _safe_manifest_relative(value: object, label: str) -> str:
+ if (not isinstance(value, str) or not value or "\\" in value
+ or "\x00" in value or ":" in value):
+ raise ManifestError(f"unsafe {label} path: {value!r}")
+ if any(ord(char) < 32 for char in value):
+ raise ManifestError(f"unsafe {label} path: {value!r}")
+ path = Path(value)
+ if value.startswith(("/", "\\")) or path.drive:
+ raise ManifestError(f"absolute {label} path: {value!r}")
+ parts = value.split("/")
+ if any(part in {"", ".", ".."} for part in parts):
+ raise ManifestError(f"traversal {label} path: {value!r}")
+ return value
+
+
+def _manifest_destinations(source: str) -> tuple[str, str]:
+ if not source.casefold().endswith(".pdf"):
+ raise ManifestError(f"manifest source is not a PDF: {source!r}")
+ rel_dest = slug_path(source)[:-4] + ".md"
+ rel_images = (Path(rel_dest).parent / (Path(rel_dest).stem + "_images")).as_posix()
+ return rel_dest, rel_images
+
+
+def _ownership_key(relative: str) -> str:
+ """Return the host filesystem's normalized key for a safe relative path."""
+ return os.path.normcase(os.path.normpath(relative))
+
+
+def validate_manifest(manifest: Mapping, out: Path, *,
+ authenticated_images: set[str] | None = None) -> dict[str, dict]:
+ """Validate a complete schema-2 manifest before any output mutation."""
+ if not isinstance(manifest, dict) or set(manifest) != {"schema", "files"}:
+ raise ManifestError("manifest must contain exactly schema and files")
+ if type(manifest["schema"]) is not int or manifest["schema"] != PDF_MANIFEST_SCHEMA:
+ raise ManifestError("unsupported PDF manifest schema")
+ files = manifest["files"]
+ if not isinstance(files, dict):
+ raise ManifestError("manifest files must be an object")
+ root = Path(os.path.abspath(os.fspath(out)))
+ authenticated_image_keys = {
+ _ownership_key(value) for value in (authenticated_images or set())
+ if isinstance(value, str)
+ }
+ seen_sources: set[str] = set()
+ seen_destinations: set[str] = set()
+ normalized: dict[str, dict] = {}
+ for source, entry in files.items():
+ source = _safe_manifest_relative(source, "source")
+ source_key = source.casefold()
+ if source_key in seen_sources:
+ raise ManifestError(f"case-colliding manifest source: {source!r}")
+ seen_sources.add(source_key)
+ if not isinstance(entry, dict) or set(entry) != {"markdown", "images"}:
+ raise ManifestError(f"invalid manifest entry for {source!r}")
+ expected_markdown, expected_images = _manifest_destinations(source)
+ markdown = _safe_manifest_relative(entry["markdown"], "markdown")
+ if markdown != expected_markdown:
+ raise ManifestError(f"markdown destination mismatch for {source!r}")
+ images = entry["images"]
+ if images is not None:
+ images = _safe_manifest_relative(images, "images")
+ if images != expected_images:
+ raise ManifestError(f"images destination mismatch for {source!r}")
+ elif first_link_in_path(root / expected_images) is not None:
+ raise ManifestError(f"images destination contains symlink/junction: {source!r}")
+ elif (root / expected_images).exists() and _ownership_key(expected_images) not in authenticated_image_keys:
+ raise ManifestError(f"null images destination has existing output: {source!r}")
+ for destination in (markdown, images):
+ if destination is None:
+ continue
+ key = destination.casefold()
+ if key in seen_destinations:
+ raise ManifestError(f"case-colliding manifest destination: {destination!r}")
+ seen_destinations.add(key)
+ if first_link_in_path(root / destination) is not None:
+ raise ManifestError(f"manifest destination contains symlink/junction: {destination!r}")
+ if _owned_path(root, destination) is None:
+ raise ManifestError(f"manifest destination escapes output root: {destination!r}")
+ normalized[source] = {"markdown": markdown, "images": images}
+ return normalized
+
+
+def _load_manifest(path: Path, out: Path) -> tuple[dict, dict[str, dict]]:
+ if first_link_in_path(path) is not None:
+ raise ManifestError("PDF manifest path contains symlink/junction")
+ try:
+ manifest = json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=_manifest_pairs)
+ except ManifestError:
+ raise
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
+ raise ManifestError(f"cannot read PDF manifest: {exc}") from exc
+ return manifest, validate_manifest(manifest, out)
+
+
+def _normalize_root(path: Path) -> Path:
+ """Normalize lexical ``..`` segments without resolving symlinks."""
+ return Path(os.path.normpath(os.fspath(Path(path).expanduser())))
+
+
+def discover_pdfs(repo: Path) -> list[Path]:
+ """Discover PDF inputs independent of filename case on the host OS."""
+ repo = _normalize_root(repo)
+ return discover_source_files(
+ repo, suffixes={".pdf"}, recursive=True, exclude_dirs={".git"}
+ )
+
+
+def destination_collisions(repo: Path, out: Path,
+ pdfs: list[Path] | None = None) -> list[tuple[str, list[str]]]:
+ """Find PDFs whose normalized destination path is claimed more than once."""
+ repo = _normalize_root(repo)
+ pdfs = pdfs if pdfs is not None else discover_pdfs(repo)
+ claimed: dict[str, tuple[str, list[str]]] = {}
+ for pdf in pdfs:
+ rel = pdf.relative_to(repo).as_posix()
+ dest = out / (slug_path(rel)[: -len(".pdf")] + ".md")
+ display = dest.relative_to(out).as_posix()
+ key = display.casefold()
+ if key not in claimed:
+ claimed[key] = (display, [])
+ claimed[key][1].append(rel)
+ return [
+ (display, sorted(rels))
+ for display, rels in sorted(claimed.values()) if len(rels) > 1
+ ]
+
+
+def _owned_path(out: Path, relative: str) -> Path | None:
+ """Return a lexical in-root path, refusing link chains."""
+ candidate = Path(os.path.abspath(os.fspath(Path(out) / relative)))
+ root = Path(os.path.abspath(os.fspath(out)))
+ if first_link_in_path(candidate) is not None:
+ raise ManifestError(f"manifest path contains symlink/junction: {relative!r}")
+ try:
+ candidate.relative_to(root)
+ except ValueError:
+ return None
+ return candidate
+
+
+def _validate_regular_file(path: Path, label: str) -> None:
+ if first_link_in_path(path) is not None or not path.exists() or not path.is_file():
+ raise ManifestError(f"{label} must be an existing regular non-link file: {path}")
+
+
+def _validate_clean_directory(path: Path, label: str) -> None:
+ if first_link_in_path(path) is not None or not path.exists() or not path.is_dir():
+ raise ManifestError(f"{label} must be an existing non-link directory: {path}")
+ pending = [path]
+ while pending:
+ current = pending.pop()
+ try:
+ entries = list(os.scandir(current))
+ except OSError as exc:
+ raise ManifestError(f"cannot inspect {label}: {current}") from exc
+ for entry in entries:
+ child = Path(entry.path)
+ if first_link_in_path(child) is not None:
+ raise ManifestError(f"{label} contains symlink/junction: {child}")
+ if entry.is_dir(follow_symlinks=False):
+ pending.append(child)
+ elif not entry.is_file(follow_symlinks=False):
+ raise ManifestError(f"{label} contains non-regular entry: {child}")
+
+
+def _validate_manifest_outputs(out: Path, stage: Path,
+ previous: dict[str, dict], current: dict[str, dict]) -> None:
+ """Validate all filesystem objects before staging or replacing anything."""
+ for files in previous.values():
+ if not isinstance(files, dict):
+ continue
+ markdown = files.get("markdown")
+ if isinstance(markdown, str):
+ path = _owned_path(out, markdown)
+ if path is None:
+ raise ManifestError(f"prior markdown escapes output: {markdown!r}")
+ _validate_regular_file(path, "prior markdown")
+ images = files.get("images")
+ if isinstance(images, str):
+ path = _owned_path(out, images)
+ if path is None:
+ raise ManifestError(f"prior images escape output: {images!r}")
+ _validate_clean_directory(path, "prior images")
+
+ prior_path_keys = {
+ _ownership_key(value)
+ for files in previous.values()
+ if isinstance(files, dict)
+ for value in (files.get("markdown"), files.get("images"))
+ if isinstance(value, str)
+ }
+ for files in current.values():
+ if not isinstance(files, dict):
+ continue
+ markdown = files.get("markdown")
+ if isinstance(markdown, str):
+ staged = _owned_path(stage, markdown)
+ if staged is None:
+ raise ManifestError(f"current markdown escapes stage: {markdown!r}")
+ _validate_regular_file(staged, "current staged markdown")
+ target = _owned_path(out, markdown)
+ if target is not None and target.exists() and _ownership_key(markdown) not in prior_path_keys:
+ raise FileExistsError(f"PDF destination is not converter-owned: {markdown}")
+ images = files.get("images")
+ if isinstance(images, str):
+ staged = _owned_path(stage, images)
+ if staged is None:
+ raise ManifestError(f"current images escape stage: {images!r}")
+ _validate_clean_directory(staged, "current staged images")
+ target = _owned_path(out, images)
+ if target is not None and target.exists() and _ownership_key(images) not in prior_path_keys:
+ raise FileExistsError(f"PDF destination is not converter-owned: {images}")
+
+
+def _remove_owned_entry(out: Path, files: dict) -> None:
+ for key in ("markdown", "images"):
+ value = files.get(key)
+ path = _owned_path(out, value) if isinstance(value, str) else None
+ if path is None or not path.exists():
+ continue
+ if path.is_dir():
+ shutil.rmtree(path)
+ else:
+ path.unlink()
+
+
+def _promote_pdf_outputs_locked(out: Path, stage: Path,
+ previous: dict, current: dict,
+ lock_path: Path | None = None,
+ fresh: dict | None = None) -> None:
+ """Replace the complete converter-owned PDF set from a finished stage."""
+ out = Path(out)
+ if lock_path is not None:
+ lock_path = _validate_lock_path(lock_path)
+ manifest_path = out / PDF_MANIFEST
+ # Validate the on-disk manifest before even creating a backup. The caller's
+ # parsed view is also checked so a stale/tampered in-memory view cannot
+ # authorize a different deletion set.
+ if first_link_in_path(manifest_path) is not None:
+ raise ManifestError("PDF manifest path contains symlink/junction")
+ if manifest_path.exists():
+ _, actual_previous = _load_manifest(manifest_path, out)
+ if previous != actual_previous:
+ raise ManifestError("previous PDF manifest does not match on-disk manifest")
+ else:
+ actual_previous = validate_manifest(
+ {"schema": PDF_MANIFEST_SCHEMA, "files": previous}, out
+ )
+ previous = actual_previous
+ prior_images = {
+ _ownership_key(files["images"]) for files in previous.values()
+ if isinstance(files, dict) and isinstance(files.get("images"), str)
+ }
+ validate_manifest(
+ {"schema": PDF_MANIFEST_SCHEMA, "files": current}, out,
+ authenticated_images=prior_images,
+ )
+ _validate_manifest_outputs(out, stage, previous, current)
+ out.mkdir(parents=True, exist_ok=True)
+ staged_manifest = stage / PDF_MANIFEST
+ if first_link_in_path(staged_manifest) is not None:
+ raise ManifestError("staged PDF manifest path contains symlink/junction")
+ staged_manifest.write_text(
+ json.dumps({"schema": PDF_MANIFEST_SCHEMA, "files": current}, indent=1,
+ ensure_ascii=False) + "\n",
+ encoding="utf-8",
+ )
+ staged_lock = stage / ".pdf-outlines.json"
+ if lock_path is not None:
+ _validate_lock_path(lock_path)
+ if first_link_in_path(staged_lock) is not None:
+ raise ManifestError("staged PDF lock path contains symlink/junction")
+ staged_lock.write_text(
+ json.dumps(fresh or {}, indent=1, ensure_ascii=False) + "\n",
+ encoding="utf-8",
+ )
+
+ # Validate targets before removing anything. Existing paths are safe to
+ # replace only when the prior manifest claimed them; CHM/unrelated output
+ # must never be overwritten by a PDF conversion.
+ prior_paths = {
+ value
+ for files in previous.values()
+ if isinstance(files, dict)
+ for value in (files.get("markdown"), files.get("images"))
+ if isinstance(value, str)
+ }
+
+ backup = Path(tempfile.mkdtemp(prefix=".pdf-backup-", dir=out.parent))
+ backed_up: list[tuple[str, Path]] = []
+ manifest_backup = backup / PDF_MANIFEST
+ had_manifest = manifest_path.exists()
+ lock_backup = backup / "pdf-outlines.json"
+ if lock_path is not None:
+ _validate_lock_path(lock_path)
+ had_lock = lock_path is not None and lock_path.exists()
+ try:
+ if had_manifest:
+ shutil.copy2(manifest_path, manifest_backup)
+ if had_lock and lock_path is not None:
+ shutil.copy2(lock_path, lock_backup)
+
+ # Keep a recoverable copy until every staged path has been promoted.
+ for value in prior_paths:
+ source = _owned_path(out, value)
+ target = _owned_path(backup, value)
+ if source is None or target is None or not source.exists():
+ continue
+ target.parent.mkdir(parents=True, exist_ok=True)
+ if source.is_dir():
+ shutil.copytree(source, target)
+ else:
+ shutil.copy2(source, target)
+ backed_up.append((value, target))
+
+ # Remove every prior owned path, including the same source's image
+ # folder: staging is a complete replacement, so omitted files die.
+ for files in previous.values():
+ if isinstance(files, dict):
+ _remove_owned_entry(out, files)
+
+ for files in current.values():
+ if not isinstance(files, dict):
+ continue
+ for key in ("markdown", "images"):
+ value = files.get(key)
+ if not isinstance(value, str):
+ continue
+ source = _owned_path(stage, value)
+ if source is None or not source.exists():
+ continue
+ if source.is_dir() and not any(source.iterdir()):
+ continue
+ target = _owned_path(out, value)
+ if target is None:
+ continue
+ target.parent.mkdir(parents=True, exist_ok=True)
+ shutil.move(str(source), str(target))
+ shutil.move(str(staged_manifest), str(manifest_path))
+ if lock_path is not None:
+ _validate_lock_path(lock_path)
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
+ shutil.move(str(staged_lock), str(lock_path))
+ except Exception:
+ # A failed move must not leave a half-promoted PDF set behind.
+ for files in current.values():
+ if isinstance(files, dict):
+ _remove_owned_entry(out, files)
+ for value, source in backed_up:
+ target = _owned_path(out, value)
+ if target is None or not source.exists():
+ continue
+ target.parent.mkdir(parents=True, exist_ok=True)
+ if source.is_dir():
+ shutil.copytree(source, target)
+ else:
+ shutil.copy2(source, target)
+ if had_manifest and manifest_backup.exists():
+ manifest_path.unlink(missing_ok=True)
+ shutil.copy2(manifest_backup, manifest_path)
+ elif not had_manifest:
+ manifest_path.unlink(missing_ok=True)
+ if lock_path is not None:
+ if had_lock and lock_backup.exists():
+ lock_path.unlink(missing_ok=True)
+ shutil.copy2(lock_backup, lock_path)
+ elif not had_lock:
+ lock_path.unlink(missing_ok=True)
+ raise
+ finally:
+ shutil.rmtree(backup, ignore_errors=True)
+
+
+def promote_pdf_outputs(out: Path, stage: Path,
+ previous: dict, current: dict,
+ lock_path: Path | None = None,
+ fresh: dict | None = None) -> None:
+ """Promote PDF output while serializing direct destination mutation."""
+ out = Path(out)
+ if lock_path is not None:
+ # Keep the same output-then-global order as ``run`` when updating the
+ # shared outline lock. The locked implementation is used internally
+ # by ``run`` to avoid recursive acquisition.
+ with export_locks(out, OUTLINES):
+ _promote_pdf_outputs_locked(
+ out, stage, previous, current, lock_path=lock_path, fresh=fresh,
+ )
+ else:
+ with export_locks(out):
+ _promote_pdf_outputs_locked(
+ out, stage, previous, current, lock_path=lock_path, fresh=fresh,
+ )
+
+
+def _source_url(rel: str, resolver=None) -> str:
+ """Resolve a stable source URL/ref through the small orchestration seam."""
+ encoded = "/".join(quote(part, safe="") for part in rel.split("/"))
+ if resolver is None:
+ value = encoded
+ elif callable(resolver):
+ value = str(resolver(rel))
+ elif isinstance(resolver, dict):
+ value = str(resolver.get(rel, encoded))
+ else:
+ value = str(resolver)
+ if "{path}" in value:
+ value = value.format(path=encoded)
+ else:
+ value = value.rstrip("/") + "/" + encoded
+ if "{path}" in value:
+ value = value.format(path=encoded)
+ parts = urlsplit(value)
+ if parts.scheme or parts.netloc:
+ value = urlunsplit((
+ parts.scheme,
+ parts.netloc,
+ quote(parts.path, safe="/%:@-._~!$&'()*+,;=%"),
+ parts.query,
+ parts.fragment,
+ ))
+ else:
+ value = quote(value, safe="/%:@-._~!$&'()*+,;=%")
+ return value
+
+
+def _sha256(path: Path) -> str:
+ digest = hashlib.sha256()
+ with path.open("rb") as source:
+ for chunk in iter(lambda: source.read(1024 * 1024), b""):
+ digest.update(chunk)
+ return digest.hexdigest()
+
+
+def replacement_provenance(source_pages: list[dict], generated_count: int) -> dict:
+ """Classify replacement characters as source-derived or exporter-created."""
+ source_count = sum(int(item.get("count", 0)) for item in source_pages)
+ return {
+ "source_count": source_count,
+ "exporter_count": max(0, generated_count - source_count),
+ }
+
+
+def source_replacement_details(text: str) -> dict | None:
+ """Identify source glyphs likely to become replacement characters."""
+ chars = [
+ char for char in text
+ if char == "\ufffd" or (ord(char) < 32 and char not in "\t\n\r")
+ ]
+ if not chars:
+ return None
+ return {
+ "count": len(chars),
+ "codepoints": sorted({f"U+{ord(char):04X}" for char in chars}),
+ }
+
+
+def _run_locked(repo: Path, out: Path, update: bool, source_url=None,
+ source_ref=None) -> tuple[dict, list[str]]:
+ """Convert all PDFs, with ``source_url`` as the repository policy seam.
+
+ ``source_url`` may be a URL template containing ``{path}``, a mapping, or
+ a callable receiving the repository-relative PDF path. ``source_ref`` is a
+ backwards-compatible alias for callers that prefer that terminology.
+ """
+ repo = _normalize_root(repo)
+ if source_ref is not None:
+ source_url = source_ref
+ _validate_lock_path(OUTLINES)
+ pins = json.loads(OUTLINES.read_text(encoding="utf-8")) if OUTLINES.exists() else {}
+ fresh: dict[str, dict] = {}
+ report: dict[str, list] = collections.defaultdict(list)
+ lines: list[str] = []
+
+ pdfs = discover_pdfs(repo)
+ collisions = destination_collisions(repo, out, pdfs)
+ if collisions:
+ report["destination_collisions"].extend(collisions)
+ lines.extend(f" COLLISION {dest}: {', '.join(rels)}" for dest, rels in collisions)
+ return {"converted": 0, "report": dict(report), "lines": lines}, lines
+
+ out.parent.mkdir(parents=True, exist_ok=True)
+ previous: dict = {}
+ manifest_path = out / PDF_MANIFEST
+ if first_link_in_path(manifest_path) is not None:
+ raise ManifestError("PDF manifest path contains symlink/junction")
+ if manifest_path.exists():
+ _, previous = _load_manifest(manifest_path, out)
+ current: dict = {}
+ stage = Path(tempfile.mkdtemp(prefix=".pdf-convert-", dir=out.parent))
+
+ try:
+ for pdf in pdfs:
+ rel = pdf.relative_to(repo).as_posix()
+ rel_dest = slug_path(rel)[: -len(".pdf")] + ".md"
+ dest = stage / rel_dest
+ images = dest.parent / (dest.stem + "_images")
+
+ try:
+ md, strategy, unconverted = convert_pdf(pdf, dest, images)
+ # A single damaged PDF is reportable while the remaining corpus
+ # continues through staging; this boundary intentionally catches
+ # converter/backend errors of varying concrete types.
+ except Exception as exc: # noqa: BLE001
+ report["pdf_failures"].append([rel, f"{type(exc).__name__}: {exc}"])
+ lines.append(f" FAILED {rel}: {exc}")
+ continue
+
+ table_warnings = [item for item in unconverted if not isinstance(item, dict)]
+ source_warnings = [
+ item for item in unconverted
+ if isinstance(item, dict) and item.get("kind") == "source_replacement"
+ ]
+ if table_warnings:
+ report["html_tables_kept"].append([rel, len(table_warnings)])
+ source_pages = [
+ page for warning in source_warnings for page in warning["pages"]
+ ]
+ provenance = replacement_provenance(source_pages, md.count("\ufffd"))
+ if provenance["source_count"]:
+ report["pdf_source_replacements"].append([rel, source_pages])
+ if provenance["exporter_count"]:
+ report["pdf_export_replacements"].append([rel, provenance])
+ with fitz.open(pdf) as _doc:
+ metadata = dict(_doc.metadata or {})
+ meta_title = metadata.get("title") or ""
+ title, body, outline = finalize_pdf(meta_title, md, pdf.stem)
+ normalized_outline = normalize_outline(outline)
+ fresh[rel] = {"outline": normalized_outline, "headings": len(outline)}
+
+ pin = pins.get(rel)
+ if not update:
+ if pin is None:
+ report["outline_unpinned"].append(rel)
+ elif not outline_matches(pin, outline):
+ expected = pin.get("outline", "")
+ report["outline_drift"].append([
+ rel,
+ f"expected ordered outline {expected!r}, got {normalized_outline!r}",
+ ])
+
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ fm = frontmatter({
+ "title": title,
+ "source": rel,
+ "source_url": _source_url(rel, source_url),
+ "sha256": _sha256(pdf),
+ "pdf_metadata": metadata,
+ "type": "pdf",
+ "outline_count": len(outline),
+ "structure": strategy,
+ })
+ dest.write_text(f"{fm}\n\n# {title}\n\n{body}",
+ encoding="utf-8")
+ n_img = len(list(images.glob("*"))) if images.exists() else 0
+ if not n_img and images.exists():
+ images.rmdir()
+ current[rel] = {
+ "markdown": rel_dest,
+ "images": (
+ (Path(rel_dest).parent / (Path(rel_dest).stem + "_images")).as_posix()
+ if n_img else None
+ ),
+ }
+ lines.append(f" {strategy:<22} {len(outline):>3} hdrs {n_img:>4} img {rel}")
+
+ # No final output or manifest is touched until every PDF and every lock
+ # check succeeds. A late failure therefore leaves the prior PDF set.
+ lock_fatal = (report.get("outline_drift")
+ or report.get("outline_unpinned")) and not update
+ fatal = (report.get("pdf_failures") or lock_fatal
+ or report.get("pdf_export_replacements"))
+ if not fatal and not report.get("pdf_failures"):
+ _promote_pdf_outputs_locked(
+ out, stage, previous, current,
+ lock_path=OUTLINES if update else None,
+ fresh=fresh if update else None,
+ )
+ if update:
+ lines.append(f"\n pinned {len(fresh)} outlines -> {OUTLINES}")
+ finally:
+ shutil.rmtree(stage, ignore_errors=True)
+
+ return {"converted": len(fresh), "report": dict(report), "lines": lines}, lines
+
+
+def run_in_private_stage(repo: Path, out: Path, update: bool, source_url=None,
+ source_ref=None) -> tuple[dict, list[str]]:
+ """Convert into a caller-owned private stage while locking global state.
+
+ The caller must guarantee that ``out`` is an unshared staging path. Such
+ a path needs no destination lock; creating one beside it would turn the
+ lockfile into generated content when the enclosing stage is promoted.
+ """
+ with export_locks(OUTLINES):
+ return _run_locked(
+ repo, Path(out), update, source_url=source_url, source_ref=source_ref,
+ )
+
+
+def run(repo: Path, out: Path, update: bool, source_url=None,
+ source_ref=None) -> tuple[dict, list[str]]:
+ """Convert PDFs under output-then-global-outline locks.
+
+ The global outline lock covers the pinned-outline read and any update
+ promotion, including the interval between those operations.
+ """
+ out = Path(out)
+ with export_locks(out, OUTLINES):
+ return _run_locked(
+ repo, out, update, source_url=source_url, source_ref=source_ref,
+ )
+
+
+def _canonical_report(producer_report: dict) -> Report:
+ """Render producer findings through the shared issue catalog."""
+ issues = []
+ for code, entries in producer_report.items():
+ for item in entries if isinstance(entries, list) else [entries]:
+ if isinstance(item, (list, tuple)) and item:
+ path = str(item[0])
+ message = str(item[1]) if len(item) > 1 else ""
+ else:
+ path = ""
+ message = str(item)
+ issues.append(make_issue(code, message, path, item))
+ return Report(issues)
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser()
+ ap.add_argument("--repo", default=".", type=Path)
+ ap.add_argument("--out", default="export", type=Path)
+ ap.add_argument("--update-outlines", action="store_true",
+ help="re-pin pdf_outlines.json after reviewing the diff")
+ args = ap.parse_args()
+
+ result, lines = run(args.repo.resolve(), args.out.resolve(), args.update_outlines)
+ print("\n".join(lines))
+ rep = result["report"]
+ print(f"\nconverted {result['converted']} PDFs")
+ canonical = _canonical_report(rep)
+ print(canonical.to_console())
+ return 1 if canonical.fatal else 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/tools/pdf_outlines.json b/tools/pdf_outlines.json
new file mode 100644
index 00000000..11b10f07
--- /dev/null
+++ b/tools/pdf_outlines.json
@@ -0,0 +1,1751 @@
+{
+ "FieldWorks Writing Systems.pdf": {
+ "outline": [
+ [
+ 2,
+ "FieldWorks"
+ ],
+ [
+ 3,
+ "Writing Systems"
+ ],
+ [
+ 4,
+ "Add a Writing System to a FieldWorks project"
+ ],
+ [
+ 4,
+ "Add a new Writing System using the Writing System Wizard"
+ ],
+ [
+ 5,
+ "Language ID: Find the relevant language in the Ethnologue"
+ ],
+ [
+ 6,
+ "Result: one match"
+ ],
+ [
+ 6,
+ "Result: duplicate language names"
+ ],
+ [
+ 6,
+ "Result: many matches"
+ ],
+ [
+ 6,
+ "Result: apparent duplicate matches"
+ ],
+ [
+ 6,
+ "Result: search by country name"
+ ],
+ [
+ 6,
+ "Result: no matches"
+ ],
+ [
+ 5,
+ "Writing System: Distinguish the Writing System"
+ ],
+ [
+ 5,
+ "Appearance: Default fonts"
+ ],
+ [
+ 5,
+ "Input: Keyboard"
+ ],
+ [
+ 5,
+ "Additional information"
+ ],
+ [
+ 6,
+ "Modifying properties of a Writing System"
+ ],
+ [
+ 6,
+ "Full name of a Writing System"
+ ],
+ [
+ 6,
+ "Information about keyboards and Keyman"
+ ]
+ ],
+ "headings": 18
+ },
+ "Language Explorer/Training/Publishing FLEx Dictionaries Using Microsoft Word.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 2,
+ "2 Exporting a dictionary to Word"
+ ],
+ [
+ 3,
+ "2.1 Preparation in FLEx prior to export"
+ ],
+ [
+ 4,
+ "Saving list reference names in the Word document"
+ ],
+ [
+ 3,
+ "2.2 Exporting a dictionary from FLEx to Word"
+ ],
+ [
+ 3,
+ "2.3 Preparing a Word dictionary for publication"
+ ],
+ [
+ 4,
+ "Page layout"
+ ],
+ [
+ 4,
+ "Exported styles in Word"
+ ],
+ [
+ 4,
+ "2.3.2.1 Context styles for Before, Between, and After text"
+ ],
+ [
+ 4,
+ "Add page headers with guidewords"
+ ],
+ [
+ 4,
+ "2.3.3.1 Changing Headers and Footers"
+ ],
+ [
+ 4,
+ "How to fix incorrect guidewords on some pages"
+ ],
+ [
+ 5,
+ "7. The remaining pages should continue with the normal page headers."
+ ],
+ [
+ 4,
+ "Convert letter headings to single column"
+ ],
+ [
+ 4,
+ "Pictures in Word"
+ ],
+ [
+ 4,
+ "Editing the internal Word docx file"
+ ],
+ [
+ 4,
+ "Dictionary front matter and back matter"
+ ],
+ [
+ 2,
+ "3 Useful features in Word"
+ ],
+ [
+ 3,
+ "3.1 Basics of Styles in Word"
+ ],
+ [
+ 3,
+ "3.2 Preventing styles from changing: dynamic updating"
+ ],
+ [
+ 3,
+ "3.3 Advanced Find and Replace"
+ ],
+ [
+ 4,
+ "3.3.1Finding styles"
+ ],
+ [
+ 4,
+ "Replacing styles"
+ ],
+ [
+ 4,
+ "Special codes"
+ ],
+ [
+ 4,
+ "Wild cards"
+ ],
+ [
+ 4,
+ "Adding text before or after a style"
+ ],
+ [
+ 3,
+ "3.4 Removing unused styles in Word"
+ ],
+ [
+ 3,
+ "3.5 Word macros"
+ ],
+ [
+ 4,
+ "Example recording a macro and running it through your document"
+ ],
+ [
+ 3,
+ "3.6 Editing the internal Word docx file"
+ ],
+ [
+ 2,
+ "4 Bidirectional dictionaries in Word"
+ ],
+ [
+ 3,
+ "4.1 Enabling right-to-left publications in Word"
+ ],
+ [
+ 3,
+ "4.2 Basic RTL layout in FLEx and Word"
+ ],
+ [
+ 4,
+ "Settings in FLEx"
+ ],
+ [
+ 4,
+ "Settings in Word"
+ ],
+ [
+ 3,
+ "4.3 Bidirectional algorithm"
+ ],
+ [
+ 3,
+ "4.4 Formatting bidirectional entries in FLEx"
+ ],
+ [
+ 5,
+ "Word:"
+ ]
+ ],
+ "headings": 38
+ },
+ "Language Explorer/Training/Technical Notes on FieldWorks Send-Receive.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Send/Receive Introduction"
+ ],
+ [
+ 2,
+ "2 Getting started"
+ ],
+ [
+ 3,
+ "2.1 Starting up a Project (FLEx) Send/Receive"
+ ],
+ [
+ 3,
+ "2.2 Starting up a Lexicon (LIFT) Send/Receive in FLEx."
+ ],
+ [
+ 3,
+ "2.3 Starting up Send/Receive in WeSay."
+ ],
+ [
+ 3,
+ "2.4 Chorus Hub startup"
+ ],
+ [
+ 3,
+ "2.5 Lexbox Internet setup"
+ ],
+ [
+ 2,
+ "3 How it works, or why it doesn’t"
+ ],
+ [
+ 3,
+ "3.1 Lexicon examples"
+ ],
+ [
+ 3,
+ "3.2 General concepts"
+ ],
+ [
+ 3,
+ "3.3 Interlinear examples"
+ ],
+ [
+ 3,
+ "3.4 WeSay/LIFT collaboration"
+ ],
+ [
+ 3,
+ "3.5 FieldWorks/Paratext collaboration"
+ ],
+ [
+ 3,
+ "3.6 Linked files"
+ ],
+ [
+ 3,
+ "3.7 FieldWorks and FLEx Bridge versions"
+ ],
+ [
+ 4,
+ "3.7.1 FLEx Bridge mercurial compatibility"
+ ],
+ [
+ 3,
+ "3.8 FieldWorks project name"
+ ],
+ [
+ 2,
+ "4 Technical details"
+ ],
+ [
+ 3,
+ "4.1 Chorus Hub issues"
+ ],
+ [
+ 3,
+ "4.2 Using FieldWorks backups and Send/Receive"
+ ],
+ [
+ 3,
+ "4.3 FLEx backups and Send/Receive"
+ ],
+ [
+ 4,
+ "4.3.1 Using FLEx Backup with Send/Receive data"
+ ],
+ [
+ 4,
+ "4.3.2 Using FLEx Backup without Send/Receive data"
+ ],
+ [
+ 3,
+ "4.4 Viewing Send/Receive history"
+ ],
+ [
+ 3,
+ "4.5 Recovering lost data from S/R"
+ ],
+ [
+ 4,
+ "4.5.1 Repairing lost or damaged local repo"
+ ],
+ [
+ 4,
+ "4.5.2 Using Repository Utility"
+ ],
+ [
+ 4,
+ "4.5.3 Using S/R to repair or change data"
+ ],
+ [
+ 4,
+ "4.5.4 Dealing with defective repos"
+ ],
+ [
+ 4,
+ "4.5.5 Recovering data in difficult situations"
+ ],
+ [
+ 3,
+ "4.6 To switch a WeSay bridge user to another FLEx user"
+ ],
+ [
+ 3,
+ "4.7 FieldWorks and WeSay compatibility issues"
+ ],
+ [
+ 3,
+ "4.8 Modifying FLEx lists outside of FLEx"
+ ],
+ [
+ 4,
+ "4.? Unfinished notes"
+ ]
+ ],
+ "headings": 34
+ },
+ "Language Explorer/Training/Technical Notes on Interlinear Import.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Importing SFM Interlinear text"
+ ],
+ [
+ 3,
+ "Special intermediate file"
+ ],
+ [
+ 3,
+ "Constructing literal translations"
+ ],
+ [
+ 2,
+ "2 FieldWorks FLExText Interlinear XML"
+ ],
+ [
+ 3,
+ "Dealing with audio in interlinear"
+ ]
+ ],
+ "headings": 5
+ },
+ "Language Explorer/Training/Technical Notes on LinguaLinks Database Import.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Data model differences"
+ ],
+ [
+ 3,
+ "1.1 Subentry and minor entry differences"
+ ],
+ [
+ 3,
+ "1.2 LinguaLinks annotations"
+ ],
+ [
+ 4,
+ "Steps: Entry Notes"
+ ],
+ [
+ 3,
+ "1.3 Interlinear text sections"
+ ],
+ [
+ 3,
+ "1.4 Interlinear text collections"
+ ],
+ [
+ 3,
+ "1.5 Thesaurus links"
+ ],
+ [
+ 3,
+ "1.6 Nested sense limitations"
+ ],
+ [
+ 3,
+ "1.7 LinguaLinks sense media files"
+ ],
+ [
+ 3,
+ "1.8 Limitations with interlinear text baselines in multiple scripts"
+ ],
+ [
+ 3,
+ "1.9 Pictures and pronunciation files"
+ ],
+ [
+ 3,
+ "1.10 LinguaLinks subentry types"
+ ],
+ [
+ 2,
+ "2 Exporting data from LinguaLinks"
+ ],
+ [
+ 3,
+ "2.1 Steps: LinguaLinks export"
+ ],
+ [
+ 3,
+ "2.2 Steps: Listing fonts used in LinguaLinks"
+ ],
+ [
+ 2,
+ "3 Importing LinguaLinks data into Language Explorer"
+ ],
+ [
+ 3,
+ "3.1 Importing into existing Language Explorer data"
+ ],
+ [
+ 3,
+ "3.2 Steps: Importing LinguaLinks data"
+ ],
+ [
+ 2,
+ "4 Technical process flow of LinguaLinks import"
+ ],
+ [
+ 2,
+ "5 Advanced import problem solving"
+ ],
+ [
+ 3,
+ "5.1 Partial processing"
+ ],
+ [
+ 3,
+ "5.2 Step: Testing encoding converters"
+ ],
+ [
+ 3,
+ "5.3 Really large databases"
+ ],
+ [
+ 3,
+ "5.3 Converting bar codes"
+ ],
+ [
+ 3,
+ "5.3 Missing writing systems"
+ ]
+ ],
+ "headings": 25
+ },
+ "Language Explorer/Training/Technical Notes on SFM Database Import.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Mapping SFM fields to the FieldWorks Language Explorer model"
+ ],
+ [
+ 3,
+ "1.1 Basic field mapping"
+ ],
+ [
+ 3,
+ "1.2 Text fields"
+ ],
+ [
+ 3,
+ "1.3 List references"
+ ],
+ [
+ 3,
+ "1.4 Semantic domains"
+ ],
+ [
+ 3,
+ "1.5 Repeating fields"
+ ],
+ [
+ 3,
+ "1.6 Multiple items per field"
+ ],
+ [
+ 3,
+ "1.7 Lexical functions"
+ ],
+ [
+ 3,
+ "1.8 References to entries and senses"
+ ],
+ [
+ 3,
+ "1.9 Multiple uses of an SFM marker"
+ ],
+ [
+ 3,
+ "1.10 In-line markers, or character mapping"
+ ],
+ [
+ 3,
+ "1.11 Variants"
+ ],
+ [
+ 4,
+ "\\lx abc"
+ ],
+ [
+ 3,
+ "1.12 Subentries"
+ ],
+ [
+ 3,
+ "1.13 Nested senses"
+ ],
+ [
+ 3,
+ "1.14 Allomorph conditions"
+ ],
+ [
+ 3,
+ "1.15 Import residue"
+ ],
+ [
+ 3,
+ "1.15 Custom fields"
+ ],
+ [
+ 3,
+ "1.16 Affix markers"
+ ],
+ [
+ 3,
+ "1.17 Picture & Pronunciation files"
+ ],
+ [
+ 3,
+ "1.18 MDF supported fields"
+ ],
+ [
+ 3,
+ "1.19 Unsupported features"
+ ],
+ [
+ 2,
+ "2 Legacy to Unicode encoding conversion"
+ ],
+ [
+ 2,
+ "3 Importing SFM data into Language Explorer"
+ ],
+ [
+ 2,
+ "4 Resolving errors reported by the SFM readiness check"
+ ],
+ [
+ 3,
+ "4.1 Eliminating Internet Explorer warnings when launching ZEdit"
+ ],
+ [
+ 2,
+ "5 Technical process flow of SFM import"
+ ],
+ [
+ 2,
+ "6 Advanced import problem solving"
+ ],
+ [
+ 3,
+ "6.1 Extra processing during importing"
+ ],
+ [
+ 3,
+ "6.2 Handling non-MDF standards"
+ ],
+ [
+ 2,
+ "7 Language Explorer V3.0 (FieldWorks 6.0) bugs"
+ ]
+ ],
+ "headings": 31
+ },
+ "Language Explorer/Training/Technical Notes on Writing Systems.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Ingredients"
+ ],
+ [
+ 3,
+ "1.1 Encoding Converter"
+ ],
+ [
+ 3,
+ "1.2 Font"
+ ],
+ [
+ 3,
+ "1.3 MSKLC or Keyman keyboard"
+ ],
+ [
+ 2,
+ "2 Notes and Recommendations"
+ ],
+ [
+ 3,
+ "2.1 Encoding Converters"
+ ],
+ [
+ 3,
+ "2.2 Same Script for different languages?"
+ ],
+ [
+ 3,
+ "2.3 Multiple Writing Systems for the same language"
+ ],
+ [
+ 4,
+ "2.3.1 Creating and adding phonetic and phonemic writing systems to your project"
+ ],
+ [
+ 4,
+ "2.3.2 Creating an alternative writing system, e.g. Roman for a non-Roman script"
+ ],
+ [
+ 4,
+ "2.3.3 Reordering writing systems"
+ ],
+ [
+ 4,
+ "2.2.4 Current limitations on multiple scripts for interlinear baseline texts"
+ ],
+ [
+ 2,
+ "3 MSKLC and Keyman setup"
+ ],
+ [
+ 4,
+ "Keyman 9 and FieldWorks 8.0.6 and later Notes"
+ ],
+ [
+ 4,
+ "Notes for older versions of Keyman and FieldWorks"
+ ],
+ [
+ 4,
+ "Keyman 7 Notes"
+ ],
+ [
+ 2,
+ "4 Writing System Error Message"
+ ],
+ [
+ 3,
+ "Scenario 1"
+ ],
+ [
+ 3,
+ "Scenario 2"
+ ],
+ [
+ 3,
+ "Scenario 3"
+ ]
+ ],
+ "headings": 20
+ },
+ "Language Explorer/Utilities/AlloGenUserDocumentation.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 3,
+ "1.1 Invoking** **_Allomorph Generator_ from within** **_FLEx_"
+ ],
+ [
+ 3,
+ "1.2 Appearance"
+ ],
+ [
+ 2,
+ "2 Edit Operations tab"
+ ],
+ [
+ 3,
+ "2.1 Operation name and description"
+ ],
+ [
+ 3,
+ "2.2 Pattern section"
+ ],
+ [
+ 3,
+ "_2.2.1 Match_"
+ ],
+ [
+ 3,
+ "_2.2.2 Morph Types_"
+ ],
+ [
+ 3,
+ "_2.2.3 Category_"
+ ],
+ [
+ 3,
+ "2.3 Actions section"
+ ],
+ [
+ 3,
+ "_2.3.1 Replace operations_"
+ ],
+ [
+ 3,
+ "_2.3.2 Environments_"
+ ],
+ [
+ 3,
+ "_2.3.3 Stem Name_"
+ ],
+ [
+ 3,
+ "2.4 Apply operations to drop-down box"
+ ],
+ [
+ 3,
+ "2.5 Save changes button"
+ ],
+ [
+ 2,
+ "3 Run Operations tab"
+ ],
+ [
+ 2,
+ "4 Edit Replace Operations tab"
+ ],
+ [
+ 2,
+ "5 Applying operations"
+ ],
+ [
+ 2,
+ "6 Restarting** **_Allomorph Generator_"
+ ],
+ [
+ 2,
+ "7 Error messages"
+ ],
+ [
+ 2,
+ "8 Known problems"
+ ],
+ [
+ 2,
+ "9 Support"
+ ]
+ ],
+ "headings": 22
+ },
+ "Language Explorer/Utilities/PcPatrFLExUserDocumentation.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 3,
+ "1.1 Invoking** **_Use PC-PATR with FLEx_ from within** **_FLEx_"
+ ],
+ [
+ 3,
+ "1.2 Initial invocation"
+ ],
+ [
+ 3,
+ "1.3 Appearance"
+ ],
+ [
+ 2,
+ "2 Buttons"
+ ],
+ [
+ 3,
+ "2.1 PC-PATR grammar Browse button"
+ ],
+ [
+ 3,
+ "2.2 Rootgloss choices"
+ ],
+ [
+ 3,
+ "2.3 Advanced"
+ ],
+ [
+ 3,
+ "2.4 Help button"
+ ],
+ [
+ 3,
+ "2.5 Refresh texts button"
+ ],
+ [
+ 3,
+ "2.6 Disambiguate button"
+ ],
+ [
+ 3,
+ "2.7 Parse button"
+ ],
+ [
+ 2,
+ "3 Parsing a segment"
+ ],
+ [
+ 2,
+ "4 Disambiguating a text"
+ ],
+ [
+ 2,
+ "5 Restarting** **_Use PC-PATR with FLEx_"
+ ],
+ [
+ 3,
+ "4. remember which segment in that text you last selected.2"
+ ],
+ [
+ 2,
+ "6 Error messages"
+ ],
+ [
+ 2,
+ "7 Known problems"
+ ],
+ [
+ 2,
+ "8 Support"
+ ],
+ [
+ 2,
+ "A. The** **_PcPatr Browser_ tool"
+ ],
+ [
+ 2,
+ "A.1 Overview"
+ ],
+ [
+ 2,
+ "A.2 Right-to-left script"
+ ],
+ [
+ 2,
+ "A.3 Keyboard shortcuts"
+ ]
+ ],
+ "headings": 23
+ },
+ "Language Explorer/Utilities/silewp2007_002.pdf": {
+ "outline": [
+ [
+ 2,
+ "Abstract"
+ ],
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 2,
+ "2 Phonological Concepts"
+ ],
+ [
+ 2,
+ "3 Implementation Issues"
+ ],
+ [
+ 3,
+ "3.1 Syllabification and TBUs"
+ ],
+ [
+ 3,
+ "3.2 Lexical Representation"
+ ],
+ [
+ 3,
+ "3.3 Tone Rules"
+ ],
+ [
+ 4,
+ "3.3.1 Operations"
+ ],
+ [
+ 4,
+ "3.3.2 Operation Parameters"
+ ],
+ [
+ 4,
+ "3.3.3 Rule Application"
+ ],
+ [
+ 4,
+ "3.3.4 Edge Rules"
+ ],
+ [
+ 4,
+ "3.3.5 Conditions on Rules"
+ ],
+ [
+ 4,
+ "3.3.6 Example"
+ ],
+ [
+ 3,
+ "3.4 Orthographic Output"
+ ],
+ [
+ 2,
+ "4 Using TonePars with AMPLE"
+ ],
+ [
+ 4,
+ "(41) AMPLE **TonePars"
+ ],
+ [
+ 2,
+ "5 Conclusion"
+ ],
+ [
+ 2,
+ "Appendix A: List of Field Codes"
+ ],
+ [
+ 2,
+ "Appendix B: Annotated Syntax for Tone Rules"
+ ],
+ [
+ 2,
+ "Endnotes"
+ ],
+ [
+ 2,
+ "References"
+ ]
+ ],
+ "headings": 21
+ },
+ "Language Explorer/Utilities/ToneParsFLExUserDocumentation.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 3,
+ "1.1 Invoking** **_Use TonePars with FLEx_ from within** **_FLEx_"
+ ],
+ [
+ 3,
+ "1.2 Initial invocation"
+ ],
+ [
+ 3,
+ "1.3 Appearance"
+ ],
+ [
+ 2,
+ "2 Buttons and check boxes"
+ ],
+ [
+ 3,
+ "2.3 Trace Tone Processing check box"
+ ],
+ [
+ 3,
+ "2.4 Tracing Options button"
+ ],
+ [
+ 3,
+ "2.5 Show Log button"
+ ],
+ [
+ 3,
+ "2.6 Help button"
+ ],
+ [
+ 3,
+ "2.7 Verify Control File Information check box"
+ ],
+ [
+ 3,
+ "2.8 Ignore Context check box"
+ ],
+ [
+ 3,
+ "2.9 Refresh Texts button"
+ ],
+ [
+ 3,
+ "2.10 Parse this text button"
+ ],
+ [
+ 3,
+ "2.11 Parse this segment button"
+ ],
+ [
+ 2,
+ "3 Maximum analyses setting for** **_XAmple_"
+ ],
+ [
+ 2,
+ "4 Restarting** **_Use TonePars with FLEx_"
+ ],
+ [
+ 2,
+ "5 Known problems"
+ ],
+ [
+ 2,
+ "6 Output files"
+ ],
+ [
+ 2,
+ "7 Error messages"
+ ],
+ [
+ 2,
+ "8 Support"
+ ]
+ ],
+ "headings": 20
+ },
+ "Language Explorer/Utilities/VarGenUserDocumentation.pdf": {
+ "outline": [
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 3,
+ "1.1 Invoking** **_Variant Generator_ from within** **_FLEx_"
+ ],
+ [
+ 3,
+ "1.2 Appearance"
+ ],
+ [
+ 2,
+ "2 Edit Operations tab"
+ ],
+ [
+ 3,
+ "2.1 Operation name and description"
+ ],
+ [
+ 3,
+ "2.2 Pattern section"
+ ],
+ [
+ 3,
+ "_2.2.1 Match_"
+ ],
+ [
+ 3,
+ "_2.2.2 Morph Types_"
+ ],
+ [
+ 3,
+ "_2.2.3 Category_"
+ ],
+ [
+ 3,
+ "2.3 Actions section"
+ ],
+ [
+ 3,
+ "_2.3.1 Replace operations_"
+ ],
+ [
+ 4,
+ "Publish entry in"
+ ],
+ [
+ 3,
+ "_2.3.2 Variant types_"
+ ],
+ [
+ 3,
+ "_2.3.3 Show minor entry_"
+ ],
+ [
+ 3,
+ "_2.3.4 Publish entry in_"
+ ],
+ [
+ 3,
+ "2.4 Apply operations to drop-down box"
+ ],
+ [
+ 3,
+ "2.5 Save changes button"
+ ],
+ [
+ 2,
+ "3 Run Operations tab"
+ ],
+ [
+ 2,
+ "4 Edit Replace Operations tab"
+ ],
+ [
+ 2,
+ "5 Applying operations"
+ ],
+ [
+ 2,
+ "6 Restarting** **_Variant Generator_"
+ ],
+ [
+ 2,
+ "7 Error messages"
+ ],
+ [
+ 2,
+ "8 Known problems"
+ ],
+ [
+ 2,
+ "9 Support"
+ ]
+ ],
+ "headings": 24
+ },
+ "WW-ConceptualIntro/ConceptualIntroFLEx.pdf": {
+ "outline": [
+ [
+ 2,
+ "Abbreviations"
+ ],
+ [
+ 2,
+ "1 Introduction"
+ ],
+ [
+ 3,
+ "1.1 Key issues"
+ ],
+ [
+ 4,
+ "1.1.1 Inflection"
+ ],
+ [
+ 4,
+ "1.1.2 Derivation"
+ ],
+ [
+ 4,
+ "1.1.3 Ambiguity"
+ ],
+ [
+ 4,
+ "1.1.4 Epenthesis"
+ ],
+ [
+ 4,
+ "1.1.5 Discontinuous morphemes"
+ ],
+ [
+ 4,
+ "1.1.6 Infixation"
+ ],
+ [
+ 4,
+ "1.1.7 Reduplication"
+ ],
+ [
+ 4,
+ "1.1.8 Root and pattern morphology"
+ ],
+ [
+ 4,
+ "1.1.9 Metathesis"
+ ],
+ [
+ 4,
+ "1.1.10 Morphemes that may be null"
+ ],
+ [
+ 3,
+ "1.2 Tasks for any morphological parser"
+ ],
+ [
+ 2,
+ "2 Morphotactics"
+ ],
+ [
+ 3,
+ "2.1 Affixation"
+ ],
+ [
+ 4,
+ "2.1.1 Unclassified affixes"
+ ],
+ [
+ 4,
+ "2.1.2 Inflectional affixes"
+ ],
+ [
+ 4,
+ "_2.1.2.1 Simple example_"
+ ],
+ [
+ 4,
+ "_2.1.2.2 Optional affix slots_"
+ ],
+ [
+ 4,
+ "_2.1.2.3 Multiple templates_"
+ ],
+ [
+ 4,
+ "_2.1.2.4 Discontinuous morpheme_"
+ ],
+ [
+ 4,
+ "_2.1.2.5 Inflection and categories considerations_"
+ ],
+ [
+ 4,
+ "_2.1.2.6 Inflection classes_"
+ ],
+ [
+ 4,
+ "2.1.2.6.1 Inflection subclasses"
+ ],
+ [
+ 4,
+ "Figure 5. How to Create Isthmus Zapotec Inflection Classes."
+ ],
+ [
+ 4,
+ "2.1.2.6.2 Inflection classes and category organization"
+ ],
+ [
+ 4,
+ "_2.1.2.7 Agreement and other inflection features_"
+ ],
+ [
+ 4,
+ "_2.1.2.8 Inflection classes versus inflection features_"
+ ],
+ [
+ 4,
+ "_2.1.2.9 Underspecified inflectional affixes_"
+ ],
+ [
+ 4,
+ "2.1.3 Derivational affixes"
+ ],
+ [
+ 4,
+ "_2.1.3.1 Major category-changing derivational affixes_"
+ ],
+ [
+ 4,
+ "_2.1.3.2 Sub-category-changing derivational affixes_"
+ ],
+ [
+ 4,
+ "_2.1.3.3 Non-category-changing derivational affixes_"
+ ],
+ [
+ 4,
+ "_2.1.3.4 Inflection class and derivational affixes_"
+ ],
+ [
+ 4,
+ "2.1.3.4.1 Inflection class may change"
+ ],
+ [
+ 4,
+ "2.1.3.4.2 Inflection class does not change"
+ ],
+ [
+ 4,
+ "_2.1.3.5 Inflection Features and Derivational Affixes_"
+ ],
+ [
+ 4,
+ "_2.1.3.6 Category-changing derivational affixes and category organization_"
+ ],
+ [
+ 4,
+ "_2.1.3.7 Underspecified derivational affixes_"
+ ],
+ [
+ 4,
+ "2.1.4 Derivation outside of inflection"
+ ],
+ [
+ 4,
+ "2.1.5 Derivation versus inflection"
+ ],
+ [
+ 4,
+ "2.1.6 Exception “features”"
+ ],
+ [
+ 3,
+ "2.2 Stem compounding"
+ ],
+ [
+ 4,
+ "2.2.1 Headed compounds"
+ ],
+ [
+ 4,
+ "2.2.2 Non-headed compounds"
+ ],
+ [
+ 4,
+ "2.2.3 Incorporation"
+ ],
+ [
+ 4,
+ "_2.2.3.1 Incorporation as a simple headed compound_"
+ ],
+ [
+ 4,
+ "_2.2.3.2 Incorporation as a headed compound with override_"
+ ],
+ [
+ 4,
+ "2.2.4 Affixes between roots in compounds"
+ ],
+ [
+ 4,
+ "2.2.5 Compound rules and categories considerations"
+ ],
+ [
+ 4,
+ "2.2.6 Restricting the productivity of a compound rule"
+ ],
+ [
+ 3,
+ "2.3 Clitics"
+ ],
+ [
+ 4,
+ "Figure 23. How to Create a Clitic."
+ ],
+ [
+ 3,
+ "2.4 Ad hoc morpheme-oriented rules"
+ ],
+ [
+ 4,
+ "2.4.1 Creating morpheme-oriented ad hoc rules"
+ ],
+ [
+ 4,
+ "2.4.2 Grouping ad hoc morpheme rules"
+ ],
+ [
+ 4,
+ "Figure 25. How to Create a Group of Morpheme-oriented Ad Hoc Rules."
+ ],
+ [
+ 2,
+ "3 Morphophonemics"
+ ],
+ [
+ 3,
+ "3.1 Overview"
+ ],
+ [
+ 4,
+ "3.1.1 Phoneme sets"
+ ],
+ [
+ 4,
+ "_3.1.1.1 Phonological features_"
+ ],
+ [
+ 4,
+ "_3.1.1.2 Digraphs_"
+ ],
+ [
+ 4,
+ "_3.1.1.3 Tones_"
+ ],
+ [
+ 4,
+ "3.1.1.3.1 No forms conditioned by tone"
+ ],
+ [
+ 4,
+ "3.1.1.3.2 Forms conditioned by tone"
+ ],
+ [
+ 4,
+ "3.1.2 Natural classes"
+ ],
+ [
+ 4,
+ "Figure 29. How to Create Natural Classes."
+ ],
+ [
+ 4,
+ "3.1.3 Allomorph environments"
+ ],
+ [
+ 4,
+ "(65) / _ [C] / _ #"
+ ],
+ [
+ 4,
+ "3.1.4 Allomorph ordering"
+ ],
+ [
+ 4,
+ "_3.1.4.1 Free fluctuation_"
+ ],
+ [
+ 3,
+ "3.2 Reduplication"
+ ],
+ [
+ 4,
+ "3.2.1 Full reduplication"
+ ],
+ [
+ 4,
+ "_3.2.1.1 Writing the pattern for full reduplication_"
+ ],
+ [
+ 4,
+ "3.2.2 Partial reduplication"
+ ],
+ [
+ 4,
+ "_3.2.2.1 Writing the pattern for partial reduplication_"
+ ],
+ [
+ 3,
+ "3.3 Infixation"
+ ],
+ [
+ 4,
+ "3.3.1 Writing the infixation environment"
+ ],
+ [
+ 4,
+ "3.3.2 Infixation and root and pattern morphology"
+ ],
+ [
+ 3,
+ "3.4 Epenthesis"
+ ],
+ [
+ 3,
+ "3.5 Metathesis"
+ ],
+ [
+ 3,
+ "3.6 Morphemes that may be null"
+ ],
+ [
+ 3,
+ "3.7 Non-phonologically conditioned allomorphy"
+ ],
+ [
+ 4,
+ "3.7.1 Stem allomorphs conditioned by morpho-syntactic features"
+ ],
+ [
+ 4,
+ "Figure 38. How to Create and Use Stem Allomorph Labels."
+ ],
+ [
+ 4,
+ "3.7.2 Affix allomorphs conditioned by morpho-syntactic features"
+ ],
+ [
+ 3,
+ "3.8 Irregularly inflected forms"
+ ],
+ [
+ 3,
+ "3.9 Coalescence"
+ ],
+ [
+ 3,
+ "3.10 Ad hoc allomorph-oriented rules"
+ ],
+ [
+ 4,
+ "3.10.1 Creating ad hoc allomorph-oriented rules"
+ ],
+ [
+ 4,
+ "3.10.2 Grouping ad hoc allomorph rules"
+ ],
+ [
+ 2,
+ "4 Lexical entry considerations"
+ ],
+ [
+ 3,
+ "4.1 Allomorphs"
+ ],
+ [
+ 4,
+ "4.1.1 Null allomorphs"
+ ],
+ [
+ 4,
+ "4.1.2 Order of allomorphs within a lexical entry"
+ ],
+ [
+ 4,
+ "Figure 43. How to Create a Null Allomorph."
+ ],
+ [
+ 3,
+ "4.2 Morpheme types"
+ ],
+ [
+ 3,
+ "4.3 Circumfixes"
+ ],
+ [
+ 3,
+ "4.4 Senses/Glosses"
+ ],
+ [
+ 2,
+ "5 Other considerations"
+ ],
+ [
+ 3,
+ "5.1 Exceptional Case for Compound Rules"
+ ],
+ [
+ 2,
+ "6 The phonological rule-based parser"
+ ],
+ [
+ 3,
+ "6.1 Item and process"
+ ],
+ [
+ 4,
+ "6.1.1 Affix process rules"
+ ],
+ [
+ 4,
+ "_6.1.1.1 Reduplication as a process_"
+ ],
+ [
+ 4,
+ "6.1.1.1.1 Full reduplication as a process"
+ ],
+ [
+ 4,
+ "6.1.1.1.2 Partial reduplication as a process"
+ ],
+ [
+ 4,
+ "_6.1.1.2 Infixation as a process_"
+ ],
+ [
+ 4,
+ "_6.1.1.3 Circumfixation as a process_"
+ ],
+ [
+ 4,
+ "6.1.2 Phonological rules"
+ ],
+ [
+ 4,
+ "_6.1.2.1 “Regular” phonological rules_"
+ ],
+ [
+ 4,
+ "6.1.2.1.1 Epenthesis"
+ ],
+ [
+ 4,
+ "Figure 45. Selaru Epenthesis Rule."
+ ],
+ [
+ 4,
+ "6.1.2.1.2 Glide becomes a vowel"
+ ],
+ [
+ 4,
+ "Figure 46. Selaru Glide Vowel Rule."
+ ],
+ [
+ 4,
+ "6.1.2.1.3 Tone processing"
+ ],
+ [
+ 4,
+ "Figure 47. How to Handle Awngi Tone."
+ ],
+ [
+ 4,
+ "Figure 48. Awngi Docking Rule."
+ ],
+ [
+ 4,
+ "Figure 49. Awngi Deletion Rule."
+ ],
+ [
+ 4,
+ "6.1.2.1.4 Nasal assimilation"
+ ],
+ [
+ 4,
+ "Figure 51. How to Handle the Unspecified Nasal in Indonesian."
+ ],
+ [
+ 4,
+ "Figure 57. How to Use an Exception “Feature” for Unspecified Nasal in Indonesian."
+ ],
+ [
+ 4,
+ "_6.1.2.2 Constraining application of “regular” phonological rules_"
+ ],
+ [
+ 4,
+ "6.1.2.2.1 Rule applies only with certain categories"
+ ],
+ [
+ 4,
+ "6.1.2.2.2 Rule applies only with certain properties"
+ ],
+ [
+ 4,
+ "_6.1.2.3 Phonological metathesis rules_"
+ ],
+ [
+ 3,
+ "6.2 Tips for making the phonological rule-based parser work effectively."
+ ],
+ [
+ 4,
+ "6.2.1 Every phoneme used in the orthography must be defined as a phoneme"
+ ],
+ [
+ 4,
+ "6.2.2 The phonological features need to uniquely identify each phoneme"
+ ],
+ [
+ 4,
+ "6.2.3 Fully specify each phoneme"
+ ],
+ [
+ 4,
+ "6.2.4 Features used in a rule should be explicit"
+ ],
+ [
+ 4,
+ "6.2.5 Avoid using archiphonemes that are uppercase equivalents of a character in your orthography"
+ ],
+ [
+ 4,
+ "6.2.6 Make sure every affix process rule is complete"
+ ],
+ [
+ 4,
+ "6.2.7 Natural classes defined by phonemes may not work as expected"
+ ],
+ [
+ 3,
+ "6.3 Known limitations"
+ ],
+ [
+ 4,
+ "6.3.1 Affixes are tried only once per word"
+ ],
+ [
+ 4,
+ "6.3.2 Natural classes defined by segments may or may not work as expected"
+ ],
+ [
+ 4,
+ "6.3.3 Ambiguous digraphs and multigraphs may not work as expected"
+ ],
+ [
+ 2,
+ "References"
+ ]
+ ],
+ "headings": 140
+ }
+}
diff --git a/tools/reporting.py b/tools/reporting.py
new file mode 100644
index 00000000..c7cce459
--- /dev/null
+++ b/tools/reporting.py
@@ -0,0 +1,213 @@
+"""Canonical quality report model and renderers."""
+
+from __future__ import annotations
+
+import json
+from collections.abc import Iterable
+from dataclasses import asdict, dataclass
+from types import MappingProxyType
+
+from issue_catalog import ISSUE_CATALOG, policy_for
+
+LABELS = MappingProxyType({code: policy.label for code, policy in ISSUE_CATALOG.items()})
+
+
+@dataclass(frozen=True)
+class Issue:
+ code: str
+ message: str
+ path: str = ""
+ fatal: bool | None = None
+ provenance: str | None = None
+ detail: object = None
+
+ def __post_init__(self) -> None:
+ original_code = self.code
+ code, policy = policy_for(original_code)
+ unknown = code == "unknown_issue" and original_code != code
+ if unknown:
+ object.__setattr__(self, "message", f"[{original_code}] {self.message}")
+ object.__setattr__(self, "code", code)
+ # The legacy constructor accepts these fields for source compatibility,
+ # but policy is always selected solely by the catalog code.
+ object.__setattr__(self, "fatal", policy.fatal)
+ object.__setattr__(self, "provenance", policy.provenance)
+
+ @property
+ def severity(self) -> str:
+ return "error" if self.fatal else "warning"
+
+ @property
+ def label(self) -> str:
+ return ISSUE_CATALOG[self.code].label
+
+ def as_dict(self) -> dict:
+ value = asdict(self)
+ value.update(label=self.label, severity=self.severity)
+ return value
+
+
+class Report:
+ def __init__(self, issues: Iterable[Issue] = (), metadata: dict | None = None) -> None:
+ self.issues = list(issues)
+ self.metadata = dict(metadata or {})
+
+ def add(self, issue: Issue) -> Issue:
+ self.issues.append(issue)
+ return issue
+
+ def extend(self, issues: Iterable[Issue]) -> None:
+ self.issues.extend(issues)
+
+ @property
+ def fatal(self) -> bool:
+ return any(issue.fatal for issue in self.issues)
+
+ def as_dict(self) -> dict:
+ by_code: dict[str, int] = {}
+ for issue in self.issues:
+ by_code[issue.code] = by_code.get(issue.code, 0) + 1
+ return {
+ "corpus": self.metadata,
+ "summary": {
+ "total": len(self.issues),
+ "fatal": sum(issue.fatal for issue in self.issues),
+ "advisory": sum(not issue.fatal for issue in self.issues),
+ "by_code": by_code,
+ },
+ "issues": [issue.as_dict() for issue in self.issues],
+ }
+
+ def to_readme(self) -> str:
+ lines = ["## Quality report", "", "| Check | Severity | Count |", "| --- | --- | ---: |"]
+ counts: dict[tuple[str, str], int] = {}
+ for issue in self.issues:
+ key = (issue.label, issue.severity)
+ counts[key] = counts.get(key, 0) + 1
+ for (label, severity), count in sorted(counts.items()):
+ lines.append(f"| {label} | {severity} | {count} |")
+ if not counts:
+ lines.append("| None | — | 0 |")
+ return "\n".join(lines)
+
+ def to_json(self) -> str:
+ return json.dumps(self.as_dict(), indent=2, ensure_ascii=False) + "\n"
+
+ @staticmethod
+ def _markdown_cell(value: object, *, code: bool = False) -> str:
+ if value is None or value == "":
+ return "—"
+ if not isinstance(value, str):
+ value = json.dumps(value, ensure_ascii=False)
+ escaped = (value.replace("&", "&").replace("<", "<")
+ .replace(">", ">").replace("\\", "\")
+ .replace("`", "`").replace("[", "[")
+ .replace("]", "]").replace("|", "\\|")
+ .replace("\r\n", " ").replace("\r", " ")
+ .replace("\n", " "))
+ if code:
+ return f"`{escaped}`"
+ return escaped
+
+ @staticmethod
+ def _repair_context(issue: Issue) -> str:
+ if issue.provenance != "source":
+ return "exporter"
+ normalized = issue.path.replace("\\", "/").casefold()
+ if normalized.startswith("pdf/") or normalized.endswith(".pdf"):
+ return "PDF source"
+ return "RoboHelp"
+
+ def to_markdown(self) -> str:
+ """Render the canonical report for source authors and maintainers."""
+ data = self.as_dict()
+ corpus = data["corpus"]
+ summary = data["summary"]
+ lines = [
+ "# Author quality report", "",
+ (
+ "This report explains every source or export finding and how to repair it. "
+ "Machine-readable detail is in [author-report.json](author-report.json)."
+ ), "",
+ "## Corpus", "", "| Item | Value |", "| --- | ---: |",
+ f"| Source ref | {self._markdown_cell(corpus.get('source_ref'), code=True)} |",
+ f"| CHMs | {corpus.get('chm_count', 0)} |",
+ f"| Topics | {corpus.get('topic_count', 0)} |",
+ f"| Images | {corpus.get('image_count', 0)} |",
+ f"| PDFs | {corpus.get('pdf_count', 0)} |", "",
+ "## Summary", "", "| Severity | Count |", "| --- | ---: |",
+ f"| Fatal errors | {summary['fatal']} |",
+ f"| Advisories | {summary['advisory']} |",
+ f"| Total | {summary['total']} |", "",
+ ]
+ grouped: dict[str, list[Issue]] = {}
+ for issue in self.issues:
+ grouped.setdefault(issue.code, []).append(issue)
+ for code in sorted(
+ grouped,
+ key=lambda item: (
+ not ISSUE_CATALOG[item].fatal, ISSUE_CATALOG[item].label.casefold(), item,
+ ),
+ ):
+ policy = ISSUE_CATALOG[code]
+ issues = grouped[code]
+ contexts = {self._repair_context(issue) for issue in issues}
+ where = "/".join(sorted(contexts))
+ lines.extend([
+ f"## {policy.label} (`{code}`)", "",
+ f"- **Severity:** {'fatal error' if policy.fatal else 'advisory'}",
+ f"- **Owner:** {where}",
+ f"- **Count:** {len(issues)}",
+ f"- **How to fix in {where}:** {policy.guidance}", "",
+ "| Source or generated path | Problem | Evidence |",
+ "| --- | --- | --- |",
+ ])
+ for issue in issues:
+ lines.append(
+ f"| {self._markdown_cell(issue.path, code=True)} "
+ f"| {self._markdown_cell(issue.message)} "
+ f"| {self._markdown_cell(issue.detail)} |"
+ )
+ lines.append("")
+ if not grouped:
+ lines.extend(["## Findings", "", "No findings.", ""])
+ return "\n".join(lines)
+
+ def to_console(self) -> str:
+ fatal = [issue for issue in self.issues if issue.fatal]
+ advisory = [issue for issue in self.issues if not issue.fatal]
+ lines = [(
+ f"quality report: {len(self.issues)} issue(s), "
+ f"{len(fatal)} fatal, {len(advisory)} advisory"
+ )]
+ lines.append(f"fatal issues ({len(fatal)}):")
+ if not fatal:
+ lines.append(" none")
+ for issue in fatal:
+ where = f" [{issue.path}]" if issue.path else ""
+ lines.append(f" FATAL {issue.label}{where}: {issue.message}")
+
+ counts: dict[tuple[str, str], int] = {}
+ for issue in advisory:
+ key = (issue.code, issue.label)
+ counts[key] = counts.get(key, 0) + 1
+ if counts:
+ lines.append(
+ f"advisories: {len(advisory)} issue(s) in {len(counts)} kind(s); "
+ "see author-report.json for details"
+ )
+ for (_, label), count in sorted(
+ counts.items(), key=lambda item: (item[0][1], item[0][0])
+ ):
+ lines.append(f" WARN {label}: {count}")
+ else:
+ lines.append("advisories: none")
+ return "\n".join(lines)
+
+
+def make_issue(code: str, message: str, path: str = "", detail: object = None) -> Issue:
+ """Construct an issue using the catalog, safely handling new producer codes."""
+ return Issue(str(code), str(message), str(path), detail=detail)
+
+
+__all__ = ["LABELS", "Issue", "Report", "make_issue"]
diff --git a/tools/source_safety.py b/tools/source_safety.py
new file mode 100644
index 00000000..7f7f9bcc
--- /dev/null
+++ b/tools/source_safety.py
@@ -0,0 +1,158 @@
+"""Safe, deterministic discovery of repository source files."""
+
+from __future__ import annotations
+
+import os
+import stat
+from pathlib import Path
+
+
+class SourceSafetyError(ValueError):
+ """A source root or source input violates the repository boundary."""
+
+
+def _absolute_lexical(path: Path | str) -> Path:
+ return Path(os.path.abspath(os.fspath(Path(path).expanduser())))
+
+
+def _is_link(path: Path) -> bool:
+ is_junction = getattr(path, "is_junction", lambda: False)
+ return path.is_symlink() or is_junction()
+
+
+def _first_link(path: Path) -> Path | None:
+ current = Path(path.anchor)
+ for part in path.parts[1:]:
+ current /= part
+ if _is_link(current):
+ return current
+ return None
+
+
+def first_link_in_path(path: Path | str) -> Path | None:
+ """Return the first symlink/junction in a lexical path chain."""
+ return _first_link(_absolute_lexical(path))
+
+
+def validate_source_tree(root: Path) -> Path:
+ """Validate a lexical tree without following links or escaping root."""
+ lexical_root = _absolute_lexical(root)
+ if (link := first_link_in_path(lexical_root)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction tree root: {link}")
+ try:
+ resolved_root = lexical_root.resolve(strict=True)
+ except OSError as exc:
+ raise SourceSafetyError(f"source tree does not exist: {root}") from exc
+ if not resolved_root.is_dir():
+ raise SourceSafetyError(f"source tree is not a directory: {root}")
+
+ pending = [lexical_root]
+ while pending:
+ current = pending.pop()
+ try:
+ entries = sorted(current.iterdir(), key=lambda item: (item.name.casefold(), item.name))
+ except OSError as exc:
+ raise SourceSafetyError(f"cannot inspect source tree: {current}") from exc
+ for entry in entries:
+ if (link := first_link_in_path(entry)) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction in source tree: {link}")
+ try:
+ mode = entry.stat(follow_symlinks=False).st_mode
+ except OSError as exc:
+ raise SourceSafetyError(f"cannot inspect source tree entry: {entry}") from exc
+ if stat.S_ISDIR(mode):
+ _resolved_inside(entry, resolved_root)
+ pending.append(entry)
+ elif stat.S_ISREG(mode):
+ _resolved_inside(entry, resolved_root)
+ return lexical_root
+
+
+def _resolved_inside(path: Path, root: Path) -> Path:
+ try:
+ resolved = path.resolve(strict=True)
+ except OSError as exc:
+ raise SourceSafetyError(f"cannot resolve source path: {path}") from exc
+ if resolved != root and root not in resolved.parents:
+ raise SourceSafetyError(f"source path resolves outside repository root: {path}")
+ return resolved
+
+
+def discover_source_files(
+ root: Path,
+ *,
+ suffixes: set[str],
+ recursive: bool,
+ exclude_dirs: set[str] | None = None,
+) -> list[Path]:
+ """Discover regular source files without following links.
+
+ Candidate files and directories that will be traversed are checked for
+ links and for a resolved path outside the resolved root. Irrelevant files
+ and excluded directories are ignored before those checks, so an unrelated
+ link cannot abort discovery or become an input boundary.
+ """
+ output_root = (
+ Path(os.path.normpath(os.fspath(Path(root).expanduser())))
+ if not Path(root).expanduser().is_absolute()
+ else _absolute_lexical(root)
+ )
+ lexical_root = _absolute_lexical(root)
+ if _first_link(lexical_root) is not None:
+ raise SourceSafetyError(f"refusing symlink/junction source root: {root}")
+ try:
+ resolved_root = lexical_root.resolve(strict=True)
+ except OSError as exc:
+ raise SourceSafetyError(f"source root does not exist: {root}") from exc
+ if not resolved_root.is_dir():
+ raise SourceSafetyError(f"source root is not a directory: {root}")
+
+ wanted = {suffix.casefold() if suffix.startswith(".") else f".{suffix.casefold()}"
+ for suffix in suffixes}
+ # Repository metadata is never a source tree, even when callers do not
+ # provide an explicit exclusion set. Additional exclusions are additive;
+ # callers cannot accidentally opt .git back into traversal.
+ excluded = {".git"} | {name.casefold() for name in (exclude_dirs or set())}
+ pending = [lexical_root]
+ discovered: list[tuple[Path, Path]] = []
+ while pending:
+ current = pending.pop()
+ try:
+ entries = sorted(current.iterdir(), key=lambda item: (item.name.casefold(), item.name))
+ except OSError as exc:
+ raise SourceSafetyError(f"cannot inspect source directory: {current}") from exc
+ for entry in entries:
+ if entry.name.casefold() in excluded:
+ continue
+ if _is_link(entry):
+ if entry.suffix.casefold() in wanted and not entry.is_dir():
+ raise SourceSafetyError(f"refusing symlink/junction source input: {entry}")
+ continue
+ mode = entry.stat(follow_symlinks=False).st_mode
+ if stat.S_ISDIR(mode):
+ if recursive:
+ _resolved_inside(entry, resolved_root)
+ pending.append(entry)
+ continue
+ if not stat.S_ISREG(mode):
+ continue
+ if entry.suffix.casefold() not in wanted:
+ continue
+ _resolved_inside(entry, resolved_root)
+ discovered.append((entry.relative_to(lexical_root), output_root / entry.relative_to(lexical_root)))
+
+ return sorted(
+ (path for _, path in discovered),
+ key=lambda item: (
+ item.relative_to(output_root).as_posix().casefold(),
+ item.relative_to(output_root).as_posix(),
+ ),
+ )
+
+
+__all__ = [
+ "SourceSafetyError",
+ "discover_source_files",
+ "first_link_in_path",
+ "validate_source_tree",
+]
diff --git a/tools/survey.py b/tools/survey.py
new file mode 100644
index 00000000..1f7ce4e6
--- /dev/null
+++ b/tools/survey.py
@@ -0,0 +1,429 @@
+"""Survey the FwHelps corpus: what is actually in the CHM and the PDFs.
+
+This is a read-only reconnaissance pass, not the converter. It answers the
+questions we need settled before designing the markdown export:
+ - how many topics, how big, how deep is the TOC
+ - which topics are reachable from the TOC and which are orphans
+ - what HTML constructs and CSS classes actually appear (what must map to md)
+ - how many internal links resolve, and where the broken ones point
+ - which PDFs carry real text vs. scanned images
+
+Usage: python tools/survey.py [--repo .] [--work DIR] [--json report.json]
+"""
+
+from __future__ import annotations
+
+import argparse
+import collections
+import html
+import json
+import re
+import sys
+from html.parser import HTMLParser
+from pathlib import Path
+from urllib.parse import unquote, urldefrag
+
+sys.path.insert(0, str(Path(__file__).parent))
+from chm_extract import extract
+
+# ---------------------------------------------------------------- sitemap ---
+
+class SitemapParser(HTMLParser):
+ """Parses the HTML Help sitemap format shared by .hhc (TOC) and .hhk (index).
+
+ Structure is