diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index af6ce84..a1ec998 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -56,10 +56,7 @@ jobs: python-version: ${{ matrix.python }} - run: pip install -r requirements.txt - name: Run the tests - # pytest exits 5 when it finds no tests. That is expected until the - # first work package adds some, so only that exit code is tolerated. - shell: bash - run: python3 -m pytest -q || [ $? -eq 5 ] + run: python3 -m pytest -q clean-clone: name: clean-clone diff --git a/.gitignore b/.gitignore index eea3a17..5ae9278 100644 --- a/.gitignore +++ b/.gitignore @@ -1,4 +1,6 @@ .DS_Store venv/ +.venv/ +.pytest_cache/ __pycache__/ *.pyc diff --git a/README.md b/README.md index 50bf467..413be06 100644 --- a/README.md +++ b/README.md @@ -2,11 +2,17 @@ Skills, pipelines, and technique write-ups for building AI-agent systems that stay correct as they scale — built by using Claude Code every day, not by reading about it. +**The 60-second tour.** + +- **What this is:** working examples of directing an AI coding agent and then checking what it built, with tests and written grades. Not a claim to machine-learning engineering. +- **See it work:** the job assessment system runs from a clean clone with no model and no API key. The quickstart is in [`pipelines/job-assessment/README.md`](pipelines/job-assessment/README.md). +- **The one file to read first:** [`pipelines/docs-pipeline/README.md`](pipelines/docs-pipeline/README.md), the documentation pipeline. + ## Start here: the best things to look at -**1. [`pipelines/docs-pipeline`](pipelines/docs-pipeline/)** is the strongest piece in this repo. It is a chain of Claude Code skills that takes a documentation change from the first request through structure review, voice, grammar, links, visuals, expert review, and publishing, with a real gate between every stage. I used a version of it every working day on real documentation. If you only open one thing, open this, and start with its [README](pipelines/docs-pipeline/README.md) and [ARCHITECTURE](pipelines/docs-pipeline/ARCHITECTURE.md). +**1. [`pipelines/docs-pipeline`](pipelines/docs-pipeline/)** is the strongest piece in this repo, and a generic export of the version I used. It is a chain of Claude Code skills that takes a documentation change from the first request through structure review, voice, grammar, links, visuals, expert review, and publishing, with a real gate between every stage. I used a version of it every working day on real documentation. If you only open one thing, open this, and start with its [README](pipelines/docs-pipeline/README.md) and [ARCHITECTURE](pipelines/docs-pipeline/ARCHITECTURE.md). -**2. [`pipelines/job-assessment`](pipelines/job-assessment/)** scores a job posting against your own written criteria and gives a plain verdict: Apply, Apply with reservations, or Skip. A model reads the posting and writes down findings, quoting the posting for every claim. Ordinary code then checks those quotes, does the scoring and picks the verdict, so the same findings always give the same answer. It was built by directing Claude Code, not hand-typed. What is shown: the whole chain runs offline on three invented postings with no model, no network and no API key, and the tests in that folder (run again in a fresh clone by CI) compare its scores, verdicts and saved output files to written-down expected values. A separate run with the `claude` CLI on Claude Sonnet is saved in [`tests/`](pipelines/job-assessment/tests/): the interview built a file that passed the checker, and the assessment gave the expected verdicts on 2 of 2 postings in its second attempt, after the first attempt got one verdict wrong ([both records are kept](pipelines/job-assessment/tests/e2e-output-run1.md)). What is not shown: that a verdict predicts getting hired, or that it beats any other method. No real postings or real person's data are in it. Its README starts with a five-minute quickstart from a clean clone, and every doc in the folder ends with prompts tested on Claude Sonnet and Claude Haiku. +**2. [`pipelines/job-assessment`](pipelines/job-assessment/)** scores a job posting against your own written criteria and gives a plain verdict: Apply, Apply with reservations, or Skip. A model reads the posting and writes down findings, quoting the posting for every claim. Ordinary code then checks those quotes, does the scoring and picks the verdict, so the same findings always give the same answer. It was built by directing Claude Code, not hand-typed. What is shown: the whole chain runs offline on three invented postings with no model, no network and no API key, and the tests in that folder (run again in a fresh clone by CI) compare its scores, verdicts and saved output files to written-down expected values. A separate run with the `claude` CLI on Claude Sonnet is saved in [`tests/`](pipelines/job-assessment/tests/): the interview built a file that passed the checker, and the assessment gave the expected verdicts on 2 of 2 postings in its second attempt, after the first attempt got one verdict wrong ([both records are kept](pipelines/job-assessment/tests/e2e-output-run1.md)). What is not shown: that a verdict predicts getting hired, or that it beats any other method. No real postings or real person's data are in it. Its README starts with a five-minute quickstart from a clean clone, and every README and guide in the folder ends with prompts tested on Claude Sonnet and Claude Haiku. **3. [`patterns/block-and-tell-hooks.md`](patterns/block-and-tell-hooks.md)** explains how to make an AI agent follow a rule every time instead of hoping it remembers. It is the idea behind most of the guards in my own setup. @@ -20,12 +26,14 @@ Every piece in here started as a real problem: a rule that kept getting skipped, ## What's inside -**`skills/`** — self-contained Claude Code skills. Each one is a single markdown file (plus a README) that defines a repeatable, well-scoped task for an AI agent to carry out — a readability pass on documentation, a way to capture an in-progress plan so it survives a context reset, an audit that checks whether an agent's own guardrails are actually strong enough to trust. +**`skills/`** — two self-contained Claude Code skills, each in its own folder with a `SKILL.md` that defines one repeatable, well-scoped task for an AI agent: `docs-readability-check` (a readability pass on documentation) and `plan-this` (capture an in-progress plan so it survives a context reset). `plan-this` also has a README. -**`pipelines/`** — multi-stage systems, not single tasks. The anchor piece here is a documentation-engineering pipeline: a chain of skills that takes a raw content change through structure review, voice/style checks, grammar, link and visual verification, and a subject-matter-expert review gate before anything publishes. This is a first pass, still being developed further — see `pipelines/docs-pipeline/` for its own README and current state. +**`pipelines/`** — multi-stage systems, not single tasks. The anchor piece here is a documentation-engineering pipeline: a chain of skills that takes a raw content change through structure review, voice/style checks, grammar, link and visual verification, and a subject-matter-expert review gate before anything publishes. See `pipelines/docs-pipeline/` for its own README and current state. The second pipeline is [`pipelines/job-assessment/`](pipelines/job-assessment/), which scores a job posting against written criteria. **`patterns/`** — written technique docs for ideas that are more valuable described in prose than shipped as literal runnable code, either because the real implementation is too specific to one project to be useful as-is, or because the idea itself is the point. Covers: how to make an "always do X first" instruction actually reliable instead of hoped-for (block-and-tell hooks), how to keep an agent's standing instructions from becoming an unmaintainable single file as they grow (rules-index architecture), and how to give an agent memory that survives months of use without turning into an unreadable dump (typed, size-bounded memory). +**Prompt blocks.** The README and guides of [`pipelines/docs-pipeline/`](pipelines/docs-pipeline/) and [`pipelines/job-assessment/`](pipelines/job-assessment/) and the three docs in `patterns/` each end with a "Prompt for your AI model" block, tested on Claude Sonnet and Claude Haiku. This top-level README, `ROADMAP.md` and the two skills do not have one. + ## How the pieces relate A skill is one task. A pipeline is several skills chained with real gates between them (nothing moves to the next stage until the current one passes). A pattern is the idea behind a mechanism, written down so it can be rebuilt in a different codebase without copying code that won't fit. @@ -36,4 +44,4 @@ Everything here was built using Claude Code, directed and reviewed by a human, n ## What's next -See `ROADMAP.md` for what's planned for the next batch — genericized versions of a few more utility scripts, further development on the documentation pipeline, and a handful of additional skills that need a lighter cleanup pass before they're ready to publish. +See `ROADMAP.md` for what is done and what is planned next: more genericized utility scripts, a fourth pattern doc, and a handful of additional skills that need a lighter cleanup pass before they are ready to publish. diff --git a/ROADMAP.md b/ROADMAP.md index 2c4f1a9..9d88aee 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -1,8 +1,8 @@ # Roadmap -What's here now is a first batch. This is what's planned next, so it's clear what's deliberately not done yet versus what's missing by accident. +What's here now is a first batch. This is what is done, and what's planned next, so it's clear what's deliberately not done yet versus what's missing by accident. -## In progress +## Done **Job assessment system.** Built in small pull requests under `pipelines/job-assessment/`. The scripts, schemas, fictional sample data, both skills and the offline end-to-end run are merged and tested; the earlier `tools/job-fit-screen` it replaces has been removed. `ARCHITECTURE.md`, `SETUP.md` and the prompt blocks, tested on Claude Sonnet and Claude Haiku, are merged too, so every item on its status list is done. @@ -17,7 +17,7 @@ What's here now is a first batch. This is what's planned next, so it's clear wha **A fourth pattern doc: always-read-first / write-back state files.** A convention for keeping a piece of live state (a machine's current status, a project's current phase) trustworthy across many separate agent sessions touching it — read the state file before acting, write back immediately after any change, and enforce both halves with hooks rather than hoping the instruction gets followed every time. -**Further development on the documentation pipeline.** The version in `pipelines/docs-pipeline/` is a first cleaned pass, not the finished shape. Ongoing work on it will land here as it develops. +**Further development on the documentation pipeline.** The version in `pipelines/docs-pipeline/` is a generic export of the version I used, not the finished shape. Ongoing work on it will land here as it develops. **A handful of additional skills**, pending a lighter genericization pass: - An AI-writing-tell detector — scans prose for patterns that read as machine-written and proposes rewrites, backed by a pattern registry with a cited source per rule, dated retirements for tells that stopped working, a documented false-positive case per rule, and a scheduled refresh step that diffs several outside source authorities to keep the pattern list current. Two small genericization items: a hardcoded personal file path and a house-style config tuned to one person's preferences. @@ -34,7 +34,3 @@ What's here now is a first batch. This is what's planned next, so it's clear wha - Anything narratively specific to one product's fictional world or brand voice — general technique only, not product content - Anything still under an employer's confidentiality terms without explicit permission to publish - Scratch/throwaway scripts that were never meant to be reused - -## Not yet decided - -Whether to link this repo from a resume, portfolio site, or LinkedIn — that's a deliberate later decision, not part of standing this repo up. diff --git a/patterns/block-and-tell-hooks.md b/patterns/block-and-tell-hooks.md index 438c0fb..ee83c75 100644 --- a/patterns/block-and-tell-hooks.md +++ b/patterns/block-and-tell-hooks.md @@ -146,4 +146,4 @@ My rule is: [YOUR "ALWAYS DO X FIRST" RULE]. My tool is: [YOUR AGENT TOOL]. Write the guard and marker scripts for my rule in my tool's hook format. Then give me a three-step test plan: one test where the guard must block, one where it must allow after the marker is written, and one where I confirm the release path works. Tell me what output proves each test passed. ``` -**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. +**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. I did not save those answers, so there is no record to read here, unlike the saved runs in `pipelines/job-assessment/tests/`. diff --git a/patterns/rules-index-architecture.md b/patterns/rules-index-architecture.md index 95a9453..4467c4e 100644 --- a/patterns/rules-index-architecture.md +++ b/patterns/rules-index-architecture.md @@ -165,4 +165,4 @@ My instructions file is pasted below. My tool is: [YOUR AGENT TOOL]. Propose a split into domain files with an index table, and tell me which files should be excluded from startup loading. Then give me a way to test that the split worked: how to measure total loaded size before and after, and one question to ask the agent that only a specific rule file can answer. ``` -**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. +**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. I did not save those answers, so there is no record to read here, unlike the saved runs in `pipelines/job-assessment/tests/`. diff --git a/patterns/typed-memory-system.md b/patterns/typed-memory-system.md index 77aab6d..9a14150 100644 --- a/patterns/typed-memory-system.md +++ b/patterns/typed-memory-system.md @@ -64,11 +64,11 @@ Run the test suite before merging. ### 3. Know the real limits -In Claude Code, the memory index is read in full at session start, and it has two limits that apply separately: **200 lines** and **25,000 bytes**. Whichever you hit first is the cutoff. +In Claude Code, the memory index is read in full at session start. In the version I use, I observed two limits on it that apply separately: **200 lines** and **25,000 bytes**. Whichever you hit first is the cutoff. I found the two numbers by reading them out of the installed Claude Code program, version 2.1.263. I have not re-confirmed them on later versions: on 2026-10-01, with version 2.1.287 installed, my check script could not find them in the program and fell back to these same values. So read them as numbers that were true for one version, and test your own. Three details that are easy to miss: -- **The limits apply to every file, not just the index.** A topic file over 200 lines gets cut the same way. +- **Topic files may be limited too, but I have not tested that.** Topic files may be read on demand rather than at session start, in which case a limit on them would only matter when one is opened, and it may not apply at all. I have not tested what your version does. Keeping topic files short is cheap, so I do it anyway. - **Going over is loud, not silent.** The harness cuts at a line boundary and adds a visible warning naming the file and the limit. Everything past the cut is still missing from that session, so treat the warning as a fault to fix, not noise. - **Measure bytes, not characters.** The limit is in bytes. Emoji and accented letters are several bytes each, so a character count undercounts. Use `wc -c file`, never a script that counts decoded characters. @@ -80,12 +80,12 @@ When the main index is capped, memories that no longer fit do not get deleted. T ```markdown ## More memories: topic indexes -- [Git and PRs](INDEX-git.md) 14 entries -- [Todoist](INDEX-todoist.md) 5 entries -- [Writing and output](INDEX-writing.md) 9 entries +- [Git and PRs](INDEX-git.md) (example: 8 entries) +- [Todoist](INDEX-todoist.md) (example: 3 entries) +- [Writing and output](INDEX-writing.md) (example: 6 entries) ``` -The agent reads the one that matches the task. The cost is that these are not auto-loaded, so the main index must say they exist and when to open them. In the real setup this tier holds a dozen-plus files and is the most important change that came after the first version of this pattern. +The agent reads the one that matches the task. The cost is that these are not auto-loaded, so the main index must say they exist and when to open them. In my own setup this tier grew to several files, and it was the most important change that came after the first version of this pattern. ### 5. A weekly automatic check @@ -137,7 +137,7 @@ A single-session tool with no saved state. It is solving index growth over month ## How this was checked -Every claim was compared against a working Claude Code memory folder: the four types (counted from file frontmatter), the index size against both limits, the topic index files, the weekly cron entry and its wrapper, and the written compaction steps. The limit numbers are the working values from that setup. They are not read from the program at run time, so re-check them for your version. The example files were written for this page. +Every claim was compared against a working Claude Code memory folder: the four types (counted from file frontmatter), the index size against both limits, the topic index files, the weekly cron entry and its wrapper, and the written compaction steps. The limit numbers are the values I observed in Claude Code 2.1.263. They are not read from the program at run time, and I could not re-confirm them on version 2.1.287, so re-check them for your version. The claim that topic files are limited is not checked here at all. The example files were written for this page. ## Prompt for your AI model @@ -171,4 +171,4 @@ My tool is: [YOUR AGENT TOOL]. Here are five things I want it to remember: [LIST Sort each into a memory type, write the topic files and the index for them, and then write a short script that checks the index against a size limit in bytes. Finish with a test: how I check the next session actually loaded the index, and how I confirm the size check fails when I add too much. ``` -**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. +**How these prompts were checked.** Each of the three prompts was run once with a small model (Claude Haiku) through the `claude` command line, with the full text of this file pasted in and sample details filled in. All three gave an on-topic answer that matched what this file says. In two runs a placeholder was left unfilled by my test setup, and the model noticed and said so or asked for the missing text instead of making something up. That is the behavior you want. One run per prompt is a light check, not a benchmark, so read the answers critically. I did not save those answers, so there is no record to read here, unlike the saved runs in `pipelines/job-assessment/tests/`. diff --git a/pipelines/job-assessment/ARCHITECTURE.md b/pipelines/job-assessment/ARCHITECTURE.md index aaef4d0..751b53d 100644 --- a/pipelines/job-assessment/ARCHITECTURE.md +++ b/pipelines/job-assessment/ARCHITECTURE.md @@ -90,7 +90,7 @@ Everything about the person lives in one file, `career-profile.yaml`. Its contra **Warnings (printed, exit stays 0):** a skill scored 3 or higher with no evidence, evidence with no dates, a strong must-have with no reason, an empty perks list, and a lane with no must-haves of its own. -**What it cannot see.** The validator checks shape and consistency, not truth. A file can pass every check and still overstate what the person can show, or say less than the person means. Those are judgment gaps. The "find gaps" prompt in `intake/README.md` asks a model to look for them, and `fixtures/gappy-profile.yaml` passes the validator with some planted for testing. +**What it cannot see.** The validator checks shape and consistency, not truth. A file can pass every check and still overstate what the person can show, or say less than the person means. Those are judgment gaps. The "find gaps" prompt in `intake/README.md` asks a model to look for them, and `fixtures/gappy-profile.yaml` and `fixtures/held-out-gappy-profile.yaml` pass the validator with some planted for testing. The prompt was tuned on the first, so the second is the fairer test. ## The findings file and the checks on it diff --git a/pipelines/job-assessment/README.md b/pipelines/job-assessment/README.md index 41ecd83..8227f28 100644 --- a/pipelines/job-assessment/README.md +++ b/pipelines/job-assessment/README.md @@ -2,7 +2,7 @@ **What it is.** A small system that reads a job posting against your own written criteria and gives a plain verdict: **Apply**, **Apply with reservations**, or **Skip**. A model reads the posting and writes down what it found, quoting the posting for every claim. Ordinary scripts then check those quotes, do the scoring and apply the verdict rules, so the same findings always give the same answer. -It is the second example in this repo. The first, and the one to read first, is the documentation pipeline in [`../docs-pipeline/`](../docs-pipeline/). These docs follow that pipeline's style guides and its rule that every doc ends with a tested prompt. +It is the second example in this repo. The first, and the one to read first, is the documentation pipeline in [`../docs-pipeline/`](../docs-pipeline/). These docs follow that pipeline's style guides and its rule that every README and guide ends with a tested prompt. The skill files and stage files do not carry one. **What you get.** @@ -55,6 +55,8 @@ Built by directing Claude Code. The author designed the rules, reviewed the outp It is a public rebuild of a private tool the author has had in daily use since late August 2026. The earliest public trace is this repo's first job-screening commit, dated 2026-08-25 (`git log --reverse --format='%ad %s' --date=short | grep job-fit-screen`). That earlier tool has since been removed; git history keeps it. Everything personal was left out: no real postings, no real person's data, no real employers. The rules were rewritten from a written list of the private tool's behaviors, and the scoring and verdict were rebuilt as scripts so they can be tested from a clean clone. +**Timeline, so the dates make sense.** The public version was assembled and reviewed on 2026-10-01, from a private system in daily use since late August. The pull requests that built it (listed in `git log`) were built by Claude Code agents under the author's direction. They merged between 16:38 and 22:02 that day, several a minute apart, because they were prepared in parallel and merged after CI passed on each. The author's check on that work is the tests and saved run records in this folder. No claim is made here about how much of each diff a person read line by line. + No production machine-learning or retrieval (RAG) work is claimed. This is prompts, a schema, and ordinary scripts with tests. ## What it proves, and what it does not @@ -74,13 +76,13 @@ No production machine-learning or retrieval (RAG) work is claimed. This is promp ## Verified -Run on 2026-10-01 from a fresh `git clone` of this repo at commit `17a49ce`, in a new virtual environment with Python 3.11, on macOS: +Run on 2026-10-01 from a fresh `git clone` of this repo, checked out at the commit just before the one that wrote this block (the hash is left out because a squash merge replaces it), in a new virtual environment with Python 3.11.11, on macOS: ```text $ python3 -m pytest -q -p no:cacheprovider SKIPPED [1] tests/live/test_live_assessment.py:97: live tests are off: set JA_LIVE=1 to run them (they call a model and cost money) SKIPPED [1] tests/live/test_live_intake.py:38: live tests are off: set JA_LIVE=1 to run them (they call a model and cost money) -461 passed, 2 skipped in 24.85s +542 passed, 2 skipped in 25.73s $ python3 scripts/validate_profile.py --fixture fixtures/robin-sample/career-profile.yaml validate_profile: clean, 0 warning(s) in career-profile.yaml $ python3 scripts/validate_profile.py fixtures/broken-profile.yaml; echo "exit=$?" @@ -91,9 +93,9 @@ ASSESSMENT: fixtures/postings/03-unlisted-pay-perks.md 🚦 VERDICT: Apply with reservations ``` -Clone, install and all of the above took 31 seconds, and left the clone with no changed files. CI repeats the same steps in a fresh clone on every pull request (the `clean-clone` job) and runs the tests on Ubuntu and macOS with Python 3.10 and 3.12. The test count grows over time; the [CI runs](https://github.com/darthrootbeer/context-engineering-toolkit/actions) show the current one. +Clone, install and all of the above took about 30 seconds, and left the clone with no changed files. CI repeats the same steps in a fresh clone on every pull request (the `clean-clone` job) and runs the tests on Ubuntu and macOS with Python 3.10 and 3.12. The test count grows over time; the [CI runs](https://github.com/darthrootbeer/context-engineering-toolkit/actions) show the current one. -The ten prompts in `prompts/` were each run on Claude Sonnet and Claude Haiku and graded against a written good-answer list. Sonnet passed all ten. Haiku passed eight and partly passed two: "Fix errors", where it still suggests values the file does not allow, and "Customize to my background", where it sometimes skipped reading an entry back and once offered an example that was the answer itself. Five prompts were rewritten after a failed or partial run, and every earlier run is kept in [`tests/prompt-runs/earlier-versions/`](tests/prompt-runs/earlier-versions/), with the grades in [`tests/prompt-runs/GRADES.md`](tests/prompt-runs/GRADES.md). +The ten prompts in `prompts/` were each run on Claude Sonnet and Claude Haiku, once per prompt per model, and graded by a Claude model (Claude Opus for the first round, Claude Sonnet 5.5 for the re-runs of prompts 03, 04, 05 and 06) against a written good-answer list that was written before each run. No claim is made that a person re-graded the answers, and one run is not a pass rate. Sonnet passed all ten. Haiku passed seven and partly passed three: "Fix errors", where it still suggests values the file does not allow, "Customize to my background", where it sometimes skipped reading an entry back and once offered an example that was the answer itself, and "Find gaps", where it missed some planted gaps. Prompt 05 was tested on a second, held-out profile whose gaps the prompt does not name: Sonnet found 6 of 6 and Haiku 5 of 6. Five prompts were rewritten after a failed or partial run (prompt 05 twice), and every earlier run is kept in [`tests/prompt-runs/earlier-versions/`](tests/prompt-runs/earlier-versions/), with the grades in [`tests/prompt-runs/GRADES.md`](tests/prompt-runs/GRADES.md). ## Where things are @@ -173,7 +175,7 @@ I'm attaching ARCHITECTURE.md, fixtures/robin-sample/career-profile.yaml, the th ```text -I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. ``` **Fix errors** @@ -183,4 +185,4 @@ I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my I'm attaching the output of a failed command from this tool (the validator, the tests, check_findings.py or assess_offline.py) and the file it names. For each problem in the output, explain the cause in one or two plain sentences, name the YAML path or file the output points to (do not guess line numbers), and propose the smallest change that fixes it, shown as before and after. Change nothing else. Never invent a value I did not give you: where the right value is my call, such as a date, a number or which entry to keep, say so, offer the choices, and write the after with a placeholder such as instead of a guess. Suggest only values the file or the output shows are allowed; if the allowed values are not shown, say they are listed in schema/career-profile.schema.json instead of guessing. If the output is not enough to know the cause, name the single command I should run next. A good answer on the validator output for fixtures/broken-profile.yaml covers all five problems and names the missing evidence id that one skill points to. ``` -**Tested on:** Claude Sonnet and Claude Haiku through the Claude Code command line, 2026-10-01. Sonnet passed all 6 prompts here. Haiku passed 4 and partly passed "Customize" (it skipped some read-backs and offered an example that was the answer) and "Fix errors" (it still suggests values the file does not allow). Every run, every rewrite and every grade is in `tests/prompt-runs/GRADES.md`. +**Tested on:** Claude Sonnet and Claude Haiku through the Claude Code command line, 2026-10-01. Sonnet passed all 6 prompts here. Haiku passed 3 and partly passed three: "Customize" (it skipped some read-backs and offered an example that was the answer), "Find gaps" (it missed a planted gap on each of two profiles) and "Fix errors" (it still suggests values the file does not allow). One run per prompt per model, graded by a Claude model, not a person. Every run, every rewrite and every grade is in `tests/prompt-runs/GRADES.md`. diff --git a/pipelines/job-assessment/SETUP.md b/pipelines/job-assessment/SETUP.md index c7bd24f..644bde1 100644 --- a/pipelines/job-assessment/SETUP.md +++ b/pipelines/job-assessment/SETUP.md @@ -37,7 +37,7 @@ On Windows, activate with `.venv\Scripts\activate` instead. python3 -m pytest -q ``` -Expect a line like `461 passed, 2 skipped`. The two skipped tests are the live runs, which call a model and only run when you ask for them (see [Live runs](#live-runs-optional)). The exact count grows as tests are added. What matters is that nothing fails. +Expect a line like `542 passed, 2 skipped`. The two skipped tests are the live runs, which call a model and only run when you ask for them (see [Live runs](#live-runs-optional)). The exact count grows as tests are added. What matters is that nothing fails. **Step 4. Check the fictional person's file, and watch a broken one fail.** diff --git a/pipelines/job-assessment/assessment/hooks/README.md b/pipelines/job-assessment/assessment/hooks/README.md index 1ecd296..a77d346 100644 --- a/pipelines/job-assessment/assessment/hooks/README.md +++ b/pipelines/job-assessment/assessment/hooks/README.md @@ -42,10 +42,10 @@ Add it to your Claude Code settings (`.claude/settings.json` in a project, or th } ``` -The guard needs `python3` and nothing else. You can also run the same check by hand: +The guard needs `python3` and nothing else. You can also run the same check by hand. This example checks the card the quickstart in the main README wrote: ```text -python3 assessment/scripts/check_email.py out/example-co-docs-writer.html +python3 assessment/scripts/check_email.py /tmp/ja-out/email/*.html ``` It prints `check_email: card passes` and exits 0, or lists each problem and exits 2. diff --git a/pipelines/job-assessment/fixtures/README.md b/pipelines/job-assessment/fixtures/README.md index 7b534b9..b8dfa35 100644 --- a/pipelines/job-assessment/fixtures/README.md +++ b/pipelines/job-assessment/fixtures/README.md @@ -32,6 +32,7 @@ The same numbers are in machine-readable form in `expected.yaml`. - `robin-sample/career-profile.yaml`: Robin's central file, two lanes, seven evidence entries, fourteen skills. - `robin-sample/intake-answers.txt`: scripted answers for the intake interview test. +- `gappy-profile.yaml` and `held-out-gappy-profile.yaml`: two profiles that pass the validator with no warnings but have judgment gaps it cannot see. The "find gaps" prompt was tuned on the first. The second is held out: nothing in the prompt describes its gaps. The planted gaps are listed in `tests/test_validate_profile.py`, not here. - `findings/*.findings.json`: hand-written model output for each posting. Every quote is copied from the posting text. - Each posting file starts with `company`, `role`, `lane`, `url` and `source` lines, then a divider, then the posting text. diff --git a/pipelines/job-assessment/fixtures/held-out-gappy-profile.yaml b/pipelines/job-assessment/fixtures/held-out-gappy-profile.yaml new file mode 100644 index 0000000..6baf733 --- /dev/null +++ b/pipelines/job-assessment/fixtures/held-out-gappy-profile.yaml @@ -0,0 +1,154 @@ +# FICTIONAL EXAMPLE DATA. Not a real person. +# +# A second career profile that passes scripts/validate_profile.py with no errors +# and no warnings, but has weak spots only a careful reader can find. It is the +# held-out test file for the "find gaps in my central file" prompt: the prompt +# was written and tuned against fixtures/gappy-profile.yaml, and nothing in the +# prompt describes the gaps in this one. The list of planted gaps is kept in +# tests/test_validate_profile.py, not here. +schema_version: 1 +person: + display_name: Quinn Example + target_roles: [API Technical Writer] + years_experience: 14 + location_label: Region C + working_style: gather_from_experts +sources: + - {key: board-a, emoji: "📋", label: Job board A} +hard_blocks: + - id: non_remote + label: "🏢 Must be in an office" + why: Quinn works from home and will not relocate. + match_hints: [remote, work from home, distributed team] + - id: crypto + label: "🪙 Cryptocurrency products" + why: Quinn does not want to write for speculative trading products. + match_hints: [crypto, token sale, nft] +named_exceptions: [] +comp: + currency: USD + floor: 64000 + min: 69000 + open_ask: 76000 + target: 83000 + stretch_ceiling: 97000 +culture: + perks: + - {id: unlimited_pto, label: Unlimited PTO, kind: big} + - {id: learning_budget, label: Learning budget, kind: nice} + low_time_off_days: 15 + hustle_phrases: [rockstar, fast-paced, always on] + free_phrases: [fast-paced, flexible hours] +soft_flags: + - {id: vague_title, label: Vague job title, why: Often means the role is still being decided.} +benefits_and_terms: + - {id: health, label: Health insurance from day one, why: Quinn pays for a dependent.} +company_criteria: + high_interest: + - {name: Fictional Api Tools Inc., why: Builds tools Quinn already uses.} + good_not_dream: [] +lanes: + - name: api-docs + emoji: "📚" + description: Roles writing and maintaining API reference and guides. + requirements: + - id: api_owner + label: "🔌 Owns the API docs" + severity: strong + why: Quinn wants to own the reference, not edit someone else's. + - id: expert_access + label: "🧑‍🔬 Regular access to engineers" + severity: strong + why: Accurate API docs need answers from the people who built it. + - id: small_team + label: "👥 Small team" + severity: soft + autonomy: + positive_signals: [own the roadmap, self-directed] + negative_signals: [tickets assigned daily] + keyword_signals: + strong_positive: + - {phrase: docs as code, why: Matches how Quinn works.} + positive: [] + negative: [] + strong_negative: [] + known_gaps: + - {id: gap_sql, label: SQL, skill_id: sql} +employers: + - id: harbor + name: Harbor Example Works + title: Technical Writer + start: 2021-03 + end: 2023-06 + summary: Wrote the admin and API guides for a scheduling product. + - id: tidewater + name: Tidewater Placeholder Inc. + title: Senior Technical Writer + start: 2023-07 + end: present + summary: Runs the help center and its build for a developer platform. +evidence: + - id: ev-harbor-onboarding + employer_id: harbor + claim: Wrote the customer onboarding guide that new accounts follow in their first week. + dates: {start: 2018-01, end: 2018-06} + authorship: WROTE + proof: checked + source: {type: document, ref: "onboarding-guide.pdf#overview", captured_on: 2026-10-01} + - id: ev-tidewater-migration + employer_id: tidewater + claim: Wrote every page of the help center again when it moved to a docs-as-code build. + dates: {start: 2024-01, end: 2024-09} + authorship: REVIEWED + proof: checked + source: {type: link, ref: "https://example.com/fictional/help-center", captured_on: 2026-10-01} + - id: ev-tidewater-glossary + employer_id: tidewater + claim: Built the product glossary and the rule for adding terms to it. + dates: {start: 2024-02, end: 2024-03} + authorship: WROTE + proof: unchecked + source: {type: interview, ref: "interview:2026-10-01:s2.q5", captured_on: 2026-10-01} +skills: + - id: docs_as_code + label: Docs as code + group: tooling + match: ['docs.as.code|docs-as-code'] + self_score: 4 + next: more + last: 2y + how: [self, team] + evidence_ids: [ev-tidewater-migration] + - id: user_guides + label: User guides and onboarding + group: writing + match: ['onboarding|user guide'] + self_score: 4 + next: neutral + last: 2y + how: [self] + evidence_ids: [ev-harbor-onboarding] + - id: glossaries + label: Glossaries and term lists + group: writing + match: ['glossary|terminology'] + self_score: 3 + next: neutral + last: 2y + how: [self] + evidence_ids: [ev-tidewater-glossary] + - id: sql + label: SQL + group: data + match: ['\bsql\b'] + self_score: 1 + next: neutral + last: 5y + how: [self] + evidence_ids: [] +writing_samples: + - {id: ws-help-center, title: Help center home page, link: "https://example.com/fictional/help-center", evidence_id: ev-tidewater-migration} +meta: + updated: 2026-10-01 + intake: + stages_done: [1, 2, 3, 4, 5, 6] diff --git a/pipelines/job-assessment/intake/README.md b/pipelines/job-assessment/intake/README.md index af68c4e..4b00984 100644 --- a/pipelines/job-assessment/intake/README.md +++ b/pipelines/job-assessment/intake/README.md @@ -59,7 +59,7 @@ I'm attaching intake/SKILL.md and intake/templates/career-profile.template.yaml. ```text -I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. ``` -**Tested on:** Claude Sonnet and Claude Haiku through the Claude Code command line, 2026-10-01. Sonnet passed both prompts here. Haiku passed one and partly passed "Customize" (it skipped some read-backs and offered an example that was the answer). Every run, every rewrite and every grade is in `../tests/prompt-runs/GRADES.md`. +**Tested on:** Claude Sonnet and Claude Haiku through the Claude Code command line, 2026-10-01. Sonnet passed both prompts here. Haiku partly passed both: "Customize" (it skipped some read-backs and offered an example that was the answer) and "Find the gaps" (it missed one planted gap on the first test profile, and on a second profile the prompt was not tuned for it found 5 of 6 and misread one field). Every run, every rewrite and every grade is in `../tests/prompt-runs/GRADES.md`. diff --git a/pipelines/job-assessment/prompts/05-find-gaps.txt b/pipelines/job-assessment/prompts/05-find-gaps.txt index 88d07c3..4a115ba 100644 --- a/pipelines/job-assessment/prompts/05-find-gaps.txt +++ b/pipelines/job-assessment/prompts/05-find-gaps.txt @@ -1 +1 @@ -I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. diff --git a/pipelines/job-assessment/requirements.txt b/pipelines/job-assessment/requirements.txt index 58c9264..f2b317c 100644 --- a/pipelines/job-assessment/requirements.txt +++ b/pipelines/job-assessment/requirements.txt @@ -1,3 +1,3 @@ -pyyaml -jsonschema -pytest +pyyaml>=6.0 +jsonschema>=4.17 +pytest>=7.0 diff --git a/pipelines/job-assessment/tests/e2e-output-run1.md b/pipelines/job-assessment/tests/e2e-output-run1.md index 60162df..f9c420c 100644 --- a/pipelines/job-assessment/tests/e2e-output-run1.md +++ b/pipelines/job-assessment/tests/e2e-output-run1.md @@ -312,7 +312,7 @@ Culture 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown -[Open the note](file:///archive/🟢%20📋%20🔧%20Copperline%20Example%20Co.%20-%20Docs%20Platform%20Engineer%20-%202026-10-01.md) +`[Open the note](file:///archive/🟢%20📋%20🔧%20Copperline%20Example%20Co.%20-%20Docs%20Platform%20Engineer%20-%202026-10-01.md)` The verdict is Apply, and no earlier rule fired: there was no hard block, no job-type override, and no score below the floor or in the reservations band. Because Copperline is a high-interest company, move fast and check for a warm introduction. The one thing to prepare for is the Go line: the posting wants you to read Go code, and you have no recorded evidence of recent Go work. That cost one Qualifications point. The Fit, Comp and Culture bars are all at 10, and Comp is based on the Region B tier ($85k–$104k). @@ -567,7 +567,7 @@ Culture 9 🟢🟢🟢🟢🟢🟢🟢🟢🟢⬛ 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown -Note: file:///archive/🟢%20📋%20✍️%20Ashgrove%20Example%20Software%20-%20Senior%20Technical%20Writer%20-%202026-10-01.md +Note: `file:///archive/🟢%20📋%20✍️%20Ashgrove%20Example%20Software%20-%20Senior%20Technical%20Writer%20-%202026-10-01.md` The verdict is Apply because no hard block, job-type override or score floor fired. Pay is not listed, so Comp stays at a neutral 5. Culture is high because the posting lists extra paid days off and a learning budget. diff --git a/pipelines/job-assessment/tests/e2e-output.md b/pipelines/job-assessment/tests/e2e-output.md index 8c82224..e61cf97 100644 --- a/pipelines/job-assessment/tests/e2e-output.md +++ b/pipelines/job-assessment/tests/e2e-output.md @@ -332,7 +332,7 @@ Culture 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown -📄 Note: [file://…/pipeline/archive/🟢 📋 🔧 Copperline Example Co. - Docs Platform Engineer - 2026-10-01.md](file:///archive/🟢%20📋%20🔧%20Copperline%20Example%20Co.%20-%20Docs%20Platform%20Engineer%20-%202026-10-01.md) +📄 Note: `[file://…/pipeline/archive/🟢 📋 🔧 Copperline Example Co. - Docs Platform Engineer - 2026-10-01.md](file:///archive/🟢%20📋%20🔧%20Copperline%20Example%20Co.%20-%20Docs%20Platform%20Engineer%20-%202026-10-01.md)` ✉️ Email card: `out/copperline-example-co-docs-platform-engineer.html` The verdict is Apply because no hard block, job-type override, score floor or reservations band fired. The one deduction is reading Go, a known gap for you, so Qualifications is 9. You meet the 7+ years ask with 9 years. Because Copperline is high-interest, move quickly and check for a warm introduction. Interviews may probe hands-on depth on OpenAPI, since that work was done by directing AI. @@ -600,7 +600,7 @@ Culture 9 🟢🟢🟢🟢🟢🟢🟢🟢🟢⬛ 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown -Note: [file:///archive/🟡 📋 ✍️ Ashgrove Example Software - Senior Technical Writer - 2026-10-01.md](file:///archive/🟡%20📋%20✍️%20Ashgrove%20Example%20Software%20-%20Senior%20Technical%20Writer%20-%202026-10-01.md) +Note: `[file:///archive/🟡 📋 ✍️ Ashgrove Example Software - Senior Technical Writer - 2026-10-01.md](file:///archive/🟡%20📋%20✍️%20Ashgrove%20Example%20Software%20-%20Senior%20Technical%20Writer%20-%202026-10-01.md)` Email card: `out/ashgrove-example-software-senior-technical-writer.html`. Its subject starts with 🟡. **Why this verdict:** Fit and Qualifications average under 7, which puts it in the reservations band. No hard block or job-type override applied. Ashgrove is a "good, not dream" company, and that did not affect the scores. diff --git a/pipelines/job-assessment/tests/prompt-runs/GRADES.md b/pipelines/job-assessment/tests/prompt-runs/GRADES.md index bad085d..bd33e24 100644 --- a/pipelines/job-assessment/tests/prompt-runs/GRADES.md +++ b/pipelines/job-assessment/tests/prompt-runs/GRADES.md @@ -3,7 +3,7 @@ Every prompt in `prompts/` was run through the Claude Code command line on 2026-10-01, on Claude Sonnet (`claude-sonnet-5-5`) and Claude Haiku (`claude-haiku-4-5-20251001`). No other model was tried. Each run was a fresh session in an empty temp folder with no tools, no MCP servers, and only project settings loaded (there were none). Claude Code still tells every session some basic facts about its environment, such as its working folder; that is how one simulated user typed this machine's real home folder path, which was scrubbed from the record. The files a prompt names were pasted in after it, the way a person attaches files in a chat window. `run_prompts.py` in this folder does all of that and saves one record per prompt and model. -The grades were written by Claude Opus, reading each saved answer against the good-answer list below, which was written before any prompt was run. Two prompts also get a check by code: prompt 03's JSON is run through `check_findings.py` and `assess_offline.py`, and prompt 02's final file is run through `validate_profile.py`. Those results are at the top of each record. +**How the grading was done, exactly.** Every grade in this file was written by a Claude model reading a saved answer against the good-answer list below. Claude Opus wrote the first round. Claude Sonnet 5.5 wrote the grades for the second round: the re-runs of prompts 03, 04, 05 and 06, made later on 2026-10-01. Each good-answer list was written before the run it was used for (the one for prompt 05 was rewritten before the second round). Every prompt was run once per model per profile. This file does not claim that a person re-graded any answer, and one run cannot show how often a prompt passes: read PASS as "passed once", not as a rate. The one case with more than one run on the same text shows why. An earlier run of the held-out profile for prompt 05 (below) is not kept, and found 6 of 6 gaps on Sonnet and 4 of 6 on Haiku. The kept re-run found 6 of 6 and 5 of 6. An outside reviewer's own Haiku run of prompt 05 on the first profile also came out PARTIAL, where the first round had graded it PASS. Two prompts also get a check by code: prompt 03's JSON is run through `check_findings.py` and `assess_offline.py`, and prompt 02's final file is run through `validate_profile.py`. Those results are at the top of each record. Three prompts are conversations: 02, s1 and r2. For those a second model plays the user from a short script (Haiku for s1 and r2, Sonnet for 02 after Haiku proved unreliable in that role). A simulated user is not a person. Where it went off script, the grade below says so. @@ -17,14 +17,16 @@ PASS means every item on the good-answer list held. PARTIAL means the main goal | Customize to my background (build my central file) | `02-customize.txt` | PASS | PARTIAL: some entries not read back after each answer, and one example equals the answer | `p2-*.md` | | Run on my first posting | `03-run-first-posting.txt` | PASS | PASS (an earlier run of the same text failed the quote check, see below) | `p3-*.md` | | Test the scoring with the fixture | `04-test-scoring.txt` | PASS | PASS | `p4-*.md` | -| Find gaps in my central file | `05-find-gaps.txt` | PASS | PASS | `p5-*.md` | +| Find gaps in my central file | `05-find-gaps.txt` | PASS on both profiles (3 of 3 gaps, then 6 of 6) | PARTIAL on both profiles (about 1.5 of 3 gaps, then 5 of 6 with one misread field) | `p5-*.md`, `p5h-*.md` | | Fix errors | `06-fix-errors.txt` | PASS | PARTIAL: suggests values the file does not allow | `p6-*.md` | | Review the rules for gaps | `r1-review-rules.txt` | PASS | PASS (an earlier run of the same text was PARTIAL, see below) | `r1-*.md` | | Walk me through setup | `s1-setup-walkthrough.txt` | PASS | PASS | `s1-*.md` | | Customize the rubric | `r2-customize-rubric.txt` | PASS | PASS | `r2-*.md` | | Understand the hook | `h1-understand-hook.txt` | PASS | PASS | `h1-*.md` | -**Timing note.** The final runs of prompts 01, 04, r1, r2 and h1 used the docs as they stood shortly before the final commit. After those runs, only two kinds of edit were made to the attached docs: the prompt blocks were copied again from `prompts/` (prompts 02, 03, 05, 06 and r1 had been rewritten), and the "Tested on" lines were filled in from this file. +**Timing note.** The first round of runs happened before a later commit changed every pay figure in the fictional fixtures. The records for prompts 03, 04, 05 and 06 were therefore run again on 2026-10-01 against the fixtures as they ship now, and saved exactly as the models answered, with nothing edited. That replaced six saved records that had carried a note saying their pay figures were changed after the run. Records for 01, 02, r1, r2, s1 and h1 were made before the pay change and were not re-run. Their models read the earlier fictional pay figures. None of those figures appears in any saved record, and no good-answer list for those prompts uses a pay amount. The final runs of prompts 01, r1, r2 and h1 used the docs as they stood shortly before the final commit of the first round. After those runs, only two kinds of edit were made to the attached docs: the prompt blocks were copied again from `prompts/`, and the "Tested on" lines were filled in from this file. + +Records in `earlier-versions/` that begin with a note about pay figures were edited after the run: fictional pay figures replaced the earlier ones, and nothing else. They test earlier wording of prompts, so they cannot be re-run as they were. Two of them (`p4-haiku-v1.md` and `p4-sonnet-v1.md`) had a wrong target of $78,000 left by that edit, and it is corrected to the fixture's $100,000 in the same pull request that re-ran the prompts. ## Good-answer lists, and what happened @@ -51,28 +53,31 @@ Good answer: the JSON passes `check_findings.py` with no problems and `assess_of - **Version 1** (asked for scores and a verdict as well, without the findings schema attached): FAIL on both. Both wrote `strong_positive_phrases` as plain strings, so the findings broke the schema. This run also showed that `check_findings.py` stops with a Python traceback on that shape instead of a plain message. That is noted for a follow-up; `assess_offline.py` reports it cleanly because it checks the schema first. - **Version 2** (schema attached, scores as a labelled preview): both JSON files passed the checks and gave Apply. Sonnet PASS. Haiku PARTIAL: its preview scored Comp 8 when the pay top was above the target, which the scripts score 10. This is the lesson of the whole system in miniature, so the next version stopped asking for scores at all. -- **Version 3** (the published text: findings and judgment calls only, no scores): Sonnet PASS. Haiku's first run quoted a sentence that is not in the posting. `check_findings.py` caught it and `assess_offline.py` stopped before any score was computed (`earlier-versions/p3-haiku-v3.md`). On the second run of the same text, which allows one round of fixes after a failed check, both models passed on the first try, so no fix round was needed. Both gave Apply. Haiku's findings scored 9, 10, 9 and 10, the same as the hand-written fixture. Sonnet's scored Fit 10 instead of 9, because it rated the manual docs build fair where the fixture says weak, and it named that exact call as one it was unsure of. The verdict did not change. +- **Version 3** (the published text: findings and judgment calls only, no scores): Sonnet PASS. Haiku's first run quoted a sentence that is not in the posting. `check_findings.py` caught it and `assess_offline.py` stopped before any score was computed (`earlier-versions/p3-haiku-v3.md`). On the second run of the same text, which allows one round of fixes after a failed check, both models passed on the first try, so no fix round was needed. The re-run of 2026-10-01 on the shipped fixtures gave the same result: both JSON files passed `check_findings.py` on the first try, no model computed a score or verdict, none cited the `do_not_use` entry, and each listed three judgment calls with the posting's words. Both gave Apply; both rated Fit 10 where the hand-written fixture says 9, and the verdict did not change. Both gave Apply. Haiku's findings scored 9, 10, 9 and 10, the same as the hand-written fixture. Sonnet's scored Fit 10 instead of 9, because it rated the manual docs build fair where the fixture says weak, and it named that exact call as one it was unsure of. The verdict did not change. ### 04 Test the scoring with the fixture Good answer: arithmetic shown for all four scores on all three postings; 12 of 12 scores and 3 of 3 verdicts match `fixtures/expected.yaml`; the comparison comes after the work. -- Sonnet and Haiku: PASS on both runs, 12 of 12 and 3 of 3 each time. Caveat: the expected file is attached in the same message, so the order "work first, then compare" cannot be enforced. Each answer's arithmetic was checked step by step against the rules, not just the totals. +- Sonnet and Haiku: PASS on both runs, 12 of 12 and 3 of 3 each time. The re-run of 2026-10-01 on the shipped fixtures also gave 12 of 12 and 3 of 3 on both, with the arithmetic shown and the target of $100,000 quoted correctly. Caveat: the expected file is attached in the same message, so the order "work first, then compare" cannot be enforced. Each answer's arithmetic was checked step by step against the rules, not just the totals. - The first Sonnet run pointed out that the document did not name the `4c average` trigger label. `ARCHITECTURE.md` now lists every trigger label. ### 05 Find gaps in my central file -Good answer: finds all three planted gaps in `fixtures/gappy-profile.yaml` (the skill scored 5 on one unchecked interview answer, the hard block with no reason, the second lane copied from the first), each with its YAML path, why it matters and one question; fills no gap itself. +Good answer (written before the second round): finds the three planted gaps in `fixtures/gappy-profile.yaml` (the skill scored 5 on one unchecked interview answer, the hard block with no reason, the second lane copied from the first), each with its YAML path, why it matters and one question; fills no gap itself; says nothing that is false of the file. On `fixtures/held-out-gappy-profile.yaml`, which the prompt was never tuned against, six gaps are planted (listed in `tests/test_validate_profile.py`): 14 years of experience against employers that start in 2021; evidence dated 2018 at an employer that starts in 2021; a claim of "wrote every page" with authorship REVIEWED; a hard block whose match hints would fire on the remote jobs the person wants; "fast-paced" in both the hustle and the free phrase lists; and a skill marked used in the last two years whose only evidence ends in 2018. PASS is at least 5 of 6 found with no false statement. PARTIAL is 3 or 4 found, or any false statement. FAIL is fewer than 3. - **Version 1**: both found all three, but the test was not fair. The attached section of `ARCHITECTURE.md` described the three gaps. That paragraph was rewritten in general terms, and the prompt was widened to say "must-haves or hard blocks with no reason" and "skills that rest on thin evidence". -- **Version 2** (published): PASS on both. Both also raised extra, reasonable weak spots, such as `last` dates that do not match the evidence dates. +- **Version 2**: graded PASS on both in the first round. A cold review pointed out that the prompt still named the kinds of gap to look for, so finding them was weak evidence. Records: `earlier-versions/p5-*-v2.md`. +- **Version 3** (published): the prompt names no kind of gap. It asks for places where the file claims more than its evidence shows, conflicts with itself, or would mislead a score, in every section. The held-out profile was added to test it. +- **Version 3 on the first profile.** Sonnet: PASS, 3 of 3 gaps, plus other reasonable weak spots, with no false statement (the missing reason on the weapons block came inside a longer item). Haiku: PARTIAL. It found the unchecked top-scored skill, and it saw that the second lane's strong "owns the docs build" must-have does not fit a writing lane, but it did not notice that the whole lane is a copy of the first, and it did not find the hard block with no reason. It made no false statement. Record: `p5-*.md`. +- **Version 3 on the held-out profile.** Sonnet: PASS, 6 of 6, no false statement. Two of its notes read the field `last` in two different ways, which the schema does not settle, so they are not graded as false. Haiku: PARTIAL, 5 of 6. It missed the hard block whose match hints would fire on the remote jobs the person wants. It also called the "vague job title" soft flag a flaw in the person's own profile, when that flag is a warning about postings, which is a misreading. Record: `p5h-*.md`. These are the results to quote: Sonnet found every planted gap on a file the prompt was not tuned for, and Haiku found most of them but not all, and not on every run. ### 06 Fix errors Good answer: covers all five problems in the broken fixture's validator output; names the missing evidence id `ev-placeholder-style-guide`; each fix is the smallest before-and-after change; where the right value is the user's call it offers choices with a placeholder instead of a guess; suggests only values the file allows. - Sonnet: PASS on versions 1, 2 and 4. On version 3 it filled in `authorship: WROTE` with "or CO-WROTE, " where that version asked for a placeholder, which is PARTIAL by a strict reading; the published version 4 is a PASS. -- Haiku: PARTIAL on all four versions. Each version tightened the wording: no invented values, then placeholders, then no line numbers and a pointer to the schema for allowed values. Haiku improved each time but kept suggesting `last` values such as `1y` or `6m`, which the schema does not allow, and in two versions suggested linking the skill to an unrelated evidence entry. In the final version it uses placeholders for the pay numbers and the `last` value, but still fills in `authorship: WROTE` itself and again offers an unrelated evidence id as an example. Records: `earlier-versions/p6-*-v1.md` to `-v3.md`, and `p6-*.md`. +- Haiku: PARTIAL on all four versions. Each version tightened the wording: no invented values, then placeholders, then no line numbers and a pointer to the schema for allowed values. Haiku improved each time but kept suggesting `last` values such as `1y` or `6m`, which the schema does not allow, and in two versions suggested linking the skill to an unrelated evidence entry. In the final version it uses placeholders for the pay numbers and the `last` value, but still fills in `authorship: WROTE` itself and again offers an unrelated evidence id as an example. Records: `earlier-versions/p6-*-v1.md` to `-v3.md`, and `p6-*.md`. The re-run of 2026-10-01 on the shipped fixtures gave the same grades: Sonnet PASS (all five problems, the missing id named, placeholders and the schema pointer where the value is the user's call). Haiku PARTIAL (all five problems and the missing id, but it wrote `authorship: WROTE` and `self_score: 3` itself and offered `1y` and `6m` as values for `last`, which the file does not allow). ### r1 Review the rules for gaps diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md index ad6057d..40c796e 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md @@ -33,7 +33,7 @@ I'll work through each posting's scoring, showing all arithmetic, then compare w 10 − 0 − 0 − 1 − 0 = **9** -**Comp:** Pay stated, Region B tier, top $104,000. Robin's target is $78,000. Top is at or above target. Score **10**. +**Comp:** Pay stated, Region B tier, top $104,000. Robin's target is $100,000. Top is at or above target. Score **10**. **Qualifications:** Start at 10. - Years required 7, Robin has 9 → 0 (not more, not narrow) diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md index dc9a7ff..0ce2e91 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md @@ -25,7 +25,7 @@ I worked all three from the rules and findings first, then compared against `exp | Score | Arithmetic | Result | |---|---|---| | Fit | 10. `ai_forward` (strong, rated strong): 0. `docs_as_code` (strong, rated strong): 0. `ci_for_docs` (soft, rated weak): -1. `solo_ownership` is a bonus: 0. Autonomy is net positive: 0. No avoided skills. 10 - 1 = 9. | **9** | -| Comp | Robin's label is Region B, so the Region B tier is used (not the model's choice, but the same). Top is $104,000, which is at or above the $78,000 target. | **10** | +| Comp | Robin's label is Region B, so the Region B tier is used (not the model's choice, but the same). Top is $104,000, which is at or above the $100,000 target. | **10** | | Qualifications | 10. The posting asks for 7 years and Robin has 9: 0. `go_lang` is a known gap: -1. Robin's self-score of 1 on Go is the same skill, so it counts once: 0. No working-style mismatch. 10 - 1 = 9. | **9** | | Culture | 5. `unlimited_pto` (big): +2. `learning_budget` (big): +2. `offsites` (nice): +1. "definition of done" (strong positive phrase): +1. 5 + 2 + 2 + 1 + 1 = 11, clamped to the cap of 10. | **10** | diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-haiku-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-haiku-v2.md new file mode 100644 index 0000000..3aeb67c --- /dev/null +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-haiku-v2.md @@ -0,0 +1,67 @@ + +# Prompt run: Find gaps in my central file, on haiku + +- Prompt file: `prompts/05-find-gaps.txt` +- Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). +- Turns: 1. Cost: $0.05 (the simulated user's calls are not included). +- Grade: see `GRADES.md`. + +## Conversation + +### User + +[the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/gappy-profile.yaml] + +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. + +### Model + +I'll systematically check what the validator cannot catch. Here are the weak spots: + +--- + +## **Planted gaps (3 major ones):** + +### 1. **`skills[1].docs_ci` — Score 5 backed only by unchecked interview** +- **YAML path:** `skills[1]` (or `skills[1].self_score`, `skills[1].evidence_ids`) +- **Why it matters:** A score of 5 signals mastery. The validator won't warn about weak evidence *within* an evidence list—only absence of evidence. This evidence is `proof: unchecked` from an interview with no artifact to verify, yet it's being scored as your highest skill. +- **Question:** Can you point to a documented artifact (a PR, a config file, CI logs, or documentation) that shows the link-checking build you set up, or was this pairing work where you implemented it together with someone else? + +--- + +### 2. **`hard_blocks[1].why` — Weapons hard block has no reason** +- **YAML path:** `hard_blocks[1]` (the weapons block) +- **Why it matters:** The validator warns about `strong` must-haves with no reason, but hard blocks aren't checked for this. Without a stated reason, a recruiter can't tell if this is an ethical boundary, a legal constraint, or something else—affecting how they'd present the role to you. +- **Question:** Why won't you work for weapons makers? Is it a personal value, a concern about the industry, or something else? + +--- + +### 3. **`lanes[1].requirements` — Tech-writing lane requirements copied verbatim from docs-platform** +- **YAML path:** `lanes[1].requirements` (and `lanes[1].autonomy`, `lanes[1].keyword_signals`) +- **Why it matters:** These roles are fundamentally different (one builds systems, one writes content), yet both demand "Owns the docs build" as a strong requirement. The validator can't catch copy-paste; only humans notice when a requirement doesn't match the role's actual needs. +- **Question:** For a pure tech-writing role, do you need to own the docs build system itself, or would you prefer to write excellent content *within* a system that already exists? + +--- + +## **Additional weak spots:** + +### 4. **`evidence[0].authorship: DIRECTED` — Thin for a claimed skill** +- **YAML path:** `skills[0].evidence_ids[0]` → `evidence[0].authorship` +- **Why it matters:** You scored OpenAPI 4, but the evidence shows you *directed* the API spec work, not that you implemented the spec generation or wrote against it. For a technical writer, this means you supervised someone else's work. +- **Question:** Did you implement the system that generates the API reference from the spec, or did you write the reference documentation after someone else built the system? + +--- + +### 5. **Last-used dates don't match evidence end dates** +- **YAML path:** `skills[0].last`, `skills[1].last`, `skills[2].last` +- **Why it matters:** All three scored skills claim `last: "2y"`, but evidence shows: + - `openapi`: last evidence 2021-09 (4.8 years ago) + - `docs_ci`: last evidence 2023-06 (3.3 years ago) + - `style_guides`: last evidence 2022-08 (4.1 years ago) + + The validator can't connect `last` to evidence dates; a mismatch suggests either the skill is rustier than claimed, or the profile wasn't kept in sync. +- **Question for each:** When did you actually *last use* each of these skills hands-on, and what did you do? + +--- + +**Summary:** The validator missed the three planted gaps (weak evidence quality, missing hard-block reason, and cloned requirements) because it checks *structure*, not *judgment*. The `last` dates are the only automatic error it could have caught but didn't, suggesting the profile wasn't updated when the `last` field was filled in. diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-sonnet-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-sonnet-v2.md new file mode 100644 index 0000000..e609d09 --- /dev/null +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p5-sonnet-v2.md @@ -0,0 +1,75 @@ + +# Prompt run: Find gaps in my central file, on sonnet + +- Prompt file: `prompts/05-find-gaps.txt` +- Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). +- Turns: 1. Cost: $0.05 (the simulated user's calls are not included). +- Grade: see `GRADES.md`. + +## Conversation + +### User + +[the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/gappy-profile.yaml] + +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. + +### Model + +I found the three planted gaps, plus a few smaller ones. I changed nothing in the file. + +## 1. Skills scored 3+ on thin evidence + +**`skills[1]` (`docs_ci`), `self_score: 5`, `evidence_ids: [ev-placeholder-docs-ci]`** +- **Why it matters:** The top score rests on one item. That item is `unchecked`, comes from an interview (`evidence[1].source`), and spans one month (`evidence[1].dates`: 2023-05 to 2023-06). `how` is `[self]` only, so nobody else vouches for it. A posting that leans on CI and build skill would get a 5 that nothing backs up. +- **Question:** "Besides the broken-link check, what else have you built or maintained in the docs pipeline, and is there a commit, config file or teammate who could confirm it?" + +**`skills[0]` (`openapi`), `self_score: 4`** +- **Why it matters:** The only evidence is `evidence[0]`, which is dated 2021 and marked `authorship: DIRECTED`. That shows you led the change, not that you hands-on wrote OpenAPI specs. `last: 2y` also doesn't match evidence that ends in 2021-09. +- **Question:** "Have you written or edited OpenAPI specs yourself, and when did you last do it?" + +**`skills[2]` (`style_guides`), `last: 2y`** (minor) +- **Why it matters:** The evidence is one document from 2022 (`evidence[2].dates`), and `last: 2y` is a stretch from there. The 4 itself is reasonably supported. +- **Question:** "Have you updated or applied the style guide since 2022, and is there anything newer to point to?" + +## 2. Unchecked evidence a posting would lean on + +**`evidence[1].proof: unchecked`** (`ev-placeholder-docs-ci`) +- **Why it matters:** This is the only evidence behind the skill scored 5, and it supports the "owns the docs build" requirement in `lanes[0].requirements[1]`. It is the most likely thing for a posting to test, and it is the least verified. +- **Question:** "Is there a link, a repo file, a PR, or a reference who can confirm the link-check work so it can move to `checked`?" + +**`writing_samples[0].evidence_id`** +- **Why it matters:** The sample is a getting-started guide, but it points at `ev-northwind-api-rebuild`, which is about generating the API reference. The sample doesn't clearly prove that claim, and the evidence is `DIRECTED`, not `WROTE`. +- **Question:** "Did you write the getting-started guide yourself, and does it belong with a different piece of evidence?" + +## 3. Evidence with no dates + +None here. Every evidence item and employer has dates, which is why the validator is silent. Two things to check anyway: +- `evidence[1].dates` is a one-month window. Was the work really that short, or is the range just when it shipped? +- Several skills claim `last: 2y`, but the dated evidence for them is older than that. See section 1. + +## 4. Must-haves or hard blocks with no reason + +**`hard_blocks[1]` (`weapons`) has no `why`** +- **Why it matters:** The validator only warns about strong must-haves, not hard blocks. A hard block removes a posting outright, so an unexplained one can't be judged on borderline cases such as defense-adjacent software or dual-use tools. +- **Question:** "What is the line for you: only makers of weapons, or also companies that sell software or services to them?" + +**`hard_blocks[2].match_hints` includes `hybrid`** (`non_remote`) +- **Why it matters:** The `why` says you won't relocate and work from home, but the hint blocks hybrid roles too. Whether hybrid is acceptable changes which postings get screened out. +- **Question:** "Is a hybrid role with a few office days within walking or commuting range ever acceptable?" + +**`lanes[0].requirements[2]` and `lanes[1].requirements[2]` (`small_team`), `severity: soft`, no `why`** +- **Why it matters:** It's soft, so the validator won't flag it. It still sways scoring, and without a reason it can't be weighed against other signals. +- **Question:** "What does 'small' mean to you in headcount, and what goes wrong for you on a larger team?" + +## 5. Lanes whose requirements look copied + +**`lanes[1]` (`tech-writing`) duplicates `lanes[0]` (`docs-platform`) in:** +- `lanes[1].requirements`: same three ids, severities and reasons. +- `lanes[1].autonomy`: same positive and negative signals. +- `lanes[1].keyword_signals`: same `docs as code` phrase. + +- **Why it matters:** The validator only checks that each lane has some must-haves, not that they differ. With identical rules, both lanes score every posting the same, so the lane choice tells you nothing. `owns_pipeline` as a strong must-have also contradicts `lanes[1].description` ("writing and editing"): a writing-focused role would normally not own the build. That makes the `tech-writing` lane unlikely to match the roles it is meant to find. +- **Question:** "For a role that is mostly writing, which of these must-haves would you actually insist on, and what would you want to see that the docs-platform lane doesn't ask for?" + +Related: `lanes[1].known_gaps` is empty while `lanes[0].known_gaps` names Kubernetes. Is Kubernetes really irrelevant to the writing lane, or did that lane just never get the same review? diff --git a/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md index 8257c9e..76082ae 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md @@ -1,8 +1,6 @@ # Prompt run: Run on my first posting, on haiku -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.10 (the simulated user's calls are not included). @@ -11,32 +9,32 @@ Note: the pay figures in this record were changed after the run to fictional one ## Checks run by code on the reply ```text -$ python3.11 assessment/scripts/check_findings.py /findings.json --posting fixtures/postings/01-strong-fit.md --profile fixtures/robin-sample/career-profile.yaml +$ python assessment/scripts/check_findings.py /findings.json --posting fixtures/postings/01-strong-fit.md --profile fixtures/robin-sample/career-profile.yaml check_findings: ok (exit code 0) ``` ```text -$ python3.11 scripts/assess_offline.py fixtures/postings/01-strong-fit.md --findings /findings.json --profile fixtures/robin-sample/career-profile.yaml --out /out --date 2026-10-01 -ASSESSMENT: ARCHIVE/2026-10-01-copperline-example-co-docs-platform-engineer.md +$ python scripts/assess_offline.py fixtures/postings/01-strong-fit.md --findings /findings.json --profile fixtures/robin-sample/career-profile.yaml --out /out --date 2026-10-01 +ASSESSMENT: fixtures/postings/01-strong-fit.md 🚦 VERDICT: Apply 🏢 Copperline Example Co. is one you have said you want to work for. Your reason: Makes tools for writers, which is the work Robin cares about most. This did not change the scores or the verdict. **Scores:** -Fit 9 🟢🟢🟢🟢🟢🟢🟢🟢🟢⬛ +Fit 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 Comp 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 Qualifications 9 🟢🟢🟢🟢🟢🟢🟢🟢🟢⬛ Culture 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 | Check | Rating | Read | |---|---|---| -| ✨ AI-forward team | 🟢 Strong | The team uses AI tools every day for drafting, which is exactly what you want. You would help decide what review rules they need. | -| 📦 Docs live in version control | 🟢 Strong | Docs are stored in Git as Markdown and go through pull request review, which matches your preference for working where review and automation matter. | -| 🧪 Docs are built and checked automatically | 🟠 Weak | Docs are currently built by hand, not automatically. You'd build the automation from scratch, but it doesn't exist yet. | -| 🦸 Owns the docs system | 🟢 Strong | You would own the docs system from day one with just peer review from one engineer. This is the ownership you're looking for. | -| How much say you'd have | 🟢 Strong | You would have wide say in how the docs system works. Your manager checks in weekly and one engineer reviews your work, but you choose tools and set priorities. | +| ✨ AI-forward team | 🟢 Strong | The team uses AI to draft every day, and you'd decide what still needs human review. That's the AI-forward setup you want. | +| 📦 Docs live in version control | 🟢 Strong | Docs live in Git as Markdown with PR reviews, exactly your working style. | +| 🧪 Docs are built and checked automatically | 🟡 Fair | Docs build is manual now; they want you to automate it. You'd own that project from the start. | +| 🦸 Owns the docs system | 🟢 Strong | You own the system day one, with one peer reviewer. That's real ownership and autonomy. | +| How much say you'd have | 🟢 Strong | You choose the tools and set your own priorities, with a weekly check-in. That's substantial say in how the work happens. | 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown 1/10 validate the profile @@ -66,9 +64,34 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil ### Model +I'll assess this posting for Robin Sample against the docs-platform lane. Let me work through the checklist. + +**Hard blocks:** Copperline is remote ("a remote company"), not gambling, weapons, or excluding locations. No blocks trip. + +**Requirements match:** +- `ai_forward` (strong): The posting explicitly puts AI into daily workflow. +- `docs_as_code` (strong): "all docs live in a Git repository as Markdown and are reviewed in pull requests" — core to Robin's method. +- `ci_for_docs` (soft): Currently manual; they want Robin to automate it. +- `solo_ownership` (soft/bonus): Robin owns the system from week one with one peer reviewer. + +**Autonomy:** "You choose the tools and set the priorities for the docs system; your manager checks in once a week." The profile's positive signals (chooses tools, sets priorities, weekly manager check-in) all appear. Net: positive. + +**Keywords:** "definition of done" hits as strong positive. "Documentation is part of the definition of done for every feature." + +**Qualifications:** +- Years required: 7; Robin has 9 total experience. +- Known gaps: "You can read Go code" — Robin scored 1 on Go. This is a lane known gap (`gap_go`). +- Skills: OpenAPI match (ev-northwind-api-rebuild). Docs-as-code match (ev-placeholder-docs-pipeline, ev-placeholder-ci-checks). + +**Pay:** Base $100k–$104k for Region B (Robin's location). Matches Robin's target exactly. + +**Culture:** Unlimited PTO with 3-week minimum, $3k learning budget, two annual meetups. No hustle language. The company is listed as high-interest in the profile: "Makes tools for writers, which is the work Robin cares about most." + +Here's the findings JSON: + ```json { - "posting_file": "ARCHIVE/2026-10-01-copperline-example-co-docs-platform-engineer.md", + "posting_file": "fixtures/postings/01-strong-fit.md", "lane": "docs-platform", "hard_block": { "tripped": false, @@ -87,31 +110,33 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil "id": "ai_forward", "rating": "strong", "quote": "Our writers and engineers draft with AI assistants every day, and you will help decide which checks a human must still make.", - "read": "The team uses AI tools every day for drafting, which is exactly what you want. You would help decide what review rules they need." + "read": "The team uses AI to draft every day, and you'd decide what still needs human review. That's the AI-forward setup you want." }, { "id": "docs_as_code", "rating": "strong", - "quote": "all docs live in a Git repository as Markdown and are reviewed in pull requests", - "read": "Docs are stored in Git as Markdown and go through pull request review, which matches your preference for working where review and automation matter." + "quote": "Keep docs-as-code habits healthy: all docs live in a Git repository as Markdown and are reviewed in pull requests.", + "read": "Docs live in Git as Markdown with PR reviews, exactly your working style." }, { "id": "ci_for_docs", - "rating": "weak", + "rating": "fair", "quote": "The docs build is still run by hand before each release, and we want you to change that.", - "read": "Docs are currently built by hand, not automatically. You'd build the automation from scratch, but it doesn't exist yet." + "read": "Docs build is manual now; they want you to automate it. You'd own that project from the start." }, { "id": "solo_ownership", "rating": "strong", "quote": "You will own the documentation system from the first week, with one other engineer to review your work.", - "read": "You would own the docs system from day one with just peer review from one engineer. This is the ownership you're looking for." + "read": "You own the system day one, with one peer reviewer. That's real ownership and autonomy." } ], "autonomy": { "net": "positive", - "quotes": ["You choose the tools and set the priorities for the docs system; your manager checks in once a week."], - "read": "You would have wide say in how the docs system works. Your manager checks in weekly and one engineer reviews your work, but you choose tools and set priorities." + "quotes": [ + "You choose the tools and set the priorities for the docs system; your manager checks in once a week." + ], + "read": "You choose the tools and set your own priorities, with a weekly check-in. That's substantial say in how the work happens." }, "pay": { "stated": true, @@ -121,12 +146,6 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil "min": 100000, "max": 104000, "quote": "Base pay range for Region B: $100,000 to $104,000." - }, - { - "label": "Region A", - "min": 102000, - "max": 118000, - "quote": "Region A: $102,000 to $118,000." } ], "tier_used": null, @@ -162,17 +181,24 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil "qualifications": { "years_required": 7, "narrow_subdomain": false, - "known_gaps_hit": ["gap_go"], + "known_gaps_hit": [ + "gap_go" + ], "self_score_gaps": [], "working_style_mismatch": false, "matches": [ { "requirement_quote": "Experience generating API reference pages from an OpenAPI file.", - "evidence_ids": ["ev-northwind-api-rebuild"] + "evidence_ids": [ + "ev-northwind-api-rebuild" + ] }, { - "requirement_quote": "The docs build is still run by hand before each release, and we want you to change that.", - "evidence_ids": ["ev-placeholder-ci-checks"] + "requirement_quote": "Keep docs-as-code habits healthy: all docs live in a Git repository as Markdown and are reviewed in pull requests.", + "evidence_ids": [ + "ev-placeholder-docs-pipeline", + "ev-placeholder-ci-checks" + ] } ], "unproven": [] @@ -183,11 +209,11 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil "name": "Copperline Example Co.", "stored_reason": "Makes tools for writers, which is the work Robin cares about most." }, - "verdict_reason": "This is a strong fit. You would own the docs platform from day one with real autonomy, the team already uses AI tools daily, and docs are part of done. The main thing to check is whether reading Go code to follow the build tool is something you want to take on.", + "verdict_reason": "This is a strong fit. You'd build the docs platform for a company making writer tools, with AI drafting and docs-as-code already standard, real autonomy, and pay at your target. The main question is how deep your Go code reading needs to be.", "keyword_signals_found": [ { "phrase": "definition of done", - "read": "Documentation is treated as part of finished work for every feature, which means the docs team has real weight." + "read": "Documentation counts as part of finished work, not extra." } ], "new_signals_to_consider": [] @@ -196,10 +222,10 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil --- -## Judgment calls least certain about: +**Judgment calls I'm least sure of:** -1. **CI automation rated "weak"** — The posting says "The docs build is still run by hand before each release, and we want you to change that." I read this as a gap (they don't have automation now), but it could also be read as an opportunity or a "fair" rating: the company clearly values automation, recognizes it matters, and is asking you to build it. The team isn't indifferent; they're asking for the fix. +1. **CI for docs rated "fair"** — Posting says "The docs build is still run by hand before each release, and we want you to change that." I read this as an opportunity for Robin to own automating the docs build. But it could also be read as: "they haven't invested in CI yet, so this could be a gap they're not actually resourced to fix." The language "we want you to change that" pushed me toward fair (opportunity) rather than weak (unmet gap). -2. **Go language only appears once, but I weighted it in the verdict** — The requirement is "You can read Go code well enough to follow how our build tool works." This is reading, not writing, and appears only in one sentence. Robin's self_score is 1, which is low. I included it as the main caution in the verdict reason, but one could argue this is overstated (it's just code reading to understand a tool, not writing Go) or understated (the skill gap is real). +2. **Autonomy as "positive"** — The posting says "your manager checks in once a week." I took this as a sign of autonomy (frequent enough for alignment, infrequent enough for independence) because it matches the profile's positive signal exactly. But "checks in once a week" could mean either real autonomy between check-ins, or just routine oversight. The broader context ("You choose the tools and set the priorities") makes the autonomy reading solid, but that one phrase alone is ambiguous. -3. **Autonomy based on one strong quote** — "You choose the tools and set the priorities for the docs system; your manager checks in once a week" is explicit and clear, but it's a single sentence. The context "with one other engineer to review your work" adds a governance layer that might read as closer to neutral autonomy rather than the strongly positive reading I gave it. However, peer code review is standard practice, not a signal of restricted autonomy. +3. **Go code in known_gaps_hit, not self_score_gaps** — The posting requires "You can read Go code well enough to follow how our build tool works." Robin's self_score on Go is 1 (minimal). Go is also listed as a known gap for this lane (`gap_go`). I listed it only in `known_gaps_hit` per the instruction to "list it once only." But one could argue it should flag both as a gap and a skills gap separately for visibility. diff --git a/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md index 159b1d7..f7f627f 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md @@ -1,23 +1,21 @@ # Prompt run: Run on my first posting, on sonnet -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). -- Turns: 1. Cost: $0.14 (the simulated user's calls are not included). +- Turns: 1. Cost: $0.13 (the simulated user's calls are not included). - Grade: see `GRADES.md`. ## Checks run by code on the reply ```text -$ python3.11 assessment/scripts/check_findings.py /findings.json --posting fixtures/postings/01-strong-fit.md --profile fixtures/robin-sample/career-profile.yaml +$ python assessment/scripts/check_findings.py /findings.json --posting fixtures/postings/01-strong-fit.md --profile fixtures/robin-sample/career-profile.yaml check_findings: ok (exit code 0) ``` ```text -$ python3.11 scripts/assess_offline.py fixtures/postings/01-strong-fit.md --findings /findings.json --profile fixtures/robin-sample/career-profile.yaml --out /out --date 2026-10-01 +$ python scripts/assess_offline.py fixtures/postings/01-strong-fit.md --findings /findings.json --profile fixtures/robin-sample/career-profile.yaml --out /out --date 2026-10-01 ASSESSMENT: fixtures/postings/01-strong-fit.md 🚦 VERDICT: Apply @@ -32,11 +30,11 @@ Culture 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 | Check | Rating | Read | |---|---|---| -| ✨ AI-forward team | 🟢 Strong | The team drafts with AI every day. A person still checks the output, and you would help decide those checks. This is what you want. | -| 📦 Docs live in version control | 🟢 Strong | The docs are kept in Git and reviewed like code. A miss here would matter a lot, and this clearly passes. | -| 🧪 Docs are built and checked automatically | 🟡 Fair | The build is not automatic yet, so the checking you like is not in place. They want you to build it, which is work you can do. | -| 🦸 Owns the docs system | 🟢 Strong | You would own the docs system from week one, with one engineer to review your work. This is a bonus and it is clearly there. | -| How much say you'd have | 🟢 Strong | You would pick the tools and set your own priorities, and your manager checks in weekly. That is a lot of say. | +| ✨ AI-forward team | 🟢 Strong | The team drafts with AI every day and still wants people to check the output. That is the setup you asked for. | +| 📦 Docs live in version control | 🟢 Strong | The docs already live in Git and go through pull request review. This is the way you like to work. | +| 🧪 Docs are built and checked automatically | 🟡 Fair | The build is not automatic today. That is a small miss on a nice-to-have, but the job is to fix it, and you have built checks like this before. | +| 🦸 Owns the docs system | 🟢 Strong | You would own the docs system, with one engineer reviewing. This is a bonus for you, not a requirement. | +| How much say you'd have | 🟢 Strong | You would pick the tools and set your own priorities. Your manager checks in once a week. That is a lot of say. | 🟢 Strong · 🟡 Fair · 🟠 Weak · 🔴 Poor · ❓ Unknown 1/10 validate the profile @@ -66,57 +64,48 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil ### Model -I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scores or verdict. No hard block trips: the posting says "We are a remote company with about 60 people," and there is no gambling or weapons content. +Hard blocks: none tripped. The posting says "We are a remote company with about 60 people", so the not-remote block does not apply. Copperline makes writing tools, so gambling and weapons don't apply either. I assessed it in the `docs-platform` lane, the one named in its lane field. + +I haven't run `check_findings.py` or `assess_offline.py`; that's for you. The `posting_file` path is the one you attached, not an archived note. ```json { "posting_file": "fixtures/postings/01-strong-fit.md", "lane": "docs-platform", - "hard_block": { - "tripped": false, - "id": null, - "quote": null, - "named_exception": null - }, - "job_type_override": { - "fired": false, - "skill": null, - "quotes": [], - "evidence_checked": [] - }, + "hard_block": {"tripped": false, "id": null, "quote": null, "named_exception": null}, + "job_type_override": {"fired": false, "skill": null, "quotes": [], "evidence_checked": []}, "requirements": [ { "id": "ai_forward", "rating": "strong", "quote": "Our writers and engineers draft with AI assistants every day, and you will help decide which checks a human must still make.", - "read": "The team drafts with AI every day. A person still checks the output, and you would help decide those checks. This is what you want." + "read": "The team drafts with AI every day and still wants people to check the output. That is the setup you asked for." }, { "id": "docs_as_code", "rating": "strong", "quote": "all docs live in a Git repository as Markdown and are reviewed in pull requests", - "read": "The docs are kept in Git and reviewed like code. A miss here would matter a lot, and this clearly passes." + "read": "The docs already live in Git and go through pull request review. This is the way you like to work." }, { "id": "ci_for_docs", "rating": "fair", "quote": "The docs build is still run by hand before each release, and we want you to change that.", - "read": "The build is not automatic yet, so the checking you like is not in place. They want you to build it, which is work you can do." + "read": "The build is not automatic today. That is a small miss on a nice-to-have, but the job is to fix it, and you have built checks like this before." }, { "id": "solo_ownership", "rating": "strong", "quote": "You will own the documentation system from the first week, with one other engineer to review your work.", - "read": "You would own the docs system from week one, with one engineer to review your work. This is a bonus and it is clearly there." + "read": "You would own the docs system, with one engineer reviewing. This is a bonus for you, not a requirement." } ], "autonomy": { "net": "positive", "quotes": [ - "You choose the tools and set the priorities for the docs system; your manager checks in once a week.", - "You will own the documentation system from the first week" + "You choose the tools and set the priorities for the docs system; your manager checks in once a week." ], - "read": "You would pick the tools and set your own priorities, and your manager checks in weekly. That is a lot of say." + "read": "You would pick the tools and set your own priorities. Your manager checks in once a week. That is a lot of say." }, "pay": { "stated": true, @@ -140,24 +129,12 @@ I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scor }, "culture": { "perks": [ - { - "perk_id": "unlimited_pto", - "quote": "Unlimited paid time off, with a three-week minimum we ask everyone to take." - }, - { - "perk_id": "learning_budget", - "quote": "A $3,000 yearly learning budget." - }, - { - "perk_id": "offsites", - "quote": "Two team meetups a year." - } + {"perk_id": "unlimited_pto", "quote": "Unlimited paid time off, with a three-week minimum we ask everyone to take."}, + {"perk_id": "learning_budget", "quote": "A $3,000 yearly learning budget."}, + {"perk_id": "offsites", "quote": "Two team meetups a year."} ], "strong_positive_phrases": [ - { - "phrase": "definition of done", - "quote": "Documentation is part of the definition of done for every feature." - } + {"phrase": "definition of done", "quote": "Documentation is part of the definition of done for every feature."} ], "low_time_off": null, "hustle": [], @@ -170,21 +147,25 @@ I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scor "known_gaps_hit": ["gap_go"], "self_score_gaps": [], "working_style_mismatch": false, + "unproven": [], "matches": [ { "requirement_quote": "Experience generating API reference pages from an OpenAPI file.", "evidence_ids": ["ev-northwind-api-rebuild"] }, { - "requirement_quote": "Build and run the system that turns our documentation into a public help site.", + "requirement_quote": "all docs live in a Git repository as Markdown and are reviewed in pull requests", "evidence_ids": ["ev-placeholder-docs-pipeline"] }, { - "requirement_quote": "all docs live in a Git repository as Markdown and are reviewed in pull requests", - "evidence_ids": ["ev-placeholder-docs-pipeline"] + "requirement_quote": "The docs build is still run by hand before each release, and we want you to change that.", + "evidence_ids": ["ev-placeholder-docs-pipeline", "ev-placeholder-ci-checks"] + }, + { + "requirement_quote": "Our writers and engineers draft with AI assistants every day", + "evidence_ids": ["ev-placeholder-style-prompts"] } - ], - "unproven": [] + ] }, "avoid_core_skills": [], "company_read": { @@ -192,7 +173,7 @@ I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scor "name": "Copperline Example Co.", "stored_reason": "Makes tools for writers, which is the work Robin cares about most." }, - "verdict_reason": "This looks like a strong match. You would own the docs system, choose the tools and work with AI drafting every day. The pay range for your region is listed, and the time off and learning budget are good. The one thing to prepare for is reading Go code, which you have not done in years.", + "verdict_reason": "You would own the docs system, choose the tools and work with AI drafting every day, at a company that makes tools for writers. Two things still need checking: how well you can read Go, and which pay region applies to you.", "keyword_signals_found": [ { "phrase": "definition of done", @@ -203,18 +184,10 @@ I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scor } ``` -**Judgment calls I was least sure of** - -1. **Go as a gap that counts (`known_gaps_hit: ["gap_go"]`).** - - Posting words: "You can read Go code well enough to follow how our build tool works." It sits under "What we are looking for." - - I counted it because it is a required line. The other reading is that "well enough to follow" is a light ask, not load-bearing, so it should not count. Dropping it would raise Qualifications. It is named once in a list and not in the duties, so I did not fire the job-type override. +## Judgment calls I was least sure of -2. **`ci_for_docs` rated `fair`.** - - Posting words: "The docs build is still run by hand before each release, and we want you to change that." - - I read this as automatic checking missing today, but wanted in the role. It could instead be `strong`, because the job is to build exactly what you listed and you have evidence for it. It could also be `weak`, because the posting confirms the opposite of the current state. +1. **Go as a known gap.** The posting says: "You can read Go code well enough to follow how our build tool works." I counted it as a required line, so `gap_go` is in `known_gaps_hit`, and I left it out of `self_score_gaps`. It could also be read as a light, read-only ask that is not load-bearing, in which case `known_gaps_hit` would be empty. I did not fire the job-type override. Go appears once, in the requirements list, and never in the duties. -3. **`working_style_mismatch: false`.** - - Posting words: "You will own the documentation system from the first week." Your profile says you gather knowledge from experts. - - I read this as owning a system, not becoming the subject expert on a product. If the posting meant you would be the sole authority on how the platform works, it could be read as a mismatch. +2. **`ci_for_docs` as `fair`.** The posting says: "The docs build is still run by hand before each release, and we want you to change that." I read it as a small miss on a soft item that the job is meant to fix. It could be `strong`, because the role is to build exactly this and you have done it. It could also be `weak`, because the team does not have this today. -I did not cite `ev-placeholder-ci-checks` or the AI prompt evidence for any match, to avoid stretching the claims. I also left the pay-range benefits term out, because the schema has no field for it. +3. **Time off left as `null`.** The posting says: "Unlimited paid time off, with a three-week minimum we ask everyone to take." I treated it as a perk, not a stated yearly day count. If you read "three-week minimum" as 15 paid days, `low_time_off` would be `{"days": 15, "quote": "..."}`. That sits exactly on your `low_time_off_days` threshold, so the script would decide whether it counts as low. diff --git a/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md index fd564ae..4638859 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md @@ -1,8 +1,6 @@ # Prompt run: Test the scoring with the fixture, on haiku -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.06 (the simulated user's calls are not included). @@ -18,109 +16,136 @@ I'm attaching ARCHITECTURE.md, fixtures/robin-sample/career-profile.yaml, the th ### Model -I'll work through each posting systematically, applying the scoring rules from ARCHITECTURE.md. +I'll work through each posting systematically, applying the scoring rules by hand. --- -## **Posting 01: strong-fit (docs-platform lane)** +## Posting 01: strong-fit (docs-platform lane) + +**FIT:** +- Start: 10 +- ai_forward (strong must-have): strong rating → 0 cost +- docs_as_code (strong must-have): strong rating → 0 cost +- ci_for_docs (soft must-have): weak rating → 1 cost +- solo_ownership (bonus): fair rating → 0 cost (bonuses never cost) +- autonomy: net positive → 0 cost + +10 − 1 = **9** -**Fit:** Start at 10. -- `ai_forward` (strong): "strong" rating → 0 -- `docs_as_code` (strong): "strong" rating → 0 -- `ci_for_docs` (soft): "weak" rating → -1 -- `solo_ownership` (soft, bonus): "fair" rating, but bonus never costs → 0 -- Autonomy: "positive" → 0 -- **10 - 1 = 9** +**COMP:** +- Pay stated. Location "Region B" matches tier exactly. Top: $104,000. +- Robin's target: $100,000. +- 104,000 ≥ 100,000 (target) → **10** -**Comp:** Pay listed. Region B tier: max $104,000. Robin's target is $78,000. $104,000 ≥ target → **10** +**QUALIFICATIONS:** +- Start: 10 +- Years: 7 required, 9 have → no cost +- Known gap (go_lang, score 1): 1 cost +- Evidence for docs_as_code and OpenAPI covered. -**Qualifications:** Start at 10. -- Years: 7+ required, Robin has 9 → 0 -- Known gap `go_lang` → -1 -- **10 - 1 = 9** +10 − 1 = **9** -**Culture:** Start at 5. -- `unlimited_pto` (big) → +2 -- `learning_budget` (big) → +2 -- `offsites` (nice) → +1 -- "definition of done" (strong positive phrase) → +1 -- **5 + 2 + 2 + 1 + 1 = 11, capped at 10 = 10** +**CULTURE:** +- Start: 5 +- unlimited_pto (big): +2 +- learning_budget (big): +2 +- offsites (nice): +1 +- "definition of done" (strong positive phrase): +1 +- Total: 5 + 2 + 2 + 1 + 1 = 11 → capped at **10** -**Verdict:** Hard block: no. Override: no. Fit 9 > 5: no skip. Qualifications 9 > 5: no skip. Comp 10 > 3 and Culture 10 > 3 and average (9+9)/2 = 9 ≥ 7: no reservations. **Apply, rule 4d** +**Verdict:** No hard block, 4a doesn't fire, Fit=9>5, Qual=9>5, Comp=10>3, Culture=10>3, average=(9+9)/2=9≥7 → **Apply (4d)** --- -## **Posting 02: job-type-override (tech-writing lane)** +## Posting 02: job-type-override (tech-writing lane) + +**FIT:** +- Start: 10 +- expert_access (strong): fair rating → 0 cost +- tooling_voice (strong): fair rating → 0 cost +- style_guide (soft): poor rating → 1 cost +- Avoid core skills: video_tutorials (core daily work, next=avoid) → 1 cost; localization_pm (core daily work, next=avoid) → 1 cost. Cap at 2 total. +- autonomy: unknown → 0 cost + +10 − 1 − 2 = **7** -**Fit:** Start at 10. -- `expert_access` (strong): "fair" → 0 -- `tooling_voice` (strong): "fair" → 0 -- `style_guide` (soft): "poor" → -1 -- Autonomy: "unknown" → 0 -- Avoid skills: `video_tutorials` (core) -1, `localization_pm` (core) -1, capped at 2 total → -2 -- **10 - 1 - 2 = 7** +**COMP:** +- Pay stated. Location "Region B" not in tiers; pick "Everywhere else" tier. Top: $78,000. +- Robin: floor $75,000, min $80,000. +- 78,000 is between floor and min → **3** -**Comp:** Pay listed. Robin's location "Region B" doesn't match any tier. Falls back to "Everywhere else" tier: max $78,000. Robin's floor is $75,000, min is $80,000. $78,000 ≥ floor but below min → **3** +**QUALIFICATIONS:** +- Start: 10 +- Years: 12 required, 9 have → 2 cost (too many years) +- Known gaps: dita_xml (score 0) → 1 cost; mobile_sdk (score 0) → 1 cost -**Qualifications:** Start at 10. -- Years: 12+ required, Robin has 9 → -2 -- Known gaps: `dita_xml` -1, `mobile_sdk` -1 -- **10 - 2 - 1 - 1 = 6** +10 − 2 − 1 − 1 = **6** -**Culture:** Start at 5. -- No perks, phrases, or flags → **5** +**CULTURE:** +- Start: 5 +- No perks, no phrases, no flags, no negatives. -**Verdict:** Hard block: no. Override: yes, `video_tutorials` core daily work with no evidence → **Skip, rule 4a** +**5** + +**Verdict:** No hard block, **4a fires**: video_tutorials is central to daily work ("most of your week") with no evidence → **Skip (4a)** --- -## **Posting 03: unlisted-pay-perks (tech-writing lane)** +## Posting 03: unlisted-pay-perks (tech-writing lane) + +**FIT:** +- Start: 10 +- expert_access (strong): weak rating → 1.5 cost +- tooling_voice (strong): unknown rating → 1.5 cost +- style_guide (soft): fair rating → 0 cost +- autonomy: unknown → 0 cost + +10 − 1.5 − 1.5 = **7** + +**COMP:** +- Pay not listed → **5** -**Fit:** Start at 10. -- `expert_access` (strong): "weak" → -1.5 -- `tooling_voice` (strong): "unknown" → -1.5 -- `style_guide` (soft): "fair" → 0 -- Autonomy: "unknown" → 0 -- **10 - 1.5 - 1.5 = 7** +**QUALIFICATIONS:** +- Start: 10 +- Years: 11 required, 9 have → 2 cost +- Known gap: mobile_sdk (score 0) → 1 cost +- Self-score gap: kubernetes (score 1) → 1 cost -**Comp:** Pay not stated → **5** +10 − 2 − 1 − 1 = **6** -**Qualifications:** Start at 10. -- Years: 11+ required, Robin has 9 → -2 -- Known gap `mobile_sdk` → -1 -- Self-score gap `kubernetes` (score 1) → -1 -- **10 - 2 - 1 - 1 = 6** +**CULTURE:** +- Start: 5 +- extra_days_off (big): +2 +- learning_budget (big): +2 +- "fast-paced" is on Robin's free_phrases list → 0 cost -**Culture:** Start at 5. -- `extra_days_off` (big) → +2 -- `learning_budget` (big) → +2 -- **5 + 2 + 2 = 9** +5 + 2 + 2 = **9** -**Verdict:** Hard block: no. Override: no. Fit 7 > 5, Qualifications 6 > 5: no skip. Comp 5 > 3, Culture 9 > 3, but average (7+6)/2 = 6.5 < 7 → **Apply with reservations, rule 4c-average** +**Verdict:** No hard block, 4a doesn't fire, Fit=7>5, Qual=6>5, Comp=5>3, Culture=9>3, **but average=(7+6)/2=6.5<7** → **Apply with reservations (4c-average)** --- -## **Comparison to fixtures/expected.yaml:** - -| Posting | Score | My answer | Expected | Match | -|---------|-------|-----------|----------|-------| -| **01** | Fit | 9 | 9 | ✓ | -| | Comp | 10 | 10 | ✓ | -| | Qualifications | 9 | 9 | ✓ | -| | Culture | 10 | 10 | ✓ | -| | Verdict | Apply | Apply | ✓ | -| | Trigger | 4d | 4d | ✓ | -| **02** | Fit | 7 | 7 | ✓ | -| | Comp | 3 | 3 | ✓ | -| | Qualifications | 6 | 6 | ✓ | -| | Culture | 5 | 5 | ✓ | -| | Verdict | Skip | Skip | ✓ | -| | Trigger | 4a | 4a | ✓ | -| **03** | Fit | 7 | 7 | ✓ | -| | Comp | 5 | 5 | ✓ | -| | Qualifications | 6 | 6 | ✓ | -| | Culture | 9 | 9 | ✓ | -| | Verdict | Apply with reservations | Apply with reservations | ✓ | -| | Trigger | 4c-average | 4c-average | ✓ | - -**All twelve scores and three verdicts match.** No mismatches. +## Comparison with expected.yaml + +| Posting | Metric | My result | Expected | ✓ | +|---------|--------|-----------|----------|---| +| 01 | Fit | 9 | 9 | ✓ | +| 01 | Comp | 10 | 10 | ✓ | +| 01 | Qual | 9 | 9 | ✓ | +| 01 | Culture | 10 | 10 | ✓ | +| 01 | Verdict | Apply | Apply | ✓ | +| 01 | Trigger | 4d | 4d | ✓ | +| 02 | Fit | 7 | 7 | ✓ | +| 02 | Comp | 3 | 3 | ✓ | +| 02 | Qual | 6 | 6 | ✓ | +| 02 | Culture | 5 | 5 | ✓ | +| 02 | Verdict | Skip | Skip | ✓ | +| 02 | Trigger | 4a | 4a | ✓ | +| 03 | Fit | 7 | 7 | ✓ | +| 03 | Comp | 5 | 5 | ✓ | +| 03 | Qual | 6 | 6 | ✓ | +| 03 | Culture | 9 | 9 | ✓ | +| 03 | Verdict | Apply with reservations | Apply with reservations | ✓ | +| 03 | Trigger | 4c-average | 4c-average | ✓ | + +**Perfect match: all 12 scores and 3 verdicts correct.** diff --git a/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md index 6b1c155..09faf8c 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md @@ -1,8 +1,6 @@ # Prompt run: Test the scoring with the fixture, on sonnet -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.12 (the simulated user's calls are not included). @@ -18,65 +16,66 @@ I'm attaching ARCHITECTURE.md, fixtures/robin-sample/career-profile.yaml, the th ### Model -I did all three by hand from the findings files, the profile and ARCHITECTURE.md. I compared against `fixtures/expected.yaml` only after finishing the arithmetic. I didn't call a tool to open it because you attached its contents, so I read it from the attachment. +I scored all three from the findings files, then compared with `fixtures/expected.yaml`. The expected file was already pasted in your message, so I didn't open anything. I wrote the arithmetic before comparing. ## Posting 01: Copperline, docs-platform lane | Score | Arithmetic | Result | |---|---|---| -| Fit | 10. `ai_forward` (strong must-have) rated strong: 0. `docs_as_code` (strong) rated strong: 0. `ci_for_docs` (soft) rated weak: -1. `solo_ownership` is a bonus: 0. Autonomy net positive: 0. No avoid-core skills: 0. | **9** | -| Comp | The tier is Region B, matching Robin's location label. Top is 104,000, which is at or above the target of 78,000. | **10** | -| Qualifications | 10. Asks for 7 years and Robin has 9: 0. Known gap `go_lang`: -1. No self-score gaps. No style mismatch. | **9** | -| Culture | 5. Unlimited PTO (big) +2, learning budget (big) +2, meetups (nice) +1, "definition of done" (strong positive) +1. That is 11, capped at 10. | **10** | +| Fit | 10. Strong must-haves `ai_forward` and `docs_as_code` are both rated strong: 0. Soft `ci_for_docs` is rated weak: -1. `solo_ownership` is a bonus: 0. Autonomy is positive: 0. No avoid-skills: 0. 10 - 1 = 9. | **9** | +| Comp | Robin's location is Region B, which matches a tier label. The top is 104,000, which is at or above the 100,000 target. | **10** | +| Qualifications | 10. The posting asks for 7 years and Robin has 9: 0. Known gap `go_lang`: -1. There are no separate self-score gaps, and Go would count once anyway. 10 - 1 = 9. | **9** | +| Culture | 5. Unlimited PTO (big) +2, learning budget (big) +2, meetups (nice) +1, "definition of done" +1. That is 11, capped at 10. | **10** | **Verdict:** -- Rule 1: no hard block. -- Rule 2: override not fired. -- Rule 3: Fit 9 and Qualifications 9 are both above 5. -- Rule 4: Comp 10 and Culture 10 are above 3, and the average of Fit and Qualifications is 9, which is at least 7. -- Rule 5 decides: **Apply (4d)**. +1. No hard block. +2. The override is not fired. +3. Fit 9 and Qualifications 9 are both above 5. +4. Comp 10 and Culture 10 are both above 3, and the average of Fit and Qualifications is 9, which is not under 7. +5. Nothing fired, so rule 4d applies: **Apply**. ## Posting 02: Driftmark, tech-writing lane | Score | Arithmetic | Result | |---|---|---| -| Fit | 10. `expert_access` (strong) fair: 0. `tooling_voice` (strong) fair: 0. `style_guide` (soft) poor: -1. Autonomy unknown: 0. Avoid-core skills are `video_tutorials` and `localization_pm`, -1 each, which is within the cap of 2: -2. 10 - 1 - 2 = 7. | **7** | -| Comp | No tier label contains "Region B", so the code uses the "Everywhere else" tier. Top is 78,000, which is at or above the floor of 75,000 but below the min of 80,000. | **3** | -| Qualifications | 10. Asks for 12 years and Robin has 9: -2, once. Known gaps `dita_xml` and `mobile_sdk`: -2. Nothing else. 10 - 2 - 2 = 6. | **6** | -| Culture | 5. No perks, phrases, flags or hustle language. | **5** | +| Fit | 10. `expert_access` and `tooling_voice` are strong must-haves rated fair: 0. Soft `style_guide` is rated poor: -1. Autonomy is unknown: 0. Avoid-skills `video_tutorials` and `localization_pm` cost 1 each, within the cap of 2: -2. 10 - 1 - 2 = 7. | **7** | +| Comp | No tier label contains "Region B", so the code takes the "Everywhere else" tier. Its top is 78,000, which is at or above the 75,000 floor but below the 80,000 min. | **3** | +| Qualifications | 10. 12 years asked against Robin's 9: -2. Known gaps `dita_xml` and `mobile_sdk`: -1 each, -2 in total, under the cap of 4. 10 - 2 - 2 = 6. | **6** | +| Culture | 5. There are no perks, hustle phrases, flags or negative phrases. | **5** | **Verdict:** -- Rule 1: no hard block. -- Rule 2 decides: the findings say a required skill central to the daily work, video tutorials, has no evidence behind it (`evidence_checked` is empty). **Skip (4a)**. - -The scores would not have changed this outcome: Comp is 3 (so 4c would also apply) and the Fit/Qualifications average is 6.5. But rule 2 comes before rule 4, so 4a decides. +1. No hard block. +2. The findings mark `job_type_override.fired = true` for `video_tutorials`, which has no evidence behind it. Rule 4a fires and decides: **Skip**. +3. Nothing after rule 2 is checked. Fit 7 and Qualifications 6 would have passed 4b anyway, and Comp 3 would have triggered 4c. ## Posting 03: Ashgrove, tech-writing lane | Score | Arithmetic | Result | |---|---|---| -| Fit | 10. `expert_access` (strong) weak: -1.5. `tooling_voice` (strong) unknown: -1.5. `style_guide` (soft) fair: 0. Autonomy unknown: 0. 10 - 1.5 - 1.5 = 7. | **7** | -| Comp | Pay is not listed, so the neutral score applies. | **5** | -| Qualifications | 10. Asks for 11 years and Robin has 9: -2. `mobile_sdk` known gap: -1. `kubernetes` self-score gap on a required line: -1. Total 3, within the cap of 4. 10 - 2 - 1 - 1 = 6. | **6** | -| Culture | 5. Extra days off (big) +2, learning budget (big) +2. "Fast-paced" is on Robin's free-phrases list, so it costs nothing. 5 + 2 + 2 = 9. | **9** | +| Fit | 10. `expert_access` is a strong must-have rated weak: -1.5. `tooling_voice` is strong and unknown: -1.5. Soft `style_guide` is rated fair: 0. Autonomy is unknown: 0. 10 - 1.5 - 1.5 = 7. | **7** | +| Comp | Pay is not listed. | **5** | +| Qualifications | 10. 11 years asked against Robin's 9: -2. Known gap `mobile_sdk`: -1. Self-score gap `kubernetes`: -1. 10 - 2 - 1 - 1 = 6. | **6** | +| Culture | 5. Extra days off (big) +2, learning budget (big) +2. "Fast-paced" is a free phrase: 0. 5 + 2 + 2 = 9. | **9** | **Verdict:** -- Rule 1: no hard block. -- Rule 2: override not fired. -- Rule 3: Fit 7 and Qualifications 6 are both above 5. -- Rule 4 decides: Comp 5 and Culture 9 are above 3, but the average of Fit and Qualifications is (7 + 6) / 2 = 6.5, which is under 7. **Apply with reservations (4c-average)**. +1. No hard block. +2. The override is not fired. +3. Fit 7 and Qualifications 6 are both above 5. +4. Comp 5 and Culture 9 are both above 3, but the average of Fit and Qualifications is (7 + 6) / 2 = 6.5, which is under 7. +5. Rule 4c (average) fires: **Apply with reservations**. ## Comparison with `fixtures/expected.yaml` | Posting | Fit | Comp | Qual | Culture | Verdict | Trigger | |---|---|---|---|---|---|---| -| 01 | 9 ✓ | 10 ✓ | 9 ✓ | 10 ✓ | Apply ✓ | 4d ✓ | -| 02 | 7 ✓ | 3 ✓ | 6 ✓ | 5 ✓ | Skip ✓ | 4a ✓ | -| 03 | 7 ✓ | 5 ✓ | 6 ✓ | 9 ✓ | Apply with reservations ✓ | 4c-average ✓ | +| 01 | 9 = 9 | 10 = 10 | 9 = 9 | 10 = 10 | Apply = Apply | 4d = 4d | +| 02 | 7 = 7 | 3 = 3 | 6 = 6 | 5 = 5 | Skip = Skip | 4a = 4a | +| 03 | 7 = 7 | 5 = 5 | 6 = 6 | 9 = 9 | Apply with reservations = same | 4c-average = 4c average | -**Mismatches: none.** All twelve scores, all three verdicts and all three trigger labels agree with the file. Posting 03 also matches the worked example in ARCHITECTURE.md. +**Mismatches: none.** All twelve scores and all three verdicts and triggers match. -Three steps could have gone differently: -- **01 Culture:** the raw sum is 11, so the cap at 10 matters. -- **02 Comp:** it only lands on 3 if you pick the "Everywhere else" tier, which is what Interpretation 7 says to do when no tier names Region B. Using Region A's top of 98,000 would give 8. -- **02 Fit:** it needs the avoid-skills cap of 2 applied across both listed skills, with no stacking beyond that. +A few steps depended on judgment or on the rules' fine print: +- **Posting 01 Culture:** the raw total is 11, so the match depends on clamping to the cap of 10 before rounding. +- **Posting 02 Comp:** it depends on Interpretation 7, where the code picks the "Everywhere else" tier. Using Region A (top 98,000) would have scored 8. +- **Posting 02 Fit:** it depends on the two avoid-skills being treated as core daily work, which the findings assert. +- **Qualifications gaps in all three:** I counted every known gap as load-bearing, as the findings imply. That is the judgment call the document says can vary. diff --git a/pipelines/job-assessment/tests/prompt-runs/p5-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p5-haiku.md index 3aeb67c..5401a35 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p5-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p5-haiku.md @@ -12,56 +12,108 @@ [the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/gappy-profile.yaml] -I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. ### Model -I'll systematically check what the validator cannot catch. Here are the weak spots: +# Gaps a careful reader should find + +I'm finding five real problems, each grounded in what's actually written: --- -## **Planted gaps (3 major ones):** +## 1. **Contradictory requirement in tech-writing lane** -### 1. **`skills[1].docs_ci` — Score 5 backed only by unchecked interview** -- **YAML path:** `skills[1]` (or `skills[1].self_score`, `skills[1].evidence_ids`) -- **Why it matters:** A score of 5 signals mastery. The validator won't warn about weak evidence *within* an evidence list—only absence of evidence. This evidence is `proof: unchecked` from an interview with no artifact to verify, yet it's being scored as your highest skill. -- **Question:** Can you point to a documented artifact (a PR, a config file, CI logs, or documentation) that shows the link-checking build you set up, or was this pairing work where you implemented it together with someone else? +**YAML path:** `lanes[1].requirements[1]` ---- +**Quote:** +```yaml +- id: owns_pipeline + label: "🛠️ Owns the docs build" + severity: strong + why: "The point of the role is building, not only writing." +``` + +**Why it matters for scoring:** The lane itself is described as *"Roles writing and editing technical content,"* yet it requires—at strong severity—that the role be about *building, not only writing.* A scoring system will penalize pure writing roles that don't own docs infrastructure, even though they fit the lane's stated purpose. This contradicts what the lane claims to target. -### 2. **`hard_blocks[1].why` — Weapons hard block has no reason** -- **YAML path:** `hard_blocks[1]` (the weapons block) -- **Why it matters:** The validator warns about `strong` must-haves with no reason, but hard blocks aren't checked for this. Without a stated reason, a recruiter can't tell if this is an ethical boundary, a legal constraint, or something else—affecting how they'd present the role to you. -- **Question:** Why won't you work for weapons makers? Is it a personal value, a concern about the industry, or something else? +**Question:** For roles that focus primarily on writing and editing content, do you really need them to own the docs build, or should that be a strong requirement only for the docs-platform lane? --- -### 3. **`lanes[1].requirements` — Tech-writing lane requirements copied verbatim from docs-platform** -- **YAML path:** `lanes[1].requirements` (and `lanes[1].autonomy`, `lanes[1].keyword_signals`) -- **Why it matters:** These roles are fundamentally different (one builds systems, one writes content), yet both demand "Owns the docs build" as a strong requirement. The validator can't catch copy-paste; only humans notice when a requirement doesn't match the role's actual needs. -- **Question:** For a pure tech-writing role, do you need to own the docs build system itself, or would you prefer to write excellent content *within* a system that already exists? +## 2. **High skill score with thin, unchecked evidence** + +**YAML path:** `skills[1]` (docs_ci) + +**Quote:** +```yaml +- id: docs_ci + label: Docs build and CI checks + self_score: 5 + ... + evidence_ids: [ev-placeholder-docs-ci] +``` + +**Evidence details:** This single evidence spans only **2023-05 to 2023-06** (two months), sourced from an **unchecked interview**, with no document link or secondary verification. + +**Why it matters for scoring:** A score of 5 (expert) typically signals deep, demonstrated mastery. Two months of work mentioned in an interview is insufficient to support expert-level scoring in most hiring contexts. A hiring manager would reasonably expect multiple projects, production evidence, or a documented body of work. + +**Question:** Beyond that 2023 project, what other docs build or CI systems have you set up or maintained? --- -## **Additional weak spots:** +## 3. **Authorship mischaracterization inflates platform claims** + +**YAML path:** `evidence[1].authorship` + +**Quote:** +```yaml +- id: ev-placeholder-docs-ci + claim: "Set up a build check that blocks a docs change with a broken link." + authorship: WROTE + source: {type: interview, ...} +``` -### 4. **`evidence[0].authorship: DIRECTED` — Thin for a claimed skill** -- **YAML path:** `skills[0].evidence_ids[0]` → `evidence[0].authorship` -- **Why it matters:** You scored OpenAPI 4, but the evidence shows you *directed* the API spec work, not that you implemented the spec generation or wrote against it. For a technical writer, this means you supervised someone else's work. -- **Question:** Did you implement the system that generates the API reference from the spec, or did you write the reference documentation after someone else built the system? +**Why it matters for scoring:** Setting up a CI/build-check system is platform or infrastructure work, not writing. Marking it as "WROTE" implies authorship of written content, not implementation of technical systems. This conflates different skill domains and makes Robin's platform engineering background appear stronger than a technical writer setting up CI tooling might justify. + +**Question:** Did you write the code or configuration that implements this check, or did you spec/design it and have an engineer implement it? --- -### 5. **Last-used dates don't match evidence end dates** -- **YAML path:** `skills[0].last`, `skills[1].last`, `skills[2].last` -- **Why it matters:** All three scored skills claim `last: "2y"`, but evidence shows: - - `openapi`: last evidence 2021-09 (4.8 years ago) - - `docs_ci`: last evidence 2023-06 (3.3 years ago) - - `style_guides`: last evidence 2022-08 (4.1 years ago) - - The validator can't connect `last` to evidence dates; a mismatch suggests either the skill is rustier than claimed, or the profile wasn't kept in sync. -- **Question for each:** When did you actually *last use* each of these skills hands-on, and what did you do? +## 4. **Writing sample does not match evidence** + +**YAML path:** `writing_samples[0]` + +**Quote:** +```yaml +- id: ws-api-guide + title: "API getting-started guide" + link: "https://example.com/fictional/getting-started" + evidence_id: ev-northwind-api-rebuild +``` + +**The evidence (ev-northwind-api-rebuild) claims:** *"Moved the API reference from hand-edited pages to pages generated from the API spec."* + +**Why it matters for scoring:** The sample title promises a guide (writing), but the evidence describes tooling and automation (not writing). A hiring manager clicking the link expects to see Robin's writing work, not an automation project. This mismatch makes the portfolio appear misleading. + +**Question:** Did you write the API getting-started guide yourself, or is the link actually to the auto-generated reference documentation? --- -**Summary:** The validator missed the three planted gaps (weak evidence quality, missing hard-block reason, and cloned requirements) because it checks *structure*, not *judgment*. The `last` dates are the only automatic error it could have caught but didn't, suggesting the profile wasn't updated when the `last` field was filled in. +## 5. **Platform engineer role aspiration vs. actual titles** + +**YAML path:** `person.target_roles[1]` + `employers[*].title` + +**Quote:** +```yaml +target_roles: [Senior Technical Writer, "Docs Platform Engineer"] +``` +vs. actual titles: +```yaml +employers: + - title: "Technical Writer" # 2017–2021 + - title: "Senior Technical Writer" # 2022–present +``` + +**Why it matters for scoring:** Robin targets "Docs Platform Engineer" but has held only writer titles. While the evidence suggests platform work within those roles, the lack of an actual platform engineer title—combined with scores of 5 in docs_ci—could read as role inflation. A hiring manager evaluating for a platform engineer role might ask whether Robin has been trusted with that title, or only did platform tasks as a side project. + +**Question:** Have you held any position officially titled "Platform Engineer," "Infrastructure Engineer," or "Systems Engineer," or has your platform work all occurred within technical writing roles? diff --git a/pipelines/job-assessment/tests/prompt-runs/p5-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p5-sonnet.md index e609d09..ec85fab 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p5-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p5-sonnet.md @@ -3,7 +3,7 @@ - Prompt file: `prompts/05-find-gaps.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). -- Turns: 1. Cost: $0.05 (the simulated user's calls are not included). +- Turns: 1. Cost: $0.09 (the simulated user's calls are not included). - Grade: see `GRADES.md`. ## Conversation @@ -12,64 +12,81 @@ [the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/gappy-profile.yaml] -I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. Find the weak spots the validator cannot catch: skills scored 3 or higher that rest on thin evidence, unchecked evidence a posting would lean on, evidence with no dates, must-haves or hard blocks with no reason, and lanes whose requirements look copied from each other. For each, give the exact YAML path, why it matters for scoring, and one question you would ask me to fix it. Do not fill any gap yourself. A good answer on fixtures/gappy-profile.yaml finds all three planted gaps. +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. ### Model -I found the three planted gaps, plus a few smaller ones. I changed nothing in the file. +I found 13 weak spots, ordered roughly by how much they would distort scoring. Each one gives the path, the quoted text, why it matters, and one question for you. I haven't filled in any answers. -## 1. Skills scored 3+ on thin evidence +## Skills and evidence -**`skills[1]` (`docs_ci`), `self_score: 5`, `evidence_ids: [ev-placeholder-docs-ci]`** -- **Why it matters:** The top score rests on one item. That item is `unchecked`, comes from an interview (`evidence[1].source`), and spans one month (`evidence[1].dates`: 2023-05 to 2023-06). `how` is `[self]` only, so nobody else vouches for it. A posting that leans on CI and build skill would get a 5 that nothing backs up. -- **Question:** "Besides the broken-link check, what else have you built or maintained in the docs pipeline, and is there a commit, config file or teammate who could confirm it?" +**1. `skills[1].self_score` (docs_ci)** +- Quote: `self_score: 5`, `how: [self]`, `evidence_ids: [ev-placeholder-docs-ci]` +- Why it matters: The top score rests on one item, `evidence[1]`, which is `proof: unchecked`, comes from an interview, and has `dates: {start: 2023-05, end: 2023-06}`. That is one month. A posting that asks for CI skill gets maximum credit for a claim nothing in the file backs up, and it is the same skill your strong `owns_pipeline` requirement leans on. +- Question: Beyond that one month in 2023, what else shows you have worked on docs builds and CI checks, and what would a reviewer be able to open to see it? -**`skills[0]` (`openapi`), `self_score: 4`** -- **Why it matters:** The only evidence is `evidence[0]`, which is dated 2021 and marked `authorship: DIRECTED`. That shows you led the change, not that you hands-on wrote OpenAPI specs. `last: 2y` also doesn't match evidence that ends in 2021-09. -- **Question:** "Have you written or edited OpenAPI specs yourself, and when did you last do it?" +**2. `skills[0].last`, `skills[2].last`, `skills[1].last` (all `2y`) against the evidence dates** +- Quote: openapi `last: 2y`, style_guides `last: 2y`, docs_ci `last: 2y` +- Why it matters: Today is 2026-10-01. The newest evidence for openapi ends 2021-09, for style_guides 2022-08, and for docs_ci 2023-06. That is about 5, 4 and 3 years ago. If recency is used in scoring, all three are overstated, and the file itself shows no more recent use. +- Question: For each of these three skills, what is the most recent piece of work you did with it, and where is it recorded? -**`skills[2]` (`style_guides`), `last: 2y`** (minor) -- **Why it matters:** The evidence is one document from 2022 (`evidence[2].dates`), and `last: 2y` is a stretch from there. The 4 itself is reasonably supported. -- **Question:** "Have you updated or applied the style guide since 2022, and is there anything newer to point to?" +**3. `skills[0].how` (openapi)** +- Quote: `how: [self, ai]` +- Why it matters: It claims AI-assisted learning or use for OpenAPI. No evidence item mentions AI and no skill covers AI tools. This is also the only support in the file for your strong `ai_forward` requirement. +- Question: What did you actually do with AI tools on OpenAPI work, and when? -## 2. Unchecked evidence a posting would lean on +**4. `evidence[2].source` and `evidence[2].proof` (style guide)** +- Quote: claim: "Wrote the team style guide **and the review checklist that goes with it**"; `proof: checked`; `ref: "style-guide.pdf#introduction"` +- Why it matters: The claim has two parts, but the checked source points at the introduction. The checklist half is marked verified without any pointer to it, so `skills[2]` (score 4) gets credit for both. +- Question: Which part of the document did you check, and does it show the checklist as well as the guide? -**`evidence[1].proof: unchecked`** (`ev-placeholder-docs-ci`) -- **Why it matters:** This is the only evidence behind the skill scored 5, and it supports the "owns the docs build" requirement in `lanes[0].requirements[1]`. It is the most likely thing for a posting to test, and it is the least verified. -- **Question:** "Is there a link, a repo file, a PR, or a reference who can confirm the link-check work so it can move to `checked`?" +**5. `evidence[0]` (Northwind API rebuild)** +- Quote: `authorship: DIRECTED`, `proof: checked`, details: "Robin's own words: the reference stopped drifting from the real API." +- Why it matters: Two things sit under one `checked` mark. The link may verify that the work existed, but the outcome is described as your own words. `DIRECTED` also sits oddly with `employers[0].title: Technical Writer`, and the file never says what you directed, who you directed, or what generated the pages. This item is the only evidence behind `skills[0]` (score 4). +- Question: What exactly did the checked link show, and who or what did you direct in this project? -**`writing_samples[0].evidence_id`** -- **Why it matters:** The sample is a getting-started guide, but it points at `ev-northwind-api-rebuild`, which is about generating the API reference. The sample doesn't clearly prove that claim, and the evidence is `DIRECTED`, not `WROTE`. -- **Question:** "Did you write the getting-started guide yourself, and does it belong with a different piece of evidence?" +**6. `writing_samples[0].evidence_id`** +- Quote: title "API getting-started guide", `evidence_id: ev-northwind-api-rebuild` +- Why it matters: The linked evidence is about generating the API reference from the spec. A getting-started guide is a different artifact, so the sample may not show the work it is attached to. The file also has no sample from Placeholder Labs. +- Question: Is the getting-started guide part of the API rebuild, and which sample best shows your current work? -## 3. Evidence with no dates +## Lanes -None here. Every evidence item and employer has dates, which is why the validator is silent. Two things to check anyway: -- `evidence[1].dates` is a one-month window. Was the work really that short, or is the range just when it shipped? -- Several skills claim `last: 2y`, but the dated evidence for them is older than that. See section 1. +**7. `lanes[0]` and `lanes[1]` are identical below the description** +- Quote: both lanes have the same `requirements` (`ai_forward`, `owns_pipeline`, `small_team`), `autonomy`, and `keyword_signals`. +- Why it matters: The two lanes cannot score a posting differently, since only the name, emoji and description differ. The validator's "lane with no must-haves of its own" warning doesn't fire because each lane has some. +- Question: What is the one thing that would make you rank a posting higher in tech-writing than in docs-platform, or the other way round? -## 4. Must-haves or hard blocks with no reason +**8. `lanes[1].requirements[1]` (owns_pipeline in tech-writing)** +- Quote: lane description "Roles writing and editing technical content"; requirement `owns_pipeline`, `severity: strong`, why "The point of the role is building, not only writing." +- Why it matters: This contradicts the lane's own description. A pure writing or editing role would take a strong miss in the lane meant for it. +- Question: Is owning the docs build a must-have for the writing lane, or only for docs-platform? -**`hard_blocks[1]` (`weapons`) has no `why`** -- **Why it matters:** The validator only warns about strong must-haves, not hard blocks. A hard block removes a posting outright, so an unexplained one can't be judged on borderline cases such as defense-adjacent software or dual-use tools. -- **Question:** "What is the line for you: only makers of weapons, or also companies that sell software or services to them?" +**9. `lanes[0].requirements[0]` and `lanes[1].requirements[0]` (ai_forward, strong)** +- Quote: `why: Robin wants to keep working with AI tools daily.` +- Why it matters: "Keep working" says this is current practice, yet the skills and evidence contain no AI-tool work apart from the `ai` tag in item 3. A strong requirement therefore penalizes postings on a preference the file doesn't back. +- Question: Which AI tools do you use daily now, and for what? -**`hard_blocks[2].match_hints` includes `hybrid`** (`non_remote`) -- **Why it matters:** The `why` says you won't relocate and work from home, but the hint blocks hybrid roles too. Whether hybrid is acceptable changes which postings get screened out. -- **Question:** "Is a hybrid role with a few office days within walking or commuting range ever acceptable?" +**10. `lanes[*].keyword_signals`** +- Quote: `strong_positive: [{phrase: docs as code, why: Matches how Robin works.}]`, with `positive: []`, `negative: []`, `strong_negative: []` +- Why it matters: Nothing in either lane lowers a score through keywords, so postings can only go up. "Docs as code" is also called how you work, but no skill or evidence names git, static site generators or Markdown tooling. +- Question: Which phrases in a posting would make you want to skip it, and which evidence shows you work docs-as-code? -**`lanes[0].requirements[2]` and `lanes[1].requirements[2]` (`small_team`), `severity: soft`, no `why`** -- **Why it matters:** It's soft, so the validator won't flag it. It still sways scoring, and without a reason it can't be weighed against other signals. -- **Question:** "What does 'small' mean to you in headcount, and what goes wrong for you on a larger team?" +## Hard blocks and culture -## 5. Lanes whose requirements look copied +**11. `hard_blocks[2].match_hints`** +- Quote: `match_hints: [on-site, in-office, hybrid]` +- Why it matters: These are broad. "Hybrid" appears in many developer-platform postings that are about hybrid cloud, and "in-office" can appear in a remote posting's perks text. Either could trigger a hard block on a remote-friendly role. Separately, `hard_blocks[1]` (weapons) has no `why`, unlike the other two blocks. +- Question: Which of these hints do you want to block a posting on its own, and why do you rule out weapons makers? -**`lanes[1]` (`tech-writing`) duplicates `lanes[0]` (`docs-platform`) in:** -- `lanes[1].requirements`: same three ids, severities and reasons. -- `lanes[1].autonomy`: same positive and negative signals. -- `lanes[1].keyword_signals`: same `docs as code` phrase. +**12. `culture.perks[0]` against `culture.low_time_off_days`** +- Quote: `{id: unlimited_pto, ..., kind: big}` and `low_time_off_days: 15` +- Why it matters: You treat 15 days as too few, but you also reward unlimited PTO as a big perk. A posting that says "unlimited PTO" with no stated days gets full credit and can't be checked against your own threshold. +- Question: If a posting says unlimited PTO and gives no number of days, should it count as a big perk, as unknown, or as a flag? -- **Why it matters:** The validator only checks that each lane has some must-haves, not that they differ. With identical rules, both lanes score every posting the same, so the lane choice tells you nothing. `owns_pipeline` as a strong must-have also contradicts `lanes[1].description` ("writing and editing"): a writing-focused role would normally not own the build. That makes the `tech-writing` lane unlikely to match the roles it is meant to find. -- **Question:** "For a role that is mostly writing, which of these must-haves would you actually insist on, and what would you want to see that the docs-platform lane doesn't ask for?" +## Pay and completeness -Related: `lanes[1].known_gaps` is empty while `lanes[0].known_gaps` names Kubernetes. Is Kubernetes really irrelevant to the writing lane, or did that lane just never get the same review? +**13. `comp.stretch_ceiling` and `meta.intake.stages_done`** +- Quote: `target: 100000`, `stretch_ceiling: 175000`; `stages_done: [1, 2, 3, 4, 5, 6]` +- Why it matters: The ceiling is 75% above target and far above `open_ask: 85000`, so a mid-range posting could score as either a good or a poor match. The file also doesn't say whether the numbers mean base or total pay, or per year. Meanwhile `stages_done` claims every intake stage is finished, yet `company_criteria.good_not_dream` is `[]`, three keyword lists are empty, and `evidence` has only three items. +- Question: What do these pay numbers cover, and which intake stage did you actually finish for the empty sections? diff --git a/pipelines/job-assessment/tests/prompt-runs/p5h-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p5h-haiku.md new file mode 100644 index 0000000..3719312 --- /dev/null +++ b/pipelines/job-assessment/tests/prompt-runs/p5h-haiku.md @@ -0,0 +1,75 @@ + +# Prompt run: Find gaps in my central file (held-out profile), on haiku + +- Prompt file: `prompts/05-find-gaps.txt` +- Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). +- Turns: 1. Cost: $0.04 (the simulated user's calls are not included). +- Grade: see `GRADES.md`. + +## Conversation + +### User + +[the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/held-out-gappy-profile.yaml] + +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. + +### Model + +I'll read through this carefully like a hiring manager would. Here are the actual gaps I found: + +--- + +**1. Evidence predates claimed employment** +- **YAML path:** `evidence[0].dates` and `employers[0].dates` +- **Quote:** Evidence `ev-harbor-onboarding` has dates `2018-01` to `2018-06`, but Harbor employment is `2021-03` to `2023-06` +- **Why it matters:** A hiring manager will assume this work was done *at* Harbor. This gap makes the timeline confusing and raises questions about what "work at Harbor" actually means. Did Quinn write this before joining, or is there an employment gap? +- **Question to fix it:** When and where did Quinn actually write this onboarding guide—was it at a different employer, or during a gap before Harbor? + +--- + +**2. Authorship claim contradicts authorship field** +- **YAML path:** `evidence[1].claim` vs `evidence[1].authorship` +- **Quote:** Claim says "Wrote every page of the help center again" but authorship field is `REVIEWED` +- **Why it matters:** Writing vs. reviewing are different skill demonstrations. A hiring manager hiring for a writer will score these very differently. The claim overstates what the field records. +- **Question to fix it:** Did Quinn write or review the help center pages? If reviewed, what percentage or which sections? + +--- + +**3. Skill recency doesn't match evidence** +- **YAML path:** `skills[1]` (user_guides) and its `evidence_ids` +- **Quote:** Skill `user_guides` says `last: 2y` but references `ev-harbor-onboarding` with dates `2018-01` to `2018-06` +- **Why it matters:** Claiming recent use (2y) but showing 8-year-old evidence makes the skill look either stale or inaccurately dated. Scoring algorithms may assume a recent reference means recent use. +- **Question to fix it:** Has Quinn written user guides in the past 2 years? If the last example is from 2018, should this say `last: 8y`? + +--- + +**4. "Fast-paced" appears in both avoid and accept lists** +- **YAML path:** `culture.hustle_phrases` and `culture.free_phrases` +- **Quote:** `hustle_phrases: [rockstar, fast-paced, always on]` and `free_phrases: [fast-paced, flexible hours]` +- **Why it matters:** These lists help recruiters filter opportunities. Listing the same phrase in both "red flag" and "acceptable" signals contradicts itself and blocks filtering logic. +- **Question to fix it:** Does Quinn like fast-paced environments or want to avoid them? It can only be one. + +--- + +**5. Vague-title soft flag doesn't match actual titles** +- **YAML path:** `soft_flags[0]` against `employers[*].title` +- **Quote:** Soft flag warns "Vague job title" but titles are "Technical Writer" and "Senior Technical Writer" +- **Why it matters:** These are standard, non-vague titles. Flagging this raises a false red flag on Quinn's profile and suggests confusion about what vagueness means. +- **Question to fix it:** Are there specific titles Quinn wants to avoid or find unclear? If not, remove this flag. + +--- + +**6. High-interest company claim is unsupported** +- **YAML path:** `company_criteria.high_interest[0]` +- **Quote:** "Builds tools Quinn already uses" +- **Why it matters:** Hiring managers want to know *which* tools Quinn uses. This is too vague to be credible and looks like generic interest rather than informed research. +- **Question to fix it:** What specific tools does Fictional Api Tools Inc. build that Quinn actually uses? Can you name them? + +--- + +**7. Years of experience gap** +- **YAML path:** `person.years_experience` vs `employers` and `evidence` +- **Quote:** `years_experience: 14` but employer history runs 2021–present (~5.5 years shown) and evidence traces back to 2018 (~8 years total) +- **Why it matters:** The 14-year claim doesn't align with the work history provided. Either there's 6 years of employment not listed, or the number is overstated. +- **Question to fix it:** Does Quinn have employment history before 2018, or should `years_experience` reflect the evidence actually shown? diff --git a/pipelines/job-assessment/tests/prompt-runs/p5h-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p5h-sonnet.md new file mode 100644 index 0000000..79f3c43 --- /dev/null +++ b/pipelines/job-assessment/tests/prompt-runs/p5h-sonnet.md @@ -0,0 +1,70 @@ + +# Prompt run: Find gaps in my central file (held-out profile), on sonnet + +- Prompt file: `prompts/05-find-gaps.txt` +- Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). +- Turns: 1. Cost: $0.07 (the simulated user's calls are not included). +- Grade: see `GRADES.md`. + +## Conversation + +### User + +[the prompt below, with attached files: ARCHITECTURE.md, section 'What the validator checks', fixtures/held-out-gappy-profile.yaml] + +I'm attaching the 'What the validator checks' section of ARCHITECTURE.md and my career-profile.yaml. The validator checks shape and consistency, not truth, so find what it cannot catch. Read my whole file the way a skeptical hiring manager and a careful editor would: look for places where the file claims more than its own evidence shows, says something that conflicts with another part of the file, or would make a posting score in a misleading way. Check every section, not only skills. For each weak spot, give the exact YAML path, quote the part that worries you, say why it matters for scoring, and give one question you would ask me to fix it. Do not fill any gap yourself, and do not report a problem you cannot point to in the file. A good answer names only things that are really in the file and gives a YAML path for each one. + +### Model + +I found 10 weak spots. Each one is in the file, and none of them is something the validator checks. + +**1. `hard_blocks[0].match_hints`** +- Quote: `match_hints: [remote, work from home, distributed team]`, under the label "Must be in an office". +- Why it matters: the block is meant to catch office-only postings, but the hints are the words remote postings use. A remote posting would be blocked, and an on-site one would pass. +- Question: Which words in a posting show that a role is office-only, as opposed to remote? + +**2. `evidence[0].dates` (`ev-harbor-onboarding`)** +- Quote: `dates: {start: 2018-01, end: 2018-06}`. The linked employer `harbor` runs `start: 2021-03` to `end: 2023-06`. +- Why it matters: the dates fall three years before you worked there, so either the dates or the employer link is wrong. This is the only evidence for `user_guides` (scored 4), so a recency or proof check on that skill would rest on a wrong date. +- Question: When did you write the onboarding guide, and was it at Harbor? + +**3. `skills[1].last` (`user_guides`)** +- Quote: `last: 2y`. The evidence ends 2023-06 at the latest, or 2018 as written, and today is 2026-10. +- Why it matters: "2y" is shorter than either date supports, so the skill looks fresher than the file shows. Recency scoring would favor it. +- Question: When did you last write user guides or onboarding material? + +**4. `evidence[1].authorship` (`ev-tidewater-migration`)** +- Quote: `claim: Wrote every page of the help center again...` next to `authorship: REVIEWED`. +- Why it matters: the claim says you wrote it and the tag says you reviewed it. This is the only evidence for `docs_as_code` (self_score 4), and it is marked `proof: checked`, so a reader would trust the stronger claim. +- Question: Did you write the pages, review someone else's rewrite, or both, and which pages were yours? + +**5. `writing_samples[0]`** +- Quote: `title: Help center home page`, with `evidence_id: ev-tidewater-migration`. +- Why it matters: the sample is one page, but the evidence claims "every page". Because the evidence is tagged `REVIEWED`, a hiring manager opening the link cannot tell whether the sample is your writing. +- Question: Is the home page text yours, and is there a page you wrote that shows the work better? + +**6. `skills[0].last` (`docs_as_code`)** +- Quote: `last: 2y`. `employers[1].summary` says "Runs the help center and its build" and `end: present`. +- Why it matters: the employer summary describes current, ongoing use of docs-as-code, but the skill says two years ago. That understates the skill and could cost it points on recency. +- Question: Are you still working in the docs-as-code build today, and if so, is `2y` meant to be something else? + +**7. `person.years_experience`** +- Quote: `years_experience: 14`. The `employers` list covers 2021-03 to present, about five and a half years. +- Why it matters: nothing in the file backs the other eight or so years. A posting with a years-required threshold would score on a number the file can't support. +- Question: Where did you work before 2021, and should those jobs be in `employers`? + +**8. API work: `person.target_roles`, `lanes[0]`, `skills`** +- Quote: `target_roles: [API Technical Writer]` and the lane `api-docs`. No skill or evidence item mentions an API. The only trace is `employers[0].summary`: "Wrote the admin and API guides". +- Why it matters: the whole target rests on API documentation, yet no skill or evidence covers it. The lane's strong requirement `api_owner` ("Owns the API docs") can't be matched to anything you have shown. +- Question: What API reference or guide work have you done, and where is it shown? + +**9. `skills[2]` (`glossaries`)** +- Quote: `self_score: 3`, backed only by `ev-tidewater-glossary` with `proof: unchecked` and `type: interview`, and dates of `2024-02` to `2024-03`. +- Why it matters: the score is the same as a verified skill's, but it rests on one two-month item from an interview. The validator only warns when a skill has no evidence at all, so this passes. +- Question: Is there a document, a link or a person who can confirm the glossary and its term rule? + +**10. Culture and keywords** +- `culture.hustle_phrases[1]` and `culture.free_phrases[0]`: `fast-paced` appears in both lists. A posting with that phrase would be flagged as a hustle sign and as a free-time sign at once. Question: Is "fast-paced" a warning or acceptable to you? +- `lanes[0].keyword_signals`: `positive`, `negative` and `strong_negative` are all `[]`, so nothing can lower a posting's score by keyword. The one `strong_positive` phrase (`docs as code`) is literal, while the skill's `match` also accepts `docs-as-code`. Question: Which words in a posting should count against it, and should the hyphenated form count as a positive too? +- `hard_blocks[1].match_hints`: `crypto` would also match "cryptography", which could block security-tool API roles you might want. Question: Do you want postings about encryption or cryptography blocked? +- `culture.perks[0]` (`unlimited_pto`, `kind: big`) next to `low_time_off_days: 15`: a posting that offers unlimited PTO gets a big perk with no days stated, while you track a 15-day floor. Question: Should unlimited PTO count as a big perk, or only when the posting states a number of days? diff --git a/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md index e0961a1..a5958e7 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md @@ -1,11 +1,9 @@ # Prompt run: Fix errors, on haiku -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). -- Turns: 1. Cost: $0.03 (the simulated user's calls are not included). +- Turns: 1. Cost: $0.02 (the simulated user's calls are not included). - Grade: see `GRADES.md`. ## Conversation @@ -18,10 +16,10 @@ I'm attaching the output of a failed command from this tool (the validator, the ### Model -# Problem 1: evidence[2].authorship -**Cause:** The authorship value "AUTHORED" is not in the allowed list. Based on the claim "Rewrote the release notes template," WROTE is the appropriate value. +## Problem 1: evidence[2].authorship +**Cause:** The value `AUTHORED` is not in the allowed list for the authorship field. -**Location:** `evidence[2].authorship` +**Path:** `evidence[2].authorship` **Fix:** ```yaml @@ -34,28 +32,28 @@ authorship: WROTE --- -# Problem 2: skills[1].evidence_ids -**Cause:** The skill references `ev-placeholder-style-guide`, but no evidence entry with that id exists in the file. The evidence section only contains `ev-northwind-api-rebuild`, `ev-placeholder-docs-ci`, and `ev-placeholder-release-notes`. +## Problem 2: skills[1].evidence_ids +**Cause:** The skill "Style guides" references `ev-placeholder-style-guide`, but that evidence id doesn't exist in the file (only three evidence items are defined). -**Location:** `skills[1].evidence_ids` +**Path:** `skills[1].evidence_ids` -**Fix:** You must either create a matching evidence entry or change the reference to an existing evidence id. If this skill should reference an existing entry, replace the reference: +**Fix:** Either remove the reference or create the missing evidence. If removing: ```yaml # Before evidence_ids: [ev-placeholder-style-guide] # After -evidence_ids: [] # e.g., ev-placeholder-docs-ci +evidence_ids: [] ``` --- -# Problem 3: comp -**Cause:** The compensation values are not in ascending order—min (98000) exceeds open_ask (85000). The correct order must be: floor ≤ min ≤ open_ask ≤ target ≤ stretch_ceiling. +## Problem 3: comp pay order +**Cause:** The `min` value (98000) exceeds `open_ask` (85000), violating the required ordering: floor ≤ min ≤ open_ask ≤ target ≤ stretch_ceiling. -**Location:** `comp.min` and `comp.open_ask` +**Path:** `comp.min` and `comp.open_ask` -**Fix:** Reorder the pay values: +**Fix:** You need to decide the correct values. Choose one: ```yaml # Before floor: 75000 @@ -64,7 +62,7 @@ open_ask: 85000 target: 100000 stretch_ceiling: 175000 -# After +# After (choose your value) floor: 75000 min: open_ask: @@ -74,44 +72,39 @@ stretch_ceiling: 175000 --- -# Problem 4: evidence[1].proof -**Cause:** Interview sources cannot have `proof: checked` because interviews cannot be independently verified like documents or links. Either change the proof to `unchecked` or change the source type to document/link. +## Problem 4: evidence[1].proof +**Cause:** Interview sources cannot be marked as `checked` proof since there's no document to verify; they must use `unchecked`. -**Location:** `evidence[1].proof` and `evidence[1].source.type` +**Path:** `evidence[1].proof` **Fix:** ```yaml # Before proof: checked -source: {type: interview, ref: "interview:2026-10-01:s2.q3", captured_on: 2026-10-01} -# After (option 1: mark proof as unchecked) +# After proof: unchecked -source: {type: interview, ref: "interview:2026-10-01:s2.q3", captured_on: 2026-10-01} - -# After (option 2: change source to document/link and keep checked) -proof: checked -source: {type: document, ref: "", captured_on: 2026-10-01} ``` --- -# Problem 5: skills[2] -**Cause:** A self_score of 5 requires recent use, but `last: never` indicates no recent use. Either lower the score or update the last field to a recent timeframe. +## Problem 5: skills[2] score/last mismatch +**Cause:** A `self_score` of 5 requires recent real-world use, but `last: never` indicates no recent experience. -**Location:** `skills[2]` (graphql skill) +**Path:** `skills[2]` -**Fix:** +**Fix:** Either lower the score or update the last-used date: ```yaml # Before self_score: 5 last: never -# After (option 1: lower the score) -self_score: # 3 or lower +# After (choose one) +# Option A: lower the score +self_score: 3 last: never -# After (option 2: update last to recent use) +# Option B: add recent use self_score: 5 -last: # e.g., 1y, 6m, 1m +last: # e.g., 1y, 2y, 6m ``` diff --git a/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md index 2e7d2ff..e8f3293 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md @@ -1,8 +1,6 @@ # Prompt run: Fix errors, on sonnet -Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. - - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -18,68 +16,63 @@ I'm attaching the output of a failed command from this tool (the validator, the ### Model -All five problems, in the order the validator printed them. - -## 1. `evidence[2].authorship` +All five problems in `fixtures/broken-profile.yaml` are covered below. No line numbers are given, only YAML paths. -**Cause:** `AUTHORED` is not one of the allowed values. The output lists them: WROTE, DIRECTED, CO-WROTE, DESIGNED, REVIEWED, OTHER-AUTHOR. +## 1. `evidence[2].authorship`: "AUTHORED" is not allowed -**Path:** `evidence[2].authorship`, the entry with id `ev-placeholder-release-notes`. +**Cause:** The evidence entry `ev-placeholder-release-notes` uses `AUTHORED`, which is not in the allowed list (WROTE, DIRECTED, CO-WROTE, DESIGNED, REVIEWED, OTHER-AUTHOR). -Which value is right depends on what you actually did, so that's your call. The claim says "Rewrote the release notes template", which points to WROTE, but you know the situation better. +**Which value is right is your call.** The claim says you "rewrote" the template, so `WROTE` fits, but `CO-WROTE` applies if someone else shared the work. ```yaml # before authorship: AUTHORED # after - authorship: + authorship: ``` -## 2. `skills[1].evidence_ids` +## 2. `skills[1].evidence_ids`: unknown evidence id -**Cause:** The skill `style_guides` points to `ev-placeholder-style-guide`, and no evidence entry in the file has that id. The missing evidence id is **`ev-placeholder-style-guide`**. +**Cause:** The Style guides skill points to `ev-placeholder-style-guide`, and no evidence entry has that id. The missing id is **`ev-placeholder-style-guide`**. The file has only three ids: `ev-northwind-api-rebuild`, `ev-placeholder-docs-ci` and `ev-placeholder-release-notes`. -**Path:** `skills[1].evidence_ids`. - -Your options: -- Point it at an evidence id that already exists: `ev-northwind-api-rebuild`, `ev-placeholder-docs-ci` or `ev-placeholder-release-notes`. The release notes entry is the closest topical match, but you should confirm. -- Add a new evidence entry with the id `ev-placeholder-style-guide`. -- Use an empty list if there's no evidence for this skill. +**Your call, with two choices:** +- Point the skill at an existing entry. The release-notes entry is the closest topical match, but only you know whether it backs this skill. +- Add a new evidence entry with the id `ev-placeholder-style-guide`. That needs a claim, dates and a source from you, which I can't supply. ```yaml -# before +# before (path: skills[1].evidence_ids) evidence_ids: [ev-placeholder-style-guide] # after - evidence_ids: [] + evidence_ids: [] ``` -## 3. `comp` +## 3. `comp`: pay numbers out of order -**Cause:** The pay numbers must satisfy `floor <= min <= open_ask <= target <= stretch_ceiling`. Here `min` (98000) is above `open_ask` (85000). +**Cause:** The required order is `floor <= min <= open_ask <= target <= stretch_ceiling`, but `min` (98000) is above `open_ask` (85000). The other values (floor 75000, target 100000, stretch_ceiling 175000) already fit the order. -**Path:** `comp.min` and `comp.open_ask`. - -Which number is wrong is your call. Either lower `min` to 85000 or less (and keep it at or above the floor of 75000), or raise `open_ask` to at least 98000 (and keep it at or below the target of 100000). +**Your call, with two choices:** +- Lower `min` to somewhere between 75000 and 85000. +- Raise `open_ask` to somewhere between 98000 and 100000. ```yaml # before min: 98000 open_ask: 85000 -# after (fix one of the two) - min: +# after (change one of the two) + min: open_ask: 85000 # or min: 98000 - open_ask: + open_ask: ``` -## 4. `evidence[1].proof` - -**Cause:** The source type is `interview`, which can't carry `proof: checked` because nothing was read to back it. The validator offers two ways out: use `unchecked`, or point `source` at a document, link or artifact. +## 4. `evidence[1].proof`: interview source cannot be "checked" -**Path:** `evidence[1].proof`, the entry with id `ev-placeholder-docs-ci`. +**Cause:** The entry `ev-placeholder-docs-ci` has `source.type: interview`, and nothing was read to back it, so `proof: checked` is not allowed. The output gives two fixes. -The smallest change is the first option: +**Your call, with two choices:** +- Set `proof` to `unchecked`. This is the smaller change and matches the facts as they stand. +- Keep `checked` and point `source` at a document, link or artifact you can actually supply. ```yaml # before @@ -88,28 +81,26 @@ The smallest change is the first option: proof: unchecked ``` -If you do have a document, link or artifact for this claim, change `source.type` and `source.ref` to it instead and keep `checked`. I won't make up that reference. - -## 5. `skills[2]` - -**Cause:** `self_score: 5` with `last: never` is a contradiction, because a score of 4 or 5 needs some real use. +## 5. `skills[2]`: `self_score` 5 with `last: never` -**Path:** `skills[2]`, the skill `graphql`. +**Cause:** The GraphQL skill claims a score of 5 but says it was never used, and a score of 4 or 5 needs some real use. -Either lower the score or correct `last`. Only you know which. The output doesn't show the allowed values for either field, so they are listed in `schema/career-profile.schema.json`. +**Your call, with two choices:** +- Lower `self_score`. The output says only that it must be below 4, so I can't pick the number for you. +- Fix `last` if you have used GraphQL. The allowed values for `last` aren't shown in the output; they are listed in `schema/career-profile.schema.json`. ```yaml # before self_score: 5 last: never -# after (fix one of the two) - self_score: +# after (change one of the two) + self_score: last: never # or self_score: 5 - last: + last: ``` -I haven't edited the file. After you pick the values, rerun `python3 scripts/validate_profile.py fixtures/broken-profile.yaml`. +Nothing else in the file needs to change. After you pick the values, rerun `python3 scripts/validate_profile.py fixtures/broken-profile.yaml` to confirm the result. -One note: this file is a deliberate fixture with five planted mistakes, and `tests/test_validate_profile.py` expects the validator to fail on exactly these. If you fix them in place, that test will break. Make the changes in a copy, or only if you meant to repair the fixture. +One note: this file is a test fixture. Its header says it must keep exactly five planted mistakes so the validator exits 1, and `tests/test_validate_profile.py` lists them. If you're fixing it rather than just reading the output, check whether the tests expect it to stay broken. diff --git a/pipelines/job-assessment/tests/prompt-runs/run_prompts.py b/pipelines/job-assessment/tests/prompt-runs/run_prompts.py index b633a86..b18c568 100644 --- a/pipelines/job-assessment/tests/prompt-runs/run_prompts.py +++ b/pipelines/job-assessment/tests/prompt-runs/run_prompts.py @@ -70,6 +70,11 @@ def validator_output(path): "attach": [("ARCHITECTURE.md, section 'What the validator checks'", lambda: section("ARCHITECTURE.md", "What the validator checks")), "fixtures/gappy-profile.yaml"]}, + # The held-out run: the same prompt on a gappy file it was never tuned against. + "p5h": {"file": "05-find-gaps.txt", "title": "Find gaps in my central file (held-out profile)", + "attach": [("ARCHITECTURE.md, section 'What the validator checks'", + lambda: section("ARCHITECTURE.md", "What the validator checks")), + "fixtures/held-out-gappy-profile.yaml"]}, "p6": {"file": "06-fix-errors.txt", "title": "Fix errors", "attach": [("output of the failed command", lambda: validator_output("fixtures/broken-profile.yaml")), "fixtures/broken-profile.yaml"]}, diff --git a/pipelines/job-assessment/tests/test_validate_profile.py b/pipelines/job-assessment/tests/test_validate_profile.py index efa73f9..49bc2fc 100644 --- a/pipelines/job-assessment/tests/test_validate_profile.py +++ b/pipelines/job-assessment/tests/test_validate_profile.py @@ -56,6 +56,23 @@ "hard_blocks[1] (weapons) has no why.", } +# The planted judgment gaps in fixtures/held-out-gappy-profile.yaml, the file the +# "find gaps" prompt was NOT tuned against. Nothing in the prompt describes them. +HELD_OUT_PLANTED_GAPS = { + "years_vs_employers": + "person.years_experience is 14, but the employers listed start in 2021-03.", + "evidence_before_employer": + "evidence[0] is dated 2018 at an employer whose tenure starts in 2021-03.", + "claim_vs_authorship": + "evidence[1] claims to have written every page, but its authorship is REVIEWED.", + "hard_block_matches_the_wanted_jobs": + "hard_blocks[0] (non_remote) has match hints that would fire on the remote jobs the person wants.", + "phrase_in_both_lists": + "culture lists 'fast-paced' as both a hustle phrase and a free phrase.", + "recent_use_with_old_evidence": + "skills[1] (user_guides) says used in the last two years, but its only evidence ends in 2018.", +} + def run_cli(*args): proc = subprocess.run([sys.executable, str(SCRIPTS / "validate_profile.py"), *map(str, args)], @@ -123,6 +140,46 @@ def test_gappy_fixture_really_holds_the_planted_gaps(): assert len(GAPPY_PLANTED_GAPS) == 3 +def test_held_out_fixture_passes_clean_with_no_warnings(): + code, out, err = run_cli("--fixture", FIXTURES / "held-out-gappy-profile.yaml") + assert code == 0 + assert out == "" + assert "clean, 0 warning(s)" in err + + +def test_held_out_fixture_really_holds_the_planted_gaps(): + data = load(FIXTURES / "held-out-gappy-profile.yaml") + starts = [str(e["start"]) for e in data["employers"]] + assert data["person"]["years_experience"] == 14 and min(starts) == "2021-03" + + old = data["evidence"][0] + employer = next(e for e in data["employers"] if e["id"] == old["employer_id"]) + assert str(old["dates"]["start"]) < str(employer["start"]) + + claim = data["evidence"][1] + assert claim["claim"].startswith("Wrote every page") and claim["authorship"] == "REVIEWED" + + block = data["hard_blocks"][0] + assert block["id"] == "non_remote" and "remote" in block["match_hints"] + + culture = data["culture"] + assert set(culture["hustle_phrases"]) & set(culture["free_phrases"]) == {"fast-paced"} + + skill = data["skills"][1] + assert skill["id"] == "user_guides" and skill["last"] == "2y" + ends = [str(e["dates"]["end"]) for e in data["evidence"] if e["id"] in skill["evidence_ids"]] + assert ends and all(end < "2020" for end in ends) + + assert len(HELD_OUT_PLANTED_GAPS) == 6 + + +def test_the_find_gaps_prompt_does_not_describe_the_held_out_gaps(): + prompt = (ROOT / "prompts" / "05-find-gaps.txt").read_text(encoding="utf-8").lower() + for giveaway in ("years_experience", "authorship", "match_hints", "hustle", "free_phrases", + "copied", "thin evidence", "no reason", "no dates"): + assert giveaway not in prompt + + @pytest.mark.skipif(not ROBIN.exists(), reason="Robin's fixture arrives in a separate package") def test_robin_sample_passes_with_fixture_flag(): code, out, _ = run_cli("--fixture", ROBIN)