From 867e55b533d4b873cf52380ed6874305a1b15917 Mon Sep 17 00:00:00 2001 From: darthrootbeer Date: Thu, 1 Oct 2026 22:01:19 -0400 Subject: [PATCH] fix(job-assessment): use unrelated fictional pay numbers and report bad findings cleanly Robin Sample's pay figures and the three posting ranges now use different round fictional values with the same ordering and bands, so every score and verdict is unchanged. check_findings.py now reports list items or sections of the wrong shape as normal problem lines instead of a traceback. Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01XV3Ggh86cf2xwUyz28XaN6 --- pipelines/job-assessment/assessment/SKILL.md | 4 +- .../assessment/scripts/check_findings.py | 65 +++++++++++++++---- .../fixtures/broken-profile.yaml | 10 +-- .../findings/01-strong-fit.findings.json | 14 ++-- .../02-job-type-override.findings.json | 14 ++-- .../fixtures/gappy-profile.yaml | 10 +-- .../fixtures/golden/01-strong-fit/email.html | 2 +- .../fixtures/golden/01-strong-fit/note.md | 2 +- .../golden/02-job-type-override/email.html | 2 +- .../golden/02-job-type-override/note.md | 8 +-- .../fixtures/postings/01-strong-fit.md | 2 +- .../fixtures/postings/02-job-type-override.md | 4 +- .../fixtures/robin-sample/career-profile.yaml | 10 +-- .../fixtures/robin-sample/intake-answers.txt | 10 +-- .../job-assessment/intake/stages/04-needs.md | 2 +- .../job-assessment/tests/e2e-output-run1.md | 6 +- pipelines/job-assessment/tests/e2e-output.md | 4 +- .../job-assessment/tests/intake-transcript.md | 12 ++-- .../earlier-versions/p3-haiku-v1.md | 16 +++-- .../earlier-versions/p3-haiku-v2.md | 16 +++-- .../earlier-versions/p3-haiku-v3.md | 14 ++-- .../earlier-versions/p3-sonnet-v1.md | 12 ++-- .../earlier-versions/p3-sonnet-v2.md | 10 +-- .../earlier-versions/p3-sonnet-v3.md | 6 +- .../earlier-versions/p4-haiku-v1.md | 6 +- .../earlier-versions/p4-sonnet-v1.md | 6 +- .../earlier-versions/p6-haiku-v1.md | 12 ++-- .../earlier-versions/p6-haiku-v2.md | 8 ++- .../earlier-versions/p6-haiku-v3.md | 6 +- .../earlier-versions/p6-sonnet-v1.md | 20 +++--- .../earlier-versions/p6-sonnet-v2.md | 20 +++--- .../earlier-versions/p6-sonnet-v3.md | 18 ++--- .../earlier-versions/r2-haiku-v1.md | 12 ++-- .../tests/prompt-runs/p3-haiku.md | 14 ++-- .../tests/prompt-runs/p3-sonnet.md | 14 ++-- .../tests/prompt-runs/p4-haiku.md | 6 +- .../tests/prompt-runs/p4-sonnet.md | 8 ++- .../tests/prompt-runs/p6-haiku.md | 20 +++--- .../tests/prompt-runs/p6-sonnet.md | 18 ++--- .../tests/test_check_findings.py | 50 ++++++++++++++ .../tests/test_load_criteria.py | 4 +- .../tests/test_score_helpers.py | 46 ++++++------- .../tests/test_validate_profile.py | 8 +-- 43 files changed, 344 insertions(+), 207 deletions(-) diff --git a/pipelines/job-assessment/assessment/SKILL.md b/pipelines/job-assessment/assessment/SKILL.md index ccf0b3e..68fc53d 100644 --- a/pipelines/job-assessment/assessment/SKILL.md +++ b/pipelines/job-assessment/assessment/SKILL.md @@ -296,8 +296,8 @@ every top-level key present, nothing extra. "autonomy": {"net": "positive", "quotes": ["You own the docs platform end to end"], "read": "You would decide how the platform works."}, "pay": {"stated": true, "tier_used": null, "top": null, "other_pay_noted": [], - "tiers": [{"label": "all other US locations", "min": 120000, "max": 150000, - "quote": "Base pay range: $120,000 - $150,000 (all other US locations)"}]}, + "tiers": [{"label": "all other US locations", "min": 98000, "max": 104000, + "quote": "Base pay range: $98,000 - $104,000 (all other US locations)"}]}, "culture": {"perks": [{"perk_id": "learning_budget", "quote": "Annual learning budget"}], "strong_positive_phrases": [], "low_time_off": null, "hustle": [], "soft_flags_tripped": [], "negative_phrases": []}, diff --git a/pipelines/job-assessment/assessment/scripts/check_findings.py b/pipelines/job-assessment/assessment/scripts/check_findings.py index 69c8b58..2aa2762 100644 --- a/pipelines/job-assessment/assessment/scripts/check_findings.py +++ b/pipelines/job-assessment/assessment/scripts/check_findings.py @@ -111,6 +111,30 @@ def quote_ok(path, quote): return False return True + def objects(value, path): + """The object entries of a list. A non-list or a non-object entry is a problem.""" + if value is None: + return [] + if not isinstance(value, list): + add(path, "must be a list.") + return [] + kept = [] + for i, entry in enumerate(value): + if isinstance(entry, dict): + kept.append((i, entry)) + else: + add(f"{path}[{i}]", "must be an object, not a plain value such as a string.") + return kept + + def section(value, path): + """A findings section that must be an object. Anything else counts as empty.""" + if value in (None, {}, [], ""): + return {} + if not isinstance(value, dict): + add(path, "must be an object.") + return {} + return value + if not isinstance(findings, dict): return ["findings: the file is not a JSON object."] @@ -131,7 +155,7 @@ def quote_ok(path, quote): requirement_ids = {r.get("id") for r in lane.get("requirements") or []} # Hard block. - block = findings.get("hard_block") or {} + block = section(findings.get("hard_block"), "hard_block") if block.get("tripped"): if block.get("id") not in blocks: add("hard_block.id", f'"{block.get("id")}" is not a hard block in the profile.') @@ -141,18 +165,21 @@ def quote_ok(path, quote): f'no named exception for "{block.get("id")}" exists in the profile.') # Job-type override. - override = findings.get("job_type_override") or {} + override = section(findings.get("job_type_override"), "job_type_override") if override.get("fired"): if not override.get("skill"): add("job_type_override.skill", "fired but names no skill.") quotes = override.get("quotes") or [] - if not quotes: + if not isinstance(quotes, list): + add("job_type_override.quotes", "must be a list.") + quotes = [] + elif not quotes: add("job_type_override.quotes", "fired but quotes nothing from the posting.") for i, q in enumerate(quotes): quote_ok(f"job_type_override.quotes[{i}]", q) # Requirement ratings. Silence is Unknown. Anything else needs a quote. - for i, item in enumerate(findings.get("requirements") or []): + for i, item in objects(findings.get("requirements"), "requirements"): path = f"requirements[{i}]" rating = str(item.get("rating", "")).lower() if rating not in RATINGS: @@ -170,10 +197,13 @@ def quote_ok(path, quote): "a read on autonomy rests only on generic phrases. The posting is silent, so rate it unknown.") # Autonomy read. - autonomy = findings.get("autonomy") or {} + autonomy = section(findings.get("autonomy"), "autonomy") if autonomy.get("net") in ("negative", "mixed"): quotes = autonomy.get("quotes") or [] - if not quotes: + if not isinstance(quotes, list): + add("autonomy.quotes", "must be a list.") + quotes = [] + elif not quotes: add("autonomy.quotes", "a read on autonomy quotes nothing from the posting.") found = [q for i, q in enumerate(quotes) if quote_ok(f"autonomy.quotes[{i}]", q)] if found and all(_generic_only(q) for q in found): @@ -181,13 +211,13 @@ def quote_ok(path, quote): "the autonomy read rests only on generic phrases. The posting is silent, so it is unknown, not negative.") # Culture: every change needs a quote. - culture = findings.get("culture") or {} - for i, perk in enumerate(culture.get("perks") or []): + culture = section(findings.get("culture"), "culture") + for i, perk in objects(culture.get("perks"), "culture.perks"): if perk.get("perk_id") not in perks: add(f"culture.perks[{i}].perk_id", f'"{perk.get("perk_id")}" is not a perk in the profile.') quote_ok(f"culture.perks[{i}].quote", perk.get("quote")) for key in ("strong_positive_phrases", "hustle", "negative_phrases", "soft_flags_tripped"): - for i, entry in enumerate(culture.get(key) or []): + for i, entry in objects(culture.get(key), f"culture.{key}"): quote_ok(f"culture.{key}[{i}].quote", entry.get("quote")) low = culture.get("low_time_off") if isinstance(low, dict): @@ -196,11 +226,14 @@ def quote_ok(path, quote): add("culture.low_time_off", "has no quote from the posting.") # Qualifications: every claim points at real, usable evidence. - qual = findings.get("qualifications") or {} - for i, match in enumerate(qual.get("matches") or []): + qual = section(findings.get("qualifications"), "qualifications") + for i, match in objects(qual.get("matches"), "qualifications.matches"): path = f"qualifications.matches[{i}]" quote_ok(f"{path}.requirement_quote", match.get("requirement_quote")) ids = match.get("evidence_ids") or [] + if not isinstance(ids, list): + add(f"{path}.evidence_ids", "must be a list of evidence ids.") + ids = [] if not ids: add(f"{path}.evidence_ids", "a match with no evidence behind it. Write it as unproven instead.") for ev_id in ids: @@ -211,7 +244,7 @@ def quote_ok(path, quote): add(f"{path}.evidence_ids", f'"{ev_id}" is marked do_not_use and cannot back a claim.') elif entry.get("authorship") == "OTHER-AUTHOR": add(f"{path}.evidence_ids", f'"{ev_id}" is work someone else wrote and cannot back a claim.') - for i, item in enumerate(qual.get("unproven") or []): + for i, item in objects(qual.get("unproven"), "qualifications.unproven"): path = f"qualifications.unproven[{i}]" if item.get("skill_id") not in skills: add(f"{path}.skill_id", f'"{item.get("skill_id")}" is not a skill in the profile.') @@ -245,7 +278,13 @@ def main(argv=None): print(f"check_findings: could not read inputs: {exc}", file=sys.stderr) return 2 meta, text = split_posting(raw) - problems = check(findings, text, meta.get("lane"), profile) + try: + problems = check(findings, text, meta.get("lane"), profile) + except (AttributeError, TypeError, KeyError) as exc: + # A shape this script does not know how to name. Still a bad findings + # file, so say so in one line and stop with the "problems found" code. + print(f"check_findings: the findings file has a shape this check cannot read ({exc}).") + return 1 for line in problems: print(line) if problems: diff --git a/pipelines/job-assessment/fixtures/broken-profile.yaml b/pipelines/job-assessment/fixtures/broken-profile.yaml index 1ce1b3d..f79f3aa 100644 --- a/pipelines/job-assessment/fixtures/broken-profile.yaml +++ b/pipelines/job-assessment/fixtures/broken-profile.yaml @@ -18,11 +18,11 @@ hard_blocks: match_hints: [sports betting, casino] comp: currency: USD - floor: 90000 - min: 130000 - open_ask: 125000 - target: 140000 - stretch_ceiling: 170000 + floor: 75000 + min: 98000 + open_ask: 85000 + target: 100000 + stretch_ceiling: 175000 culture: perks: - {id: unlimited_pto, label: Unlimited PTO, kind: big} diff --git a/pipelines/job-assessment/fixtures/findings/01-strong-fit.findings.json b/pipelines/job-assessment/fixtures/findings/01-strong-fit.findings.json index 2f7945b..fc543f7 100644 --- a/pipelines/job-assessment/fixtures/findings/01-strong-fit.findings.json +++ b/pipelines/job-assessment/fixtures/findings/01-strong-fit.findings.json @@ -52,19 +52,19 @@ "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ], "tier_used": "Region B", - "top": 150000, + "top": 104000, "other_pay_noted": [] }, "culture": { diff --git a/pipelines/job-assessment/fixtures/findings/02-job-type-override.findings.json b/pipelines/job-assessment/fixtures/findings/02-job-type-override.findings.json index 6f4df1e..bc47016 100644 --- a/pipelines/job-assessment/fixtures/findings/02-job-type-override.findings.json +++ b/pipelines/job-assessment/fixtures/findings/02-job-type-override.findings.json @@ -47,19 +47,19 @@ "tiers": [ { "label": "Region A", - "min": 105000, - "max": 120000, - "quote": "Region A: $105,000 to $120,000." + "min": 82000, + "max": 98000, + "quote": "Region A: $82,000 to $98,000." }, { "label": "Everywhere else", - "min": 85000, - "max": 100000, - "quote": "Everywhere else: $85,000 to $100,000." + "min": 72000, + "max": 78000, + "quote": "Everywhere else: $72,000 to $78,000." } ], "tier_used": "Everywhere else", - "top": 100000, + "top": 78000, "other_pay_noted": [] }, "culture": { diff --git a/pipelines/job-assessment/fixtures/gappy-profile.yaml b/pipelines/job-assessment/fixtures/gappy-profile.yaml index 48a8780..137681c 100644 --- a/pipelines/job-assessment/fixtures/gappy-profile.yaml +++ b/pipelines/job-assessment/fixtures/gappy-profile.yaml @@ -34,11 +34,11 @@ named_exceptions: added: 2026-10-01 comp: currency: USD - floor: 90000 - min: 110000 - open_ask: 125000 - target: 140000 - stretch_ceiling: 170000 + floor: 75000 + min: 80000 + open_ask: 85000 + target: 100000 + stretch_ceiling: 175000 culture: perks: - {id: unlimited_pto, label: Unlimited PTO, kind: big} diff --git a/pipelines/job-assessment/fixtures/golden/01-strong-fit/email.html b/pipelines/job-assessment/fixtures/golden/01-strong-fit/email.html index d60cc9e..1b38189 100644 --- a/pipelines/job-assessment/fixtures/golden/01-strong-fit/email.html +++ b/pipelines/job-assessment/fixtures/golden/01-strong-fit/email.html @@ -128,7 +128,7 @@
● Unknown
-
The posting lists base pay up to 150000.
+
The posting lists base pay up to 104000.
diff --git a/pipelines/job-assessment/fixtures/golden/01-strong-fit/note.md b/pipelines/job-assessment/fixtures/golden/01-strong-fit/note.md index 5bbbe15..27f4610 100644 --- a/pipelines/job-assessment/fixtures/golden/01-strong-fit/note.md +++ b/pipelines/job-assessment/fixtures/golden/01-strong-fit/note.md @@ -148,7 +148,7 @@ Copperline makes writing tools for software teams. We are a remote company with ## Pay and benefits -- Base pay range for Region B: $125,000 to $150,000. Region A: $140,000 to $170,000. +- Base pay range for Region B: $100,000 to $104,000. Region A: $102,000 to $118,000. - Unlimited paid time off, with a three-week minimum we ask everyone to take. - A $3,000 yearly learning budget. - Two team meetups a year. diff --git a/pipelines/job-assessment/fixtures/golden/02-job-type-override/email.html b/pipelines/job-assessment/fixtures/golden/02-job-type-override/email.html index 5ed7c9c..ec90019 100644 --- a/pipelines/job-assessment/fixtures/golden/02-job-type-override/email.html +++ b/pipelines/job-assessment/fixtures/golden/02-job-type-override/email.html @@ -119,7 +119,7 @@
● Unknown
-
The posting lists base pay up to 100000.
+
The posting lists base pay up to 78000.
diff --git a/pipelines/job-assessment/fixtures/golden/02-job-type-override/note.md b/pipelines/job-assessment/fixtures/golden/02-job-type-override/note.md index 4040d5c..b69c835 100644 --- a/pipelines/job-assessment/fixtures/golden/02-job-type-override/note.md +++ b/pipelines/job-assessment/fixtures/golden/02-job-type-override/note.md @@ -101,8 +101,8 @@ The split follows Modestino, Shoag and Ballance (*Review of Economics and Statis | 12+ years of technical writing experience. | experience | likely filter | | Fluent in DITA XML and a component content management system. | skill | likely real | | Experience documenting iOS and Android SDKs. | skill | likely real | -| Region A: $105,000 to $120,000. | skill | likely real | -| Everywhere else: $85,000 to $100,000. | skill | likely real | +| Region A: $82,000 to $98,000. | skill | likely real | +| Everywhere else: $72,000 to $78,000. | skill | likely real | --- @@ -142,6 +142,6 @@ Driftmark sells training courses for a software product. We are a friendly team ## Pay -- Region A: $105,000 to $120,000. -- Everywhere else: $85,000 to $100,000. +- Region A: $82,000 to $98,000. +- Everywhere else: $72,000 to $78,000. diff --git a/pipelines/job-assessment/fixtures/postings/01-strong-fit.md b/pipelines/job-assessment/fixtures/postings/01-strong-fit.md index 516a3c6..13aabf9 100644 --- a/pipelines/job-assessment/fixtures/postings/01-strong-fit.md +++ b/pipelines/job-assessment/fixtures/postings/01-strong-fit.md @@ -34,7 +34,7 @@ Copperline makes writing tools for software teams. We are a remote company with ## Pay and benefits -- Base pay range for Region B: $125,000 to $150,000. Region A: $140,000 to $170,000. +- Base pay range for Region B: $100,000 to $104,000. Region A: $102,000 to $118,000. - Unlimited paid time off, with a three-week minimum we ask everyone to take. - A $3,000 yearly learning budget. - Two team meetups a year. diff --git a/pipelines/job-assessment/fixtures/postings/02-job-type-override.md b/pipelines/job-assessment/fixtures/postings/02-job-type-override.md index c34b3a2..87e2258 100644 --- a/pipelines/job-assessment/fixtures/postings/02-job-type-override.md +++ b/pipelines/job-assessment/fixtures/postings/02-job-type-override.md @@ -32,5 +32,5 @@ Driftmark sells training courses for a software product. We are a friendly team ## Pay -- Region A: $105,000 to $120,000. -- Everywhere else: $85,000 to $100,000. +- Region A: $82,000 to $98,000. +- Everywhere else: $72,000 to $78,000. diff --git a/pipelines/job-assessment/fixtures/robin-sample/career-profile.yaml b/pipelines/job-assessment/fixtures/robin-sample/career-profile.yaml index 9b97a94..93c010f 100644 --- a/pipelines/job-assessment/fixtures/robin-sample/career-profile.yaml +++ b/pipelines/job-assessment/fixtures/robin-sample/career-profile.yaml @@ -18,11 +18,11 @@ named_exceptions: - {company: Lanternfield Example Co., block_id: non_remote, why: "Robin lives a ten minute walk from this office and would go in.", added: "2026-10-01"} comp: currency: USD - floor: 90000 - min: 110000 - open_ask: 125000 - target: 140000 - stretch_ceiling: 170000 + floor: 75000 + min: 80000 + open_ask: 85000 + target: 100000 + stretch_ceiling: 175000 culture: perks: - {id: unlimited_pto, label: Unlimited PTO, kind: big} diff --git a/pipelines/job-assessment/fixtures/robin-sample/intake-answers.txt b/pipelines/job-assessment/fixtures/robin-sample/intake-answers.txt index 9df2009..662ae1d 100644 --- a/pipelines/job-assessment/fixtures/robin-sample/intake-answers.txt +++ b/pipelines/job-assessment/fixtures/robin-sample/intake-answers.txt @@ -77,11 +77,11 @@ s3.localization_pm: 2, would rather avoid, within 5 years, self, no evidence # Stage 4. Needs s4.q1: Gambling and betting, weapons makers, and any job that is not remote. s4.q2: Yes, Lanternfield Example Co. I live a ten minute walk from its office. -s4.q3: 90000 US dollars. -s4.q4: 110000. -s4.q5: 125000. -s4.q6: 140000. -s4.q7: 170000. +s4.q3: 75000 US dollars. +s4.q4: 80000. +s4.q5: 85000. +s4.q6: 100000. +s4.q7: 175000. s4.q8: Region B. s4.q9: Unlimited time off, a learning budget and extra paid days off matter most. Meetups, a home-office stipend and paid parental leave are nice. s4.q10: Fewer than 15 days a year. diff --git a/pipelines/job-assessment/intake/stages/04-needs.md b/pipelines/job-assessment/intake/stages/04-needs.md index 0a915f5..b346044 100644 --- a/pipelines/job-assessment/intake/stages/04-needs.md +++ b/pipelines/job-assessment/intake/stages/04-needs.md @@ -116,7 +116,7 @@ YAML ```bash python3 scripts/add_entry.py PROFILE --section comp --entry - <<'YAML' currency: USD -floor: 90000 +floor: 75000 YAML ``` diff --git a/pipelines/job-assessment/tests/e2e-output-run1.md b/pipelines/job-assessment/tests/e2e-output-run1.md index 50d7835..60162df 100644 --- a/pipelines/job-assessment/tests/e2e-output-run1.md +++ b/pipelines/job-assessment/tests/e2e-output-run1.md @@ -1,6 +1,8 @@ # Live run, first attempt (posting 03 FAILED, kept as evidence of model variance) +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + The second attempt, which passed, is in `e2e-output.md`. Nothing was changed between the two runs. Date: 2026-10-01. Model: `sonnet`. A model followed `assessment/SKILL.md` and wrote its own findings. The test recomputes scores and the verdict with the scripts and does not trust the model's numbers. Tool output is cut to 500 characters. Folder names are replaced with `` and ``. @@ -228,7 +230,7 @@ override=v.override_skill(f),hard_block=v.hard_block_id(f))))" > work/verdict.js "fit": [], "comp": [ "Tier used: Region B", - "Top of range 150000 against target 140000, min 110000, floor 90000" + "Top of range 104000 against target 100000, min 80000, floor 75000" ], "qualifications": [ "1 load-bearing gap(s), -1 (cap 4)" @@ -312,7 +314,7 @@ Culture 10 🌟🌟🌟🌟🌟🌟🌟🌟🌟🌟 [Open the note](file:///archive/🟒%20πŸ“‹%20πŸ”§%20Copperline%20Example%20Co.%20-%20Docs%20Platform%20Engineer%20-%202026-10-01.md) -The verdict is Apply, and no earlier rule fired: there was no hard block, no job-type override, and no score below the floor or in the reservations band. Because Copperline is a high-interest company, move fast and check for a warm introduction. The one thing to prepare for is the Go line: the posting wants you to read Go code, and you have no recorded evidence of recent Go work. That cost one Qualifications point. The Fit, Comp and Culture bars are all at 10, and Comp is based on the Region B tier ($125k–$150k). +The verdict is Apply, and no earlier rule fired: there was no hard block, no job-type override, and no score below the floor or in the reservations band. Because Copperline is a high-interest company, move fast and check for a warm introduction. The one thing to prepare for is the Go line: the posting wants you to read Go code, and you have no recorded evidence of recent Go work. That cost one Qualifications point. The Fit, Comp and Culture bars are all at 10, and Comp is based on the Region B tier ($85k–$104k). The email card is at `out/copperline-example-co-docs-platform-engineer.html`. It was written and passed its own check, and nothing was sent. diff --git a/pipelines/job-assessment/tests/e2e-output.md b/pipelines/job-assessment/tests/e2e-output.md index 8b8860b..8c82224 100644 --- a/pipelines/job-assessment/tests/e2e-output.md +++ b/pipelines/job-assessment/tests/e2e-output.md @@ -1,6 +1,8 @@ # Live run: assessment on the fictional postings +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + Date: 2026-10-01. Model: `sonnet`. A model followed `assessment/SKILL.md` and wrote its own findings. The test recomputes scores and the verdict with the scripts and does not trust the model's numbers. Tool output is cut to 500 characters. Folder names are replaced with `` and ``. One path to a saved tool-output file, which named a home folder, was replaced by hand with `` after the run. ## Posting 01-strong-fit @@ -242,7 +244,7 @@ check_findings: ok "fit": [], "comp": [ "Tier used: Region B", - "Top of range 150000 against target 140000, min 110000, floor 90000" + "Top of range 104000 against target 100000, min 80000, floor 75000" ], "qualifications": [ "1 load-bearing gap(s), -1 (cap 4)" diff --git a/pipelines/job-assessment/tests/intake-transcript.md b/pipelines/job-assessment/tests/intake-transcript.md index 21235cf..e9cffc1 100644 --- a/pipelines/job-assessment/tests/intake-transcript.md +++ b/pipelines/job-assessment/tests/intake-transcript.md @@ -1,6 +1,8 @@ # Live run: intake transcript +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + Date: 2026-10-01. Model: `sonnet`. Commands run by the model: 32. Turns: 33. Cost: $0.81. A model followed `intake/SKILL.md` and answered from `fixtures/robin-sample/intake-answers.txt`. Tool output is cut to the first 700 characters. Folder names are replaced with `` and ``. @@ -796,11 +798,11 @@ validate_profile: clean, 7 warning(s) in career-profile.yaml $ add(){ python3 scripts/add_entry.py career-profile.yaml "$@" | grep -v "no dates\|perks list"; } add --section comp --entry - <<'YAML' currency: USD -floor: 90000 -min: 110000 -open_ask: 125000 -target: 140000 -stretch_ceiling: 170000 +floor: 75000 +min: 80000 +open_ask: 85000 +target: 100000 +stretch_ceiling: 175000 YAML python3 - <<'E' import yaml;d=yaml.safe_load(open('career-profile.yaml'));print(d['person'],d['culture']) diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v1.md index c9d59f5..d3e259a 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v1.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.12 (the simulated user's calls are not included). @@ -115,15 +117,15 @@ Proceeding with the full read. "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ] }, @@ -192,7 +194,7 @@ Proceeding with the full read. **Fit: 10** β€” All four requirements rated strong or better; autonomy reads positive; no avoidance conflicts. -**Comp: 8** β€” Region B top pay ($150k) falls between your open ask (125k) and target (140k), per the scoring rules. +**Comp: 8** β€” Region B top pay ($104k) falls between your open ask (85k) and target (100k), per the scoring rules. **Qualifications: 9** β€” You have 9 years against their 7-year ask; OpenAPI and docs-as-code are backed. Go is a known gap for this lane; costs 1 point. diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v2.md index 8040f9a..b2806dd 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v2.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v2.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.10 (the simulated user's calls are not included). @@ -125,15 +127,15 @@ Now I'll read the posting systematically and write the findings. "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ], "tier_used": null, @@ -219,7 +221,7 @@ Now I'll read the posting systematically and write the findings. **Fit: 10** – All four must-haves (AI-forward, docs-as-code, CI automation, ownership) rate strong or better; you'd have real autonomy over tools and priorities. -**Comp: 8** – Region B tops out at $150,000, above your minimum ($110,000) but below your target ($140,000). +**Comp: 8** – Region B tops out at $104,000, above your minimum ($80,000) but below your target ($100,000). **Qualifications: 9** – You exceed the 7 years required, your API docs work matches the OpenAPI requirement (ev-northwind-api-rebuild), your CI pipeline matches the build automation they need (ev-placeholder-ci-checks), and your docs-as-code pipeline matches their current setup (ev-placeholder-docs-pipeline). Only gap: reading Go code is self-score 1, not load-bearing for daily work. diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v3.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v3.md index db24a4b..32078c0 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v3.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-haiku-v3.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.10 (the simulated user's calls are not included). @@ -96,15 +98,15 @@ Let me read the posting against each lane requirement and skill: "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ], "tier_used": null, diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v1.md index da70330..db16777 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v1.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.17 (the simulated user's calls are not included). @@ -75,10 +77,10 @@ No hard block trips. The posting says "We are a remote company", so the non-remo "read": "You would pick the tools and set your own priorities, with a weekly check-in from your manager."}, "pay": {"stated": true, "tier_used": null, "top": null, "other_pay_noted": [], "tiers": [ - {"label": "Region B", "min": 125000, "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000"}, - {"label": "Region A", "min": 140000, "max": 170000, - "quote": "Region A: $140,000 to $170,000"}]}, + {"label": "Region B", "min": 100000, "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000"}, + {"label": "Region A", "min": 102000, "max": 118000, + "quote": "Region A: $102,000 to $118,000"}]}, "culture": {"perks": [ {"perk_id": "unlimited_pto", "quote": "Unlimited paid time off"}, {"perk_id": "learning_budget", "quote": "A $3,000 yearly learning budget"}, @@ -116,7 +118,7 @@ Before you run the scripts: | Score | Arithmetic | Result | |---|---|---| | Fit | Start at 10. Strong must-haves `ai_forward` and `docs_as_code` cost 0. Soft `ci_for_docs` rated fair costs 0. Bonus `solo_ownership` costs 0. Autonomy is positive and no avoid-skills are core, so 0. | **10** | -| Comp | Region B tier, top $150,000, which is at or above the $140,000 target. | **10** | +| Comp | Region B tier, top $104,000, which is at or above the $100,000 target. | **10** | | Qualifications | Start at 10. Asks for 7 years against your 9, so 0. `gap_go` costs 1. No working-style mismatch. | **9** | | Culture | Start at 5. Unlimited PTO +2, learning budget +2, meetups +1, "definition of done" +1 gives 11, capped at 10. Time off is not low and there is no hustle language. | **10** | diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v2.md index 9a48c88..9c4d4e1 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v2.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v2.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.18 (the simulated user's calls are not included). @@ -112,8 +114,8 @@ I couldn't run `check_findings.py` or `assess_offline.py` from here, so none of "top": null, "other_pay_noted": [], "tiers": [ - {"label": "Region B", "min": 125000, "max": 150000, "quote": "Base pay range for Region B: $125,000 to $150,000."}, - {"label": "Region A", "min": 140000, "max": 170000, "quote": "Region A: $140,000 to $170,000."} + {"label": "Region B", "min": 100000, "max": 104000, "quote": "Base pay range for Region B: $100,000 to $104,000."}, + {"label": "Region A", "min": 102000, "max": 118000, "quote": "Region A: $102,000 to $118,000."} ] }, "culture": { @@ -162,7 +164,7 @@ I couldn't run `check_findings.py` or `assess_offline.py` from here, so none of "name": "Copperline Example Co.", "stored_reason": "Makes tools for writers, which is the work Robin cares about most." }, - "verdict_reason": "You would own the docs system from the first week and choose the tools, with AI drafting and docs kept in Git every day. The pay range reaches $150,000 for your region, and the one thing to check is how much Go code you would need to read.", + "verdict_reason": "You would own the docs system from the first week and choose the tools, with AI drafting and docs kept in Git every day. The pay range reaches $104,000 for your region, and the one thing to check is how much Go code you would need to read.", "keyword_signals_found": [ { "phrase": "definition of done", @@ -180,7 +182,7 @@ I couldn't run `check_findings.py` or `assess_offline.py` from here, so none of | Score | Result | What moved it | |---|---|---| | Fit | **10** | Both strong must-haves are rated strong, so there are no deductions. The soft item (automatic docs checks) is fair and costs 0, the bonus item never costs anything, and autonomy is net positive. No avoid-skills are core work. | -| Comp | **10** | The Region B tier tops out at $150,000, which is above your $140,000 target. | +| Comp | **10** | The Region B tier tops out at $104,000, which is above your $100,000 target. | | Qualifications | **9** | 10 minus 1 for the Go known gap (`gap_go`). The posting asks for 7+ years and you have 9, and there is no working-style mismatch. | | Culture | **10** | 5, plus 2 for unlimited PTO, 2 for the learning budget, 1 for meetups and 1 for "definition of done". That is 11, capped at 10. | diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v3.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v3.md index 5dc8045..9904111 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v3.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p3-sonnet-v3.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.14 (the simulated user's calls are not included). @@ -109,8 +111,8 @@ I read the posting directly and ran no scripts, so nothing below has been throug "pay": { "stated": true, "tiers": [ - {"label": "Region B", "min": 125000, "max": 150000, "quote": "Base pay range for Region B: $125,000 to $150,000."}, - {"label": "Region A", "min": 140000, "max": 170000, "quote": "Region A: $140,000 to $170,000."} + {"label": "Region B", "min": 100000, "max": 104000, "quote": "Base pay range for Region B: $100,000 to $104,000."}, + {"label": "Region A", "min": 102000, "max": 118000, "quote": "Region A: $102,000 to $118,000."} ], "tier_used": null, "top": null, diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md index 8dc7221..ad6057d 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-haiku-v1.md @@ -1,6 +1,8 @@ # Prompt run: Test the scoring with the fixture, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.07 (the simulated user's calls are not included). @@ -31,7 +33,7 @@ I'll work through each posting's scoring, showing all arithmetic, then compare w 10 βˆ’ 0 βˆ’ 0 βˆ’ 1 βˆ’ 0 = **9** -**Comp:** Pay stated, Region B tier, top $150,000. Robin's target is $140,000. Top is at or above target. Score **10**. +**Comp:** Pay stated, Region B tier, top $104,000. Robin's target is $78,000. Top is at or above target. Score **10**. **Qualifications:** Start at 10. - Years required 7, Robin has 9 β†’ 0 (not more, not narrow) @@ -62,7 +64,7 @@ I'll work through each posting's scoring, showing all arithmetic, then compare w 10 βˆ’ 0 βˆ’ 0 βˆ’ 1 βˆ’ 2 = **7** -**Comp:** Pay stated, tier "Everywhere else," top $100,000. Floor $90K, min $110K. $100K is above floor but below min. Score **3**. +**Comp:** Pay stated, tier "Everywhere else," top $78,000. Floor $75K, min $80K. $78K is above floor but below min. Score **3**. **Qualifications:** Start at 10. - Years required 12, Robin has 9 β†’ 2 diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md index a804b1c..dc9a7ff 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p4-sonnet-v1.md @@ -1,6 +1,8 @@ # Prompt run: Test the scoring with the fixture, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.12 (the simulated user's calls are not included). @@ -23,7 +25,7 @@ I worked all three from the rules and findings first, then compared against `exp | Score | Arithmetic | Result | |---|---|---| | Fit | 10. `ai_forward` (strong, rated strong): 0. `docs_as_code` (strong, rated strong): 0. `ci_for_docs` (soft, rated weak): -1. `solo_ownership` is a bonus: 0. Autonomy is net positive: 0. No avoided skills. 10 - 1 = 9. | **9** | -| Comp | Robin's label is Region B, so the Region B tier is used (not the model's choice, but the same). Top is $150,000, which is at or above the $140,000 target. | **10** | +| Comp | Robin's label is Region B, so the Region B tier is used (not the model's choice, but the same). Top is $104,000, which is at or above the $78,000 target. | **10** | | Qualifications | 10. The posting asks for 7 years and Robin has 9: 0. `go_lang` is a known gap: -1. Robin's self-score of 1 on Go is the same skill, so it counts once: 0. No working-style mismatch. 10 - 1 = 9. | **9** | | Culture | 5. `unlimited_pto` (big): +2. `learning_budget` (big): +2. `offsites` (nice): +1. "definition of done" (strong positive phrase): +1. 5 + 2 + 2 + 1 + 1 = 11, clamped to the cap of 10. | **10** | @@ -39,7 +41,7 @@ I worked all three from the rules and findings first, then compared against `exp | Score | Arithmetic | Result | |---|---|---| | Fit | 10. `expert_access` (strong, fair): 0. `tooling_voice` (strong, fair): 0. `style_guide` (soft, poor): -1. Autonomy unknown: 0. Avoided skills that are core daily work: `video_tutorials` -1 and `localization_pm` -1, which is -2 and within the cap of 2. 10 - 1 - 2 = 7. | **7** | -| Comp | No tier label contains "Region B", so the "Everywhere else" tier is used. Top is $100,000: at or above the $90,000 floor, below the $110,000 min. | **3** | +| Comp | No tier label contains "Region B", so the "Everywhere else" tier is used. Top is $78,000: at or above the $75,000 floor, below the $80,000 min. | **3** | | Qualifications | 10. 12 years asked against Robin's 9: -2 (once). Known gaps `dita_xml` and `mobile_sdk`: -2. No self-score gaps. 10 - 2 - 2 = 6. | **6** | | Culture | 5. No perks, phrases, flags or negatives. | **5** | diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v1.md index 2b70245..d10dc47 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v1.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -41,16 +43,16 @@ I'm attaching the output of a failed command from this tool (the validator, the ``` **Problem 3: comp pay numbers out of order** -- **Cause:** The min value (130000) exceeds open_ask (125000), but the validator requires min ≀ open_ask. +- **Cause:** The min value (98000) exceeds open_ask (85000), but the validator requires min ≀ open_ask. - **Location:** Lines 20–22, comp section, fields `min` and `open_ask` - **Fix:** Adjust the values so min ≀ open_ask (your choice which values are correct): ```yaml # Before: - min: 130000 - open_ask: 125000 + min: 98000 + open_ask: 85000 # After (example): - min: 125000 - open_ask: 130000 + min: 85000 + open_ask: 98000 ``` **Problem 4: evidence[1].proof** diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v2.md index 055f207..bcb9d24 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v2.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v2.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -31,10 +33,10 @@ The reference "ev-placeholder-style-guide" doesn't exist in the evidence section - After: `evidence_ids: [ev-placeholder-docs-ci]` **Problem 3: comp** -The order is wrong: min (130000) must be ≀ open_ask (125000). You decide which to change to fix the sequence floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. +The order is wrong: min (98000) must be ≀ open_ask (85000). You decide which to change to fix the sequence floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. - Lines: `comp.min` and `comp.open_ask` -- Before: `min: 130000` / `open_ask: 125000` -- After: `min: 125000` / `open_ask: 130000` (or adjust other values) +- Before: `min: 98000` / `open_ask: 85000` +- After: `min: 85000` / `open_ask: 98000` (or adjust other values) **Problem 4: evidence[1].proof** An interview source cannot carry proof: checked because no document was read to verify it. Change to unchecked. diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v3.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v3.md index e549c62..a882524 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v3.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-haiku-v3.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -35,10 +37,10 @@ I'm attaching the output of a failed command from this tool (the validator, the --- ## Problem 3: Compensation order violation -**Cause:** Lines 28–30 have `min: 130000` and `open_ask: 125000`, violating the required order of floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. +**Cause:** Lines 28–30 have `min: 98000` and `open_ask: 85000`, violating the required order of floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. **Fix (choose one):** -- Before: `min: 130000` / `open_ask: 125000` +- Before: `min: 98000` / `open_ask: 85000` - After: `min: ` / `open_ask: ` (ensure min ≀ open_ask) --- diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v1.md index 5e57c78..e3f8906 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v1.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.04 (the simulated user's calls are not included). @@ -49,21 +51,21 @@ All five problems are fixable with one-line edits. Three need a choice from you, ## 3. `comp.min` and `comp.open_ask` -**Cause:** `min` is 130000 but `open_ask` is 125000. The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`. +**Cause:** `min` is 98000 but `open_ask` is 85000. The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`. **Your call:** which number is wrong? -- **A.** If `min` is wrong, lower it to a value between 90000 (`floor`) and 125000. -- **B.** If `open_ask` is wrong, raise it to a value between 130000 and 140000 (`target`). +- **A.** If `min` is wrong, lower it to a value between 75000 (`floor`) and 85000. +- **B.** If `open_ask` is wrong, raise it to a value between 98000 and 100000 (`target`). ```yaml # before - min: 130000 - open_ask: 125000 -# after (option A; is yours to choose, 90000 to 125000) + min: 98000 + open_ask: 85000 +# after (option A; is yours to choose, 75000 to 85000) min: - open_ask: 125000 -# after (option B; is yours to choose, 130000 to 140000) - min: 130000 + open_ask: 85000 +# after (option B; is yours to choose, 98000 to 100000) + min: 98000 open_ask: ``` diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v2.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v2.md index 91c0d24..7ccbfb3 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v2.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v2.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.04 (the simulated user's calls are not included). @@ -43,20 +45,20 @@ evidence_ids: [ev-placeholder-release-notes] ``` ## 3. `comp`: `min` is above `open_ask` -**Cause:** The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`, but `min` is 130000 and `open_ask` is 125000. +**Cause:** The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`, but `min` is 98000 and `open_ask` is 85000. **Choices:** The number is your decision. Change one of the two: -- Lower `min` to a value from 90000 (`floor`) to 125000. -- Raise `open_ask` to a value from 130000 to 140000 (`target`). +- Lower `min` to a value from 75000 (`floor`) to 85000. +- Raise `open_ask` to a value from 98000 to 100000 (`target`). ```yaml # before -min: 130000 -open_ask: 125000 -# after (option A: you pick min, 90000–125000) +min: 98000 +open_ask: 85000 +# after (option A: you pick min, 75000–85000) min: -open_ask: 125000 -# after (option B: you pick open_ask, 130000–140000) -min: 130000 +open_ask: 85000 +# after (option B: you pick open_ask, 98000–100000) +min: 98000 open_ask: ``` diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v3.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v3.md index b740b8a..e714bf6 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v3.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/p6-sonnet-v3.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -48,20 +50,20 @@ All five problems are in `fixtures/broken-profile.yaml`. Nothing was changed. ## 3. `comp` (`min` / `open_ask`) -**Cause:** The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`, but `min` (130000) is above `open_ask` (125000). +**Cause:** The order must be `floor <= min <= open_ask <= target <= stretch_ceiling`, but `min` (98000) is above `open_ask` (85000). -**Which number is wrong is your call.** Either value fits between `floor` (90000) and `target` (140000), so keep one and change the other. +**Which number is wrong is your call.** Either value fits between `floor` (75000) and `target` (100000), so keep one and change the other. ```yaml # before - min: 130000 - open_ask: 125000 + min: 98000 + open_ask: 85000 # after (option A: lower min) - min: - open_ask: 125000 + min: + open_ask: 85000 # after (option B: raise open_ask) - min: 130000 - open_ask: + min: 98000 + open_ask: ``` ## 4. `evidence[1].proof` (ev-placeholder-docs-ci) diff --git a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/r2-haiku-v1.md b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/r2-haiku-v1.md index 686e5b0..f6b1607 100644 --- a/pipelines/job-assessment/tests/prompt-runs/earlier-versions/r2-haiku-v1.md +++ b/pipelines/job-assessment/tests/prompt-runs/earlier-versions/r2-haiku-v1.md @@ -1,6 +1,8 @@ # Prompt run: Customize the rubric (assessment/README.md), on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/r2-customize-rubric.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 6. Cost: $0.33 (the simulated user's calls are not included). @@ -137,11 +139,11 @@ named_exceptions: - {company: Lanternfield Example Co., block_id: non_remote, why: "Robin lives a ten minute walk from this office and would go in.", added: "2026-10-01"} comp: currency: USD - floor: 90000 - min: 110000 - open_ask: 125000 - target: 140000 - stretch_ceiling: 170000 + floor: 75000 + min: 80000 + open_ask: 85000 + target: 100000 + stretch_ceiling: 175000 culture: perks: - {id: unlimited_pto, label: Unlimited PTO, kind: big} diff --git a/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md index 8a49439..8257c9e 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p3-haiku.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.10 (the simulated user's calls are not included). @@ -116,15 +118,15 @@ I'm attaching assessment/SKILL.md, schema/findings.schema.json, my career-profil "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ], "tier_used": null, diff --git a/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md index e44cf1d..159b1d7 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p3-sonnet.md @@ -1,6 +1,8 @@ # Prompt run: Run on my first posting, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/03-run-first-posting.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.14 (the simulated user's calls are not included). @@ -121,15 +123,15 @@ I haven't run `check_findings.py` or `assess_offline.py`, and I computed no scor "tiers": [ { "label": "Region B", - "min": 125000, - "max": 150000, - "quote": "Base pay range for Region B: $125,000 to $150,000." + "min": 100000, + "max": 104000, + "quote": "Base pay range for Region B: $100,000 to $104,000." }, { "label": "Region A", - "min": 140000, - "max": 170000, - "quote": "Region A: $140,000 to $170,000." + "min": 102000, + "max": 118000, + "quote": "Region A: $102,000 to $118,000." } ], "tier_used": null, diff --git a/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md index 62538cc..fd564ae 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p4-haiku.md @@ -1,6 +1,8 @@ # Prompt run: Test the scoring with the fixture, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.06 (the simulated user's calls are not included). @@ -30,7 +32,7 @@ I'll work through each posting systematically, applying the scoring rules from A - Autonomy: "positive" β†’ 0 - **10 - 1 = 9** -**Comp:** Pay listed. Region B tier: max $150,000. Robin's target is $140,000. $150,000 β‰₯ target β†’ **10** +**Comp:** Pay listed. Region B tier: max $104,000. Robin's target is $78,000. $104,000 β‰₯ target β†’ **10** **Qualifications:** Start at 10. - Years: 7+ required, Robin has 9 β†’ 0 @@ -58,7 +60,7 @@ I'll work through each posting systematically, applying the scoring rules from A - Avoid skills: `video_tutorials` (core) -1, `localization_pm` (core) -1, capped at 2 total β†’ -2 - **10 - 1 - 2 = 7** -**Comp:** Pay listed. Robin's location "Region B" doesn't match any tier. Falls back to "Everywhere else" tier: max $100,000. Robin's floor is $90,000, min is $110,000. $100,000 β‰₯ floor but below min β†’ **3** +**Comp:** Pay listed. Robin's location "Region B" doesn't match any tier. Falls back to "Everywhere else" tier: max $78,000. Robin's floor is $75,000, min is $80,000. $78,000 β‰₯ floor but below min β†’ **3** **Qualifications:** Start at 10. - Years: 12+ required, Robin has 9 β†’ -2 diff --git a/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md index c77f33b..6b1c155 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p4-sonnet.md @@ -1,6 +1,8 @@ # Prompt run: Test the scoring with the fixture, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/04-test-scoring.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.12 (the simulated user's calls are not included). @@ -23,7 +25,7 @@ I did all three by hand from the findings files, the profile and ARCHITECTURE.md | Score | Arithmetic | Result | |---|---|---| | Fit | 10. `ai_forward` (strong must-have) rated strong: 0. `docs_as_code` (strong) rated strong: 0. `ci_for_docs` (soft) rated weak: -1. `solo_ownership` is a bonus: 0. Autonomy net positive: 0. No avoid-core skills: 0. | **9** | -| Comp | The tier is Region B, matching Robin's location label. Top is 150,000, which is at or above the target of 140,000. | **10** | +| Comp | The tier is Region B, matching Robin's location label. Top is 104,000, which is at or above the target of 78,000. | **10** | | Qualifications | 10. Asks for 7 years and Robin has 9: 0. Known gap `go_lang`: -1. No self-score gaps. No style mismatch. | **9** | | Culture | 5. Unlimited PTO (big) +2, learning budget (big) +2, meetups (nice) +1, "definition of done" (strong positive) +1. That is 11, capped at 10. | **10** | @@ -39,7 +41,7 @@ I did all three by hand from the findings files, the profile and ARCHITECTURE.md | Score | Arithmetic | Result | |---|---|---| | Fit | 10. `expert_access` (strong) fair: 0. `tooling_voice` (strong) fair: 0. `style_guide` (soft) poor: -1. Autonomy unknown: 0. Avoid-core skills are `video_tutorials` and `localization_pm`, -1 each, which is within the cap of 2: -2. 10 - 1 - 2 = 7. | **7** | -| Comp | No tier label contains "Region B", so the code uses the "Everywhere else" tier. Top is 100,000, which is at or above the floor of 90,000 but below the min of 110,000. | **3** | +| Comp | No tier label contains "Region B", so the code uses the "Everywhere else" tier. Top is 78,000, which is at or above the floor of 75,000 but below the min of 80,000. | **3** | | Qualifications | 10. Asks for 12 years and Robin has 9: -2, once. Known gaps `dita_xml` and `mobile_sdk`: -2. Nothing else. 10 - 2 - 2 = 6. | **6** | | Culture | 5. No perks, phrases, flags or hustle language. | **5** | @@ -76,5 +78,5 @@ The scores would not have changed this outcome: Comp is 3 (so 4c would also appl Three steps could have gone differently: - **01 Culture:** the raw sum is 11, so the cap at 10 matters. -- **02 Comp:** it only lands on 3 if you pick the "Everywhere else" tier, which is what Interpretation 7 says to do when no tier names Region B. Using Region A's top of 120,000 would give 8. +- **02 Comp:** it only lands on 3 if you pick the "Everywhere else" tier, which is what Interpretation 7 says to do when no tier names Region B. Using Region A's top of 98,000 would give 8. - **02 Fit:** it needs the avoid-skills cap of 2 applied across both listed skills, with no stacking beyond that. diff --git a/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md b/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md index f7151f3..e0961a1 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md +++ b/pipelines/job-assessment/tests/prompt-runs/p6-haiku.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on haiku +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `haiku`, which ran as `claude-haiku-4-5-20251001` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -49,25 +51,25 @@ evidence_ids: [] # e.g., ev-placeholder-docs-ci --- # Problem 3: comp -**Cause:** The compensation values are not in ascending orderβ€”min (130000) exceeds open_ask (125000). The correct order must be: floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. +**Cause:** The compensation values are not in ascending orderβ€”min (98000) exceeds open_ask (85000). The correct order must be: floor ≀ min ≀ open_ask ≀ target ≀ stretch_ceiling. **Location:** `comp.min` and `comp.open_ask` **Fix:** Reorder the pay values: ```yaml # Before -floor: 90000 -min: 130000 -open_ask: 125000 -target: 140000 -stretch_ceiling: 170000 +floor: 75000 +min: 98000 +open_ask: 85000 +target: 100000 +stretch_ceiling: 175000 # After -floor: 90000 +floor: 75000 min: open_ask: -target: 140000 -stretch_ceiling: 170000 +target: 100000 +stretch_ceiling: 175000 ``` --- diff --git a/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md b/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md index 61d1a18..2e7d2ff 100644 --- a/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md +++ b/pipelines/job-assessment/tests/prompt-runs/p6-sonnet.md @@ -1,6 +1,8 @@ # Prompt run: Fix errors, on sonnet +Note: the pay figures in this record were changed after the run to fictional ones. Nothing else was edited; the scores and verdicts are as recorded. + - Prompt file: `prompts/06-fix-errors.txt` - Date: 2026-10-01. Model alias `sonnet`, which ran as `claude-sonnet-5-5` (Claude Code can add a small helper model call of its own). - Turns: 1. Cost: $0.03 (the simulated user's calls are not included). @@ -53,22 +55,22 @@ Your options: ## 3. `comp` -**Cause:** The pay numbers must satisfy `floor <= min <= open_ask <= target <= stretch_ceiling`. Here `min` (130000) is above `open_ask` (125000). +**Cause:** The pay numbers must satisfy `floor <= min <= open_ask <= target <= stretch_ceiling`. Here `min` (98000) is above `open_ask` (85000). **Path:** `comp.min` and `comp.open_ask`. -Which number is wrong is your call. Either lower `min` to 125000 or less (and keep it at or above the floor of 90000), or raise `open_ask` to at least 130000 (and keep it at or below the target of 140000). +Which number is wrong is your call. Either lower `min` to 85000 or less (and keep it at or above the floor of 75000), or raise `open_ask` to at least 98000 (and keep it at or below the target of 100000). ```yaml # before - min: 130000 - open_ask: 125000 + min: 98000 + open_ask: 85000 # after (fix one of the two) - min: - open_ask: 125000 + min: + open_ask: 85000 # or - min: 130000 - open_ask: + min: 98000 + open_ask: ``` ## 4. `evidence[1].proof` diff --git a/pipelines/job-assessment/tests/test_check_findings.py b/pipelines/job-assessment/tests/test_check_findings.py index 86b96e7..3652692 100644 --- a/pipelines/job-assessment/tests/test_check_findings.py +++ b/pipelines/job-assessment/tests/test_check_findings.py @@ -290,3 +290,53 @@ def test_cli_exit_codes(tmp_path, capsys): assert cf.main([str(tmp_path / "bad.json")] + args) == 1 assert "was not found in the posting text" in capsys.readouterr().out assert cf.main([str(tmp_path / "missing.json")] + args) == 2 + + +# A culture list holding plain strings instead of objects used to end in a Python +# traceback. These pin the clean one-line report and exit code 1. +@pytest.mark.parametrize("key", ["strong_positive_phrases", "hustle", + "negative_phrases", "soft_flags_tripped", "perks"]) +def test_a_culture_list_of_strings_is_reported_not_a_crash(key): + f = mutate(lambda f: f["culture"].update({key: ["a plain string"]})) + problems = cf.check(f, POSTING_BODY, "lane-a", PROFILE) + assert any(p.startswith(f"culture.{key}[0]: must be an object") for p in problems) + + +@pytest.mark.parametrize("mutator", [ + lambda f: f.update(requirements=["no_micromanagement"]), + lambda f: f.update(culture="great"), + lambda f: f.update(hard_block="none"), + lambda f: f["qualifications"].update(matches=["a string"]), + lambda f: f["qualifications"].update(unproven="a string"), +]) +def test_other_wrong_shapes_report_problems_without_crashing(mutator): + problems = cf.check(mutate(mutator), POSTING_BODY, "lane-a", PROFILE) + assert problems and all(isinstance(p, str) for p in problems) + + +def test_cli_wrong_shape_prints_clean_lines_and_exits_1(tmp_path, capsys): + (tmp_path / "p.md").write_text(POSTING) + (tmp_path / "prof.yaml").write_text(yaml.safe_dump(PROFILE)) + bad = good() + bad["culture"]["strong_positive_phrases"] = ["Documentation is part of the definition of done."] + (tmp_path / "bad.json").write_text(json.dumps(bad)) + args = ["--posting", str(tmp_path / "p.md"), "--profile", str(tmp_path / "prof.yaml")] + assert cf.main([str(tmp_path / "bad.json")] + args) == 1 + out = capsys.readouterr() + assert "Traceback" not in out.out + out.err + assert "culture.strong_positive_phrases[0]: must be an object" in out.out + assert len(out.out.strip().splitlines()) == 2 # the problem, then the count + + +def test_unknown_shape_failure_is_one_clean_line_and_exit_1(tmp_path, capsys, monkeypatch): + (tmp_path / "p.md").write_text(POSTING) + (tmp_path / "prof.yaml").write_text(yaml.safe_dump(PROFILE)) + (tmp_path / "f.json").write_text(json.dumps(good())) + + def boom(*a, **k): + raise AttributeError("'str' object has no attribute 'get'") + monkeypatch.setattr(cf, "check", boom) + args = ["--posting", str(tmp_path / "p.md"), "--profile", str(tmp_path / "prof.yaml")] + assert cf.main([str(tmp_path / "f.json")] + args) == 1 + out = capsys.readouterr().out.strip().splitlines() + assert len(out) == 1 and out[0].startswith("check_findings: the findings file has a shape") diff --git a/pipelines/job-assessment/tests/test_load_criteria.py b/pipelines/job-assessment/tests/test_load_criteria.py index 9df070b..29b2af0 100644 --- a/pipelines/job-assessment/tests/test_load_criteria.py +++ b/pipelines/job-assessment/tests/test_load_criteria.py @@ -29,8 +29,8 @@ "named_exceptions": [ {"company": "Placeholder Labs", "block_id": "non_remote", "why": "One-off.", "added": "2026-10-01"}, ], - "comp": {"currency": "USD", "floor": 90000, "min": 110000, "open_ask": 125000, - "target": 140000, "stretch_ceiling": 170000}, + "comp": {"currency": "USD", "floor": 75000, "min": 80000, "open_ask": 85000, + "target": 100000, "stretch_ceiling": 175000}, "culture": { "perks": [{"id": "unlimited_pto", "label": "Unlimited PTO", "kind": "big"}, {"id": "offsites", "label": "Regular offsites", "kind": "nice"}], diff --git a/pipelines/job-assessment/tests/test_score_helpers.py b/pipelines/job-assessment/tests/test_score_helpers.py index 83f9617..535da1c 100644 --- a/pipelines/job-assessment/tests/test_score_helpers.py +++ b/pipelines/job-assessment/tests/test_score_helpers.py @@ -19,8 +19,8 @@ def make_profile(**over): profile = { "person": {"display_name": "Robin Sample", "years_experience": 9, "location_label": "Region B"}, - "comp": {"floor": 90000, "min": 110000, "open_ask": 125000, - "target": 140000, "stretch_ceiling": 170000}, + "comp": {"floor": 75000, "min": 80000, "open_ask": 85000, + "target": 100000, "stretch_ceiling": 175000}, "culture": { "perks": [ {"id": "unlimited_pto", "kind": "big"}, @@ -197,14 +197,14 @@ def test_comp_missing_pay_block_is_unlisted(): @pytest.mark.parametrize("top,expected", [ - (170000, 10), # above target - (140000, 10), # exactly the target - (139999, 8), # just under target - (125000, 8), # the open ask is still the 8 band - (110000, 8), # exactly the minimum - (109999, 3), # just under the minimum - (90000, 3), # exactly the floor - (89999, 1), # below the floor + (175000, 10), # above target + (100000, 10), # exactly the target + (99999, 8), # just under target + (85000, 8), # the open ask is still the 8 band + (80000, 8), # exactly the minimum + (79999, 3), # just under the minimum + (75000, 3), # exactly the floor + (74999, 1), # below the floor (1, 1), ]) def test_comp_top_of_range_bands(top, expected): @@ -212,39 +212,39 @@ def test_comp_top_of_range_bands(top, expected): def test_comp_scores_the_top_of_a_range_not_the_bottom(): - tiers = [{"label": "Region B", "min": 95000, "max": 145000}] + tiers = [{"label": "Region B", "min": 72000, "max": 102000}] assert sh.comp_score(pay(tiers=tiers), P) == 10 def test_comp_named_location_tier_wins_over_nationwide(): tiers = [ - {"label": "Nationwide", "min": 150000, "max": 180000}, - {"label": "Region B", "min": 80000, "max": 100000}, + {"label": "Nationwide", "min": 104000, "max": 118000}, + {"label": "Region B", "min": 72000, "max": 78000}, ] assert sh.comp_score(pay(tiers=tiers), P) == 3 def test_comp_nationwide_tier_when_location_not_named(): tiers = [ - {"label": "Region A", "min": 80000, "max": 95000}, - {"label": "Everywhere else", "min": 130000, "max": 150000}, + {"label": "Region A", "min": 72000, "max": 78000}, + {"label": "Everywhere else", "min": 102000, "max": 118000}, ] assert sh.comp_score(pay(tiers=tiers), P) == 10 def test_comp_lowest_tier_when_no_location_and_no_nationwide(): tiers = [ - {"label": "Region A", "min": 150000, "max": 170000}, - {"label": "Region C", "min": 85000, "max": 100000}, + {"label": "Region A", "min": 102000, "max": 118000}, + {"label": "Region C", "min": 72000, "max": 78000}, ] assert sh.comp_score(pay(tiers=tiers), P) == 3 def test_comp_tier_order_is_named_then_nationwide_then_lowest(): tiers = [ - {"label": "Region A", "max": 90000}, - {"label": "All other locations", "max": 120000}, - {"label": "Region B", "max": 150000}, + {"label": "Region A", "max": 78000}, + {"label": "All other locations", "max": 98000}, + {"label": "Region B", "max": 118000}, ] assert sh.pick_tier(tiers, "Region B")["label"] == "Region B" assert sh.pick_tier(tiers, "Region Z")["label"] == "All other locations" @@ -252,7 +252,7 @@ def test_comp_tier_order_is_named_then_nationwide_then_lowest(): def test_comp_base_pay_only_other_pay_is_ignored(): - f = pay(top=100000) + f = pay(top=78000) f["pay"]["other_pay_noted"] = ["Bonus up to 50000", "Stock grant 200000"] assert sh.comp_score(f, P) == 3 @@ -522,10 +522,10 @@ def test_a_pay_range_written_as_the_schema_describes_scores_end_to_end(tmp_path) props = _schema_tier_properties() lo_key, hi_key = "min", "max" assert lo_key in props and hi_key in props - tier = {"label": "all other US locations", lo_key: 100000, hi_key: 150000, + tier = {"label": "all other US locations", lo_key: 98000, hi_key: 104000, "quote": "q"} findings = pay(tiers=[tier]) - # top 150000 is above the profile target of 140000, so comp is 10 + # top 104000 is above the profile target of 100000, so comp is 10 assert sh.comp_score(findings, P) == 10 # and through the command line entry point, as the skill runs it fpath = tmp_path / "findings.json" diff --git a/pipelines/job-assessment/tests/test_validate_profile.py b/pipelines/job-assessment/tests/test_validate_profile.py index a2201c4..efa73f9 100644 --- a/pipelines/job-assessment/tests/test_validate_profile.py +++ b/pipelines/job-assessment/tests/test_validate_profile.py @@ -34,7 +34,7 @@ # rule 2, references: a skill points at an evidence id that does not exist 'skills[1].evidence_ids: "ev-placeholder-style-guide" is not an evidence id in this file.', # rule 4, pay order: min is above open_ask - "comp: pay numbers out of order: min (130000) is above open_ask (125000). " + "comp: pay numbers out of order: min (98000) is above open_ask (85000). " "Expected floor <= min <= open_ask <= target <= stretch_ceiling.", # rule 5, proof: an interview answer marked as checked "evidence[1].proof: source.type interview cannot carry proof: checked. Nothing was read to back it; " @@ -137,7 +137,7 @@ def fill_template_with_robins_ids(data: dict) -> dict: data["person"].update(display_name="Robin Sample", target_roles=["Senior Technical Writer"], years_experience=9, location_label="Region B", working_style="gather_from_experts") - data["comp"].update(floor=90000, min=110000, open_ask=125000, target=140000, stretch_ceiling=170000) + data["comp"].update(floor=75000, min=80000, open_ask=85000, target=100000, stretch_ceiling=175000) data["culture"]["perks"] = [{"id": "unlimited_pto", "label": "Unlimited PTO", "kind": "big"}] data["culture"]["low_time_off_days"] = 15 data["hard_blocks"] = [{"id": "gambling", "label": "Gambling or betting", "why": "Fictional reason."}] @@ -245,9 +245,9 @@ def test_pay_order_skips_missing_numbers(base): base["comp"]["open_ask"] = None base["comp"]["target"] = None assert errors_of(base) == [] - base["comp"]["stretch_ceiling"] = 100000 + base["comp"]["stretch_ceiling"] = 78000 assert errors_of(base) == [ - "comp: pay numbers out of order: min (110000) is above stretch_ceiling (100000). " + "comp: pay numbers out of order: min (80000) is above stretch_ceiling (78000). " "Expected floor <= min <= open_ask <= target <= stretch_ceiling."]