From d750068bc1627244eada7a2f83a050344dc252f2 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 16 Sep 2026 19:17:55 +0000 Subject: [PATCH] Add TypeSafe/Jev System One pass for Pareto candidate scoring Optional tools/typesafe-pareto helper fans out Score + Choice judgments over a small U-NSGA-III front sample. Live calls only when TYPESAFE_API_KEY is set; CI runs mocks and writes metrics JSONL. Co-authored-by: Jason --- .github/workflows/ci.yml | 15 + .gitignore | 3 + CHANGELOG.md | 4 + CONTRIBUTING.md | 13 +- README.md | 27 + SECURITY.md | 4 +- docs/NOTICE.md | 1 + metrics/.gitkeep | 0 .../fixtures/zdt1_candidates.json | 82 ++ .../typesafe-pareto/requirements-optional.txt | 3 + tools/typesafe-pareto/score_pareto.py | 770 ++++++++++++++++++ tools/typesafe-pareto/test_score_pareto.py | 224 +++++ 12 files changed, 1144 insertions(+), 2 deletions(-) create mode 100644 metrics/.gitkeep create mode 100644 tools/typesafe-pareto/fixtures/zdt1_candidates.json create mode 100644 tools/typesafe-pareto/requirements-optional.txt create mode 100644 tools/typesafe-pareto/score_pareto.py create mode 100644 tools/typesafe-pareto/test_score_pareto.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 963004f..3e67f37 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -38,3 +38,18 @@ jobs: name: nupkg path: artifacts/*.nupkg if-no-files-found: error + + typesafe-pareto: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + + - name: Mock tests + run: python -m unittest tools/typesafe-pareto/test_score_pareto.py -v + + - name: Smoke (mock) + run: python tools/typesafe-pareto/score_pareto.py --smoke --force-mock --metrics metrics/typesafe-runs.jsonl diff --git a/.gitignore b/.gitignore index 7103cbb..a6bb1fe 100644 --- a/.gitignore +++ b/.gitignore @@ -33,6 +33,9 @@ __pycache__/ venv/ .pytest_cache/ +# TypeSafe / Jev live run log (regenerated by tools/typesafe-pareto) +metrics/typesafe-runs.jsonl + # Local secrets *.pfx appsettings.*.local.json diff --git a/CHANGELOG.md b/CHANGELOG.md index 3cb053f..e4c35fe 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/). ## [Unreleased] +### Added + +- Optional **TypeSafe / Jev System One** pass (`tools/typesafe-pareto`) for Score + Choice judgments over a small Pareto candidate sample. Additive semantic layer — does not replace NSGA-III / U-NSGA-III objectives. Live calls only when `TYPESAFE_API_KEY` is set; CI uses mocks. Metrics append to `metrics/typesafe-runs.jsonl`. + ## [0.1.3] — 2026-08-10 ### Added diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index de934c9..21fc8c0 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -12,6 +12,7 @@ Thanks for your interest. This library aims to be a **faithful, well-tested** U- - **.NET 10 SDK** ([download](https://dotnet.microsoft.com/download)) - Optional (oracle / multi-seed stats): Python 3.10+ with `pip install pymoo` +- Optional (TypeSafe / Jev Pareto scores): Python 3.10+; live calls need `TYPESAFE_API_KEY` (never commit it) ```bash git clone https://github.com/AppSprout-dev/Unsga3.git @@ -34,6 +35,15 @@ python tools/oracle/run_multiseed_wilcoxon.py --problems zdt1 dtlz2 --seeds 15 See [`docs/EQUIVALENCE.md`](docs/EQUIVALENCE.md) and [`docs/ORACLE-RESULTS.md`](docs/ORACLE-RESULTS.md). +### TypeSafe / Jev Pareto scoring (optional) + +```bash +python -m unittest tools/typesafe-pareto/test_score_pareto.py -v +python tools/typesafe-pareto/score_pareto.py --smoke --force-mock +``` + +Live System One calls run only when `TYPESAFE_API_KEY` is set. Metrics append to `metrics/typesafe-runs.jsonl` (gitignored). This path does not change algorithm defaults or oracle results. + Grok Build skills (maintainers): `/unsga3-oracle` · `/unsga3-release` under [`.grok/skills/`](.grok/skills/). ## What we welcome @@ -64,7 +74,8 @@ Grok Build skills (maintainers): `/unsga3-oracle` · `/unsga3-release` under [`. - One logical change per PR - Update [`CHANGELOG.md`](CHANGELOG.md) under **Unreleased** when user-visible - If you touch survival / normalization / metrics, note oracle impact (or re-run multi-seed) -- Do not commit `tools/oracle/out/` front CSVs +- Do not commit `tools/oracle/out/` front CSVs or `metrics/typesafe-runs.jsonl` +- Do not commit API keys (`.env` is gitignored; use `TYPESAFE_API_KEY` locally only) ## Reporting bugs diff --git a/README.md b/README.md index be9b787..77f51fa 100644 --- a/README.md +++ b/README.md @@ -90,6 +90,31 @@ dotnet run --project samples/BasicUsage -c Release Requires **.NET 10** SDK. Optional oracle: Python 3 + `pip install pymoo` (see [CONTRIBUTING.md](CONTRIBUTING.md)). +## TypeSafe / Jev Pareto scoring (optional) + +After a classical U-NSGA-III front, you can run a **TypeSafe System One** (Jev) pass over a small candidate sample and compare calibrated Score / Choice answers to raw objective vectors. This is an **additive semantic layer** — it does **not** replace NSGA-III / U-NSGA-III objectives, constraint-domination, or IGD. + +The helper lives in `tools/typesafe-pareto/` (stdlib Python; official `typesafe-sdk` is optional for live calls). One `POST /v1/systemone` fans out per-candidate **Score** (`constraint_satisfaction`, `diversity_value`, `exploit_vs_explore`) and **Choice** (`keep` | `drop` | `review`). Smoke batches ≤10 fixture candidates. + +```bash +# Mock (default when TYPESAFE_API_KEY is unset; what CI runs) +python tools/typesafe-pareto/score_pareto.py --smoke --force-mock + +# Live Jev (key from the environment only — never commit it) +export TYPESAFE_API_KEY=... # https://docs.typesafe.ai/sdk/python.md +pip install -r tools/typesafe-pareto/requirements-optional.txt # optional SDK +python tools/typesafe-pareto/score_pareto.py --smoke +python tools/typesafe-pareto/score_pareto.py --candidates path/to/front.json +``` + +Candidate JSON is an object with `candidates` (or `NonDominatedSolutions`), each with `objectives` and optional `variables` / `constraints` / `constraint_violation` / `feasible` / `rank`. A synthetic ZDT1-like fixture is at `tools/typesafe-pareto/fixtures/zdt1_candidates.json`. + +Metrics append one JSON line per run to **`metrics/typesafe-runs.jsonl`** (gitignored): + +`{ts, experiment:"unsga3_pareto_score", repo:"AppSprout-dev/Unsga3", model, latency_ms, usage, candidate_count, answers, notes}` + +API docs used: [HTTP](https://docs.typesafe.ai/api.md) · [Python SDK](https://docs.typesafe.ai/sdk/python.md) · [fan-out](https://docs.typesafe.ai/patterns/fan-out.md). No API keys in this repo. + ## Equivalence & research | Doc | Contents | @@ -110,6 +135,8 @@ Unsga3/ ├── samples/BasicUsage/ ├── tools/oracle/ # pymoo oracle + multi-seed stats (optional) ├── tools/OracleCompare/ # C# side of the oracle +├── tools/typesafe-pareto/ # optional TypeSafe / Jev Score+Choice on a front sample +├── metrics/ # typesafe-runs.jsonl (local; gitignored) ├── docs/ └── .github/workflows/ # CI + GitHub Packages publish ``` diff --git a/SECURITY.md b/SECURITY.md index f91574b..c1b15fd 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -7,7 +7,9 @@ | 0.1.x | Yes | | < 0.1 | No | -This is a pure numerical optimization library with **no network surface** and no default deserialization of untrusted input. Risk is mainly supply-chain (NuGet) and misuse in safety-critical decision systems. +This is a pure numerical optimization library with **no network surface** in the NuGet package and no default deserialization of untrusted input. Risk is mainly supply-chain (NuGet) and misuse in safety-critical decision systems. + +The optional `tools/typesafe-pareto` helper may call `https://api.typesafe.ai/v1/systemone` when `TYPESAFE_API_KEY` is set. Do not commit API keys or `.env` files. ## Reporting a vulnerability diff --git a/docs/NOTICE.md b/docs/NOTICE.md index b6cfbf0..24eabf2 100644 --- a/docs/NOTICE.md +++ b/docs/NOTICE.md @@ -25,6 +25,7 @@ See also `CITATION.cff` for citing **this** software. |---------|-------------------|---------------| | [pymoo](https://pymoo.org/) | Apache-2.0 | **Oracle only** — external Python scripts under `tools/oracle/` compare IGD and fronts. **Not** linked into the NuGet package. | | NumPy / SciPy (via pymoo env) | BSD | Optional stats / oracle scripts | +| [TypeSafe](https://docs.typesafe.ai/) / Jev | hosted API | Optional `tools/typesafe-pareto` System One Score/Choice pass. **Not** linked into the NuGet package; live calls need a local `TYPESAFE_API_KEY`. | No pymoo or SciPy code is vendored in `src/`. diff --git a/metrics/.gitkeep b/metrics/.gitkeep new file mode 100644 index 0000000..e69de29 diff --git a/tools/typesafe-pareto/fixtures/zdt1_candidates.json b/tools/typesafe-pareto/fixtures/zdt1_candidates.json new file mode 100644 index 0000000..f6837d1 --- /dev/null +++ b/tools/typesafe-pareto/fixtures/zdt1_candidates.json @@ -0,0 +1,82 @@ +{ + "problem": "zdt1", + "algorithm": "U-NSGA-III", + "source": "synthetic-fixture", + "notes": "Synthetic ZDT1-like points for TypeSafe / Jev smoke. Pareto-optimal ZDT1 satisfies f2 = 1 - sqrt(f1).", + "candidates": [ + { + "id": "c0", + "variables": [0.0], + "objectives": [0.0, 1.0], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c1", + "variables": [0.1], + "objectives": [0.1, 0.6838], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c2", + "variables": [0.25], + "objectives": [0.25, 0.5], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c3", + "variables": [0.5], + "objectives": [0.5, 0.2929], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c4", + "variables": [0.75], + "objectives": [0.75, 0.134], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c5", + "variables": [1.0], + "objectives": [1.0, 0.0], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 0 + }, + { + "id": "c6", + "variables": [0.5], + "objectives": [0.5, 0.6], + "constraints": [], + "constraint_violation": 0.0, + "feasible": true, + "rank": 1, + "notes": "Dominated / off-front: ZDT1 PF at f1=0.5 is f2≈0.2929" + }, + { + "id": "c7", + "variables": [0.4], + "objectives": [0.4, 0.3675], + "constraints": [0.25], + "constraint_violation": 0.25, + "feasible": false, + "rank": 0, + "notes": "Infeasible (g>0); included so constraint_satisfaction has a negative case" + } + ] +} diff --git a/tools/typesafe-pareto/requirements-optional.txt b/tools/typesafe-pareto/requirements-optional.txt new file mode 100644 index 0000000..1cb021a --- /dev/null +++ b/tools/typesafe-pareto/requirements-optional.txt @@ -0,0 +1,3 @@ +# Live System One calls only. CI and --force-mock use stdlib + fixtures. +# Docs: https://docs.typesafe.ai/sdk/python.md +typesafe-sdk diff --git a/tools/typesafe-pareto/score_pareto.py b/tools/typesafe-pareto/score_pareto.py new file mode 100644 index 0000000..e54339e --- /dev/null +++ b/tools/typesafe-pareto/score_pareto.py @@ -0,0 +1,770 @@ +#!/usr/bin/env python3 +"""TypeSafe / Jev System One pass over U-NSGA-III Pareto candidates. + +Additive semantic layer: Score + Choice judgments for a small candidate +sample. This does not replace NSGA-III / U-NSGA-III objectives. + +Docs (do not invent APIs): + https://docs.typesafe.ai/api.md + https://docs.typesafe.ai/sdk/python.md + https://docs.typesafe.ai/primitives.md + https://docs.typesafe.ai/patterns/fan-out.md + +Usage: + python tools/typesafe-pareto/score_pareto.py --smoke + python tools/typesafe-pareto/score_pareto.py --candidates path.json + +Live POST https://api.typesafe.ai/v1/systemone only when TYPESAFE_API_KEY +is set (and --force-mock is not). CI uses the in-process mock. +""" +from __future__ import annotations + +import argparse +import json +import os +import sys +import time +import urllib.error +import urllib.request +from datetime import datetime, timezone +from pathlib import Path +from typing import Any, Callable, Mapping, Sequence +from urllib.parse import urljoin + +# Optional official SDK (https://docs.typesafe.ai/sdk/python.md). +# CI does not install it; HTTP fallback matches the documented REST API. +try: + from typesafe_sdk import TypeSafeClient +except ImportError: # pragma: no cover - exercised when the extra is absent + TypeSafeClient = None + +EXPERIMENT = "unsga3_pareto_score" +REPO = "AppSprout-dev/Unsga3" +DEFAULT_MODEL = "jev-latest" +DEFAULT_BASE_URL = "https://api.typesafe.ai" +SYSTEMONE_PATH = "/v1/systemone" +API_KEY_ENV = "TYPESAFE_API_KEY" +BASE_URL_ENV = "TYPESAFE_BASE_URL" +DEFAULT_MODEL_ENV = "TYPESAFE_DEFAULT_MODEL" +DEFAULT_TIMEOUT_S = 10.0 +SMOKE_LIMIT = 10 +METRIC_FILENAME = "typesafe-runs.jsonl" + +# Ordered Score levels (index 0 = worst). See https://docs.typesafe.ai/primitives/score.md +CONSTRAINT_LEVELS = ( + "Severe constraint violation; candidate is clearly infeasible.", + "Feasibility is marginal; slack is thin or a constraint is close to breaking.", + "Constraints are satisfied with little unused slack.", + "Constraints are comfortably satisfied.", +) +DIVERSITY_LEVELS = ( + "Nearly duplicates another candidate on the supplied front.", + "Occupies a moderately populated region of the front.", + "Adds coverage in a sparsely sampled region.", +) +EXPLOIT_LEVELS = ( + "Pure exploitation of a known knee or dense cluster.", + "Balanced local refinement and spread along the front.", + "Explores a sparse or extreme region of the front.", +) +DISPOSITION_CRITERIA = { + "keep": "Retain as a high-value Pareto representative.", + "drop": "Redundant, infeasible, or low semantic value; drop from the shortlist.", + "review": "Ambiguous; a human should inspect before keeping.", +} + +HERE = Path(__file__).resolve().parent +REPO_ROOT = HERE.parents[1] +DEFAULT_FIXTURE = HERE / "fixtures" / "zdt1_candidates.json" +DEFAULT_METRICS = REPO_ROOT / "metrics" / METRIC_FILENAME + + +class TypeSafeRequestError(RuntimeError): + """Raised when the TypeSafe HTTP API returns an error or cannot be reached.""" + + +def repo_root() -> Path: + return REPO_ROOT + + +def default_metrics_path() -> Path: + return DEFAULT_METRICS + + +def load_export(path: Path) -> dict[str, Any]: + """Load a candidate JSON export (object with candidates, or a raw list).""" + raw = json.loads(path.read_text(encoding="utf-8")) + return normalize_export(raw, source_path=str(path)) + + +def normalize_export(raw: Any, *, source_path: str | None = None) -> dict[str, Any]: + if isinstance(raw, list): + payload: dict[str, Any] = {"candidates": raw} + elif isinstance(raw, dict): + payload = dict(raw) + else: + raise ValueError("Candidate export must be a JSON object or array.") + + candidates = payload.get("candidates") + if candidates is None and "NonDominatedSolutions" in payload: + candidates = payload["NonDominatedSolutions"] + payload["candidates"] = candidates + if not isinstance(candidates, list) or not candidates: + raise ValueError("Candidate export needs a non-empty 'candidates' array.") + + normalized: list[dict[str, Any]] = [] + for i, item in enumerate(candidates): + if not isinstance(item, dict): + raise ValueError(f"Candidate {i} must be an object.") + cand = dict(item) + cand.setdefault("id", f"c{i}") + objectives = cand.get("objectives") + if not isinstance(objectives, list) or not objectives: + raise ValueError(f"Candidate {cand['id']} needs a non-empty 'objectives' array.") + cand.setdefault("variables", []) + cand.setdefault("constraints", []) + cv = cand.get("constraint_violation") + if cv is None: + cv = sum(float(g) for g in cand["constraints"] if isinstance(g, (int, float)) and g > 0) + cand["constraint_violation"] = cv + cand.setdefault("feasible", float(cand["constraint_violation"]) <= 0) + normalized.append(cand) + payload["candidates"] = normalized + if source_path: + payload.setdefault("source_path", source_path) + payload.setdefault("algorithm", "U-NSGA-III") + return payload + + +def sample_candidates(payload: Mapping[str, Any], limit: int = SMOKE_LIMIT) -> list[dict[str, Any]]: + if limit < 1: + raise ValueError("limit must be >= 1") + candidates = list(payload["candidates"]) + return candidates[:limit] + + +def build_state(payload: Mapping[str, Any], candidates: Sequence[Mapping[str, Any]]) -> dict[str, Any]: + """Structured state for System One. Question instructions use dotted paths.""" + return { + "repo": REPO, + "experiment": EXPERIMENT, + "algorithm": payload.get("algorithm", "U-NSGA-III"), + "problem": payload.get("problem"), + "source": payload.get("source"), + "layer": ( + "Additive TypeSafe / Jev semantic layer beside classical Pareto fronts. " + "It does not replace NSGA-III / U-NSGA-III objective vectors." + ), + "candidates": [dict(c) for c in candidates], + } + + +def _qid(candidate_id: str, suffix: str) -> str: + safe = "".join(ch if ch.isalnum() or ch in "-_" else "_" for ch in str(candidate_id)) + return f"{safe}_{suffix}" + + +def build_questions(candidates: Sequence[Mapping[str, Any]]) -> dict[str, dict[str, Any]]: + """One System One call: per-candidate Score + Choice (speculative fan-out).""" + questions: dict[str, dict[str, Any]] = {} + for i, cand in enumerate(candidates): + cid = str(cand["id"]) + path = f"candidates[{i}]" + questions[_qid(cid, "constraint_satisfaction")] = { + "type": "score", + "instructions": ( + f"How well does `{path}` satisfy constraints, given " + f"`{path}.feasible` and `{path}.constraint_violation` " + f"(g<=0 form; 0 means feasible)?" + ), + "criteria": list(CONSTRAINT_LEVELS), + } + questions[_qid(cid, "diversity_value")] = { + "type": "score", + "instructions": ( + f"How much unique coverage does `{path}` add relative to the other " + f"entries in `candidates`, looking at `{path}.objectives`?" + ), + "criteria": list(DIVERSITY_LEVELS), + } + questions[_qid(cid, "exploit_vs_explore")] = { + "type": "score", + "instructions": ( + f"On the exploit-vs-explore spectrum, where does `{path}` sit " + f"given `{path}.objectives` and the spread of `candidates`?" + ), + "criteria": list(EXPLOIT_LEVELS), + } + questions[_qid(cid, "disposition")] = { + "type": "choice", + "instructions": ( + f"Should `{path}` be kept, dropped, or reviewed as a shortlist " + f"member? Use feasibility, front rank if present, objective " + f"vectors, and redundancy versus the rest of `candidates`." + ), + "criteria": dict(DISPOSITION_CRITERIA), + } + return questions + + +def expected_question_ids(candidates: Sequence[Mapping[str, Any]]) -> list[str]: + ids: list[str] = [] + for cand in candidates: + cid = str(cand["id"]) + ids.extend( + [ + _qid(cid, "constraint_satisfaction"), + _qid(cid, "diversity_value"), + _qid(cid, "exploit_vs_explore"), + _qid(cid, "disposition"), + ] + ) + return ids + + +def _zdt1_pf_gap(objectives: Sequence[Any]) -> float | None: + if len(objectives) < 2: + return None + try: + f1 = float(objectives[0]) + f2 = float(objectives[1]) + except (TypeError, ValueError): + return None + if f1 < 0: + return None + pf = 1.0 - (f1 ** 0.5) + return max(0.0, f2 - pf) + + +def _nearest_objective_gap(index: int, candidates: Sequence[Mapping[str, Any]]) -> float: + obj = candidates[index].get("objectives") or [] + best = float("inf") + for j, other in enumerate(candidates): + if j == index: + continue + other_obj = other.get("objectives") or [] + n = min(len(obj), len(other_obj)) + if n == 0: + continue + dist = sum((float(obj[k]) - float(other_obj[k])) ** 2 for k in range(n)) ** 0.5 + best = min(best, dist) + return 0.0 if best is float("inf") else best + + +def _round_dist(dist: Mapping[str, float]) -> dict[str, float]: + rounded = {k: round(float(v), 6) for k, v in dist.items()} + if rounded: + peak_key = max(rounded, key=rounded.get) + rounded[peak_key] = round(rounded[peak_key] + (1.0 - sum(rounded.values())), 6) + return rounded + + +def _peaked(options: Sequence[str], winner: str, peak: float = 0.78) -> dict[str, float]: + rest = (1.0 - peak) / max(len(options) - 1, 1) + return _round_dist({opt: (peak if opt == winner else rest) for opt in options}) + + +def mock_system_one( + state: Mapping[str, Any], + questions: Mapping[str, Mapping[str, Any]], + *, + model: str = DEFAULT_MODEL, +) -> dict[str, Any]: + """Deterministic stand-in matching the documented System One response shape.""" + candidates = list(state.get("candidates") or []) + by_id = {str(c.get("id")): (i, c) for i, c in enumerate(candidates)} + answers: dict[str, Any] = {} + dimensions = ( + "constraint_satisfaction", + "diversity_value", + "exploit_vs_explore", + "disposition", + ) + + for qid, question in questions.items(): + qtype = question["type"] + dimension = qid + cand_id = qid + for dim in dimensions: + token = f"_{dim}" + if qid.endswith(token): + cand_id = qid[: -len(token)] + dimension = dim + break + if cand_id in by_id: + idx, cand = by_id[cand_id] + else: + idx = 0 + cand = candidates[0] if candidates else {} + + feasible = bool(cand.get("feasible", True)) + cv = float(cand.get("constraint_violation") or 0.0) + gap = _zdt1_pf_gap(cand.get("objectives") or []) + spread = _nearest_objective_gap(idx, candidates) + + if qtype == "score": + criteria = list(question["criteria"]) + if dimension == "constraint_satisfaction": + if not feasible or cv > 0.1: + level = 0 + elif cv > 0: + level = 1 + elif gap is not None and gap > 0.15: + level = 2 + else: + level = len(criteria) - 1 + elif dimension == "diversity_value": + if spread < 0.08: + level = 0 + elif spread < 0.25: + level = 1 + else: + level = 2 + elif dimension == "exploit_vs_explore": + objs = cand.get("objectives") or [0.5] + f1 = float(objs[0]) + if f1 <= 0.05 or f1 >= 0.95: + level = 2 + elif 0.35 <= f1 <= 0.65: + level = 0 + else: + level = 1 + else: + level = min(1, len(criteria) - 1) + level = max(0, min(level, len(criteria) - 1)) + peak = 0.8 + raw_probs = { + str(i): (peak if i == level else (1.0 - peak) / max(len(criteria) - 1, 1)) + for i in range(len(criteria)) + } + probs = _round_dist(raw_probs) + score = round(sum(int(k) * v for k, v in probs.items()), 6) + answers[qid] = { + "type": "score", + "score": score, + "legend": {str(i): criteria[i] for i in range(len(criteria))}, + "probabilities": probs, + "confidence": peak, + } + elif qtype == "choice": + options = list(question["criteria"].keys()) + if not feasible or cv > 0: + winner = "drop" if "drop" in options else options[-1] + elif gap is not None and gap > 0.15: + winner = "drop" if "drop" in options else options[-1] + elif spread < 0.08: + winner = "review" if "review" in options else options[0] + else: + winner = "keep" if "keep" in options else options[0] + answers[qid] = { + "type": "choice", + "choice": winner, + "probabilities": _peaked(options, winner), + "confidence": 0.78, + } + else: + raise ValueError(f"Unsupported question type in mock: {qtype}") + + return { + "model": model, + "answers": answers, + "usage": {"input_tokens": 0, "output_tokens": 0}, + } + + +def _systemone_url(base_url: str) -> str: + root = base_url.rstrip("/") + "/" + return urljoin(root, SYSTEMONE_PATH.lstrip("/")) + + +def _http_system_one( + state: Mapping[str, Any], + questions: Mapping[str, Mapping[str, Any]], + *, + model: str, + api_key: str, + base_url: str, + timeout: float, +) -> dict[str, Any]: + body = json.dumps({"state": state, "model": model, "questions": questions}).encode("utf-8") + request = urllib.request.Request( + _systemone_url(base_url), + data=body, + method="POST", + headers={ + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + "Accept": "application/json", + }, + ) + try: + with urllib.request.urlopen(request, timeout=timeout) as resp: + payload = json.loads(resp.read().decode("utf-8")) + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", errors="replace") + raise TypeSafeRequestError(f"TypeSafe HTTP {exc.code}: {detail[:500]}") from exc + except urllib.error.URLError as exc: + raise TypeSafeRequestError(f"TypeSafe connection failed: {exc.reason}") from exc + if not isinstance(payload, dict) or "answers" not in payload: + raise TypeSafeRequestError("TypeSafe response missing 'answers'.") + return payload + + +def _sdk_system_one( + state: Mapping[str, Any], + questions: Mapping[str, Mapping[str, Any]], + *, + model: str, + api_key: str, + base_url: str, + timeout: float, +) -> dict[str, Any]: + if TypeSafeClient is None: + raise TypeSafeRequestError("typesafe-sdk is not installed.") + with TypeSafeClient(api_key=api_key, model=model, base_url=base_url, timeout=timeout) as client: + result = client.system_one(state=state, questions=questions, model=model) + raw = getattr(result, "raw_http_response", None) + if raw is not None: + payload = raw.json() + if isinstance(payload, dict) and "answers" in payload: + return payload + usage = getattr(result, "usage", None) + answers: dict[str, Any] = {} + for key, answer in getattr(result, "answers", {}).items(): + answers[key] = _answer_to_dict(answer) + return { + "model": getattr(result, "model", model), + "answers": answers, + "usage": { + "input_tokens": getattr(usage, "input_tokens", None), + "output_tokens": getattr(usage, "output_tokens", None), + }, + } + + +def _answer_to_dict(answer: Any) -> dict[str, Any]: + if isinstance(answer, dict): + return dict(answer) + qtype = getattr(answer, "type", None) + if qtype == "score" or hasattr(answer, "score"): + legend = getattr(answer, "legend", {}) + probs = getattr(answer, "probabilities", {}) + return { + "type": "score", + "score": getattr(answer, "score"), + "legend": {str(k): v for k, v in dict(legend).items()}, + "probabilities": {str(k): v for k, v in dict(probs).items()}, + "confidence": getattr(answer, "confidence", None), + } + if qtype == "choice" or hasattr(answer, "choice"): + return { + "type": "choice", + "choice": getattr(answer, "choice"), + "probabilities": dict(getattr(answer, "probabilities", {})), + "confidence": getattr(answer, "confidence", None), + } + if qtype == "noul" or hasattr(answer, "noul"): + return {"type": "noul", "noul": getattr(answer, "noul")} + raise TypeError(f"Unrecognized TypeSafe answer: {answer!r}") + + +def resolve_api_key(explicit: str | None = None) -> str | None: + key = explicit if explicit is not None else os.environ.get(API_KEY_ENV) + if key is None: + return None + key = key.strip() + return key or None + + +def resolve_model(explicit: str | None = None) -> str: + if explicit: + return explicit + env = (os.environ.get(DEFAULT_MODEL_ENV) or "").strip() + return env or DEFAULT_MODEL + + +def resolve_base_url(explicit: str | None = None) -> str: + if explicit: + return explicit.rstrip("/") + env = (os.environ.get(BASE_URL_ENV) or "").strip() + return (env or DEFAULT_BASE_URL).rstrip("/") + + +def call_system_one( + state: Mapping[str, Any], + questions: Mapping[str, Mapping[str, Any]], + *, + model: str, + api_key: str | None, + base_url: str, + timeout: float = DEFAULT_TIMEOUT_S, + force_mock: bool = False, + http_post: Callable[..., dict[str, Any]] | None = None, +) -> tuple[dict[str, Any], str, float]: + """Return (response, mode, latency_ms). mode is 'mock', 'sdk', or 'http'.""" + started = time.perf_counter() + if force_mock or not api_key: + response = mock_system_one(state, questions, model=model) + mode = "mock" + elif http_post is not None: + response = http_post(state, questions, model=model, api_key=api_key, base_url=base_url, timeout=timeout) + mode = "http" + elif TypeSafeClient is not None: + response = _sdk_system_one( + state, questions, model=model, api_key=api_key, base_url=base_url, timeout=timeout + ) + mode = "sdk" + else: + response = _http_system_one( + state, questions, model=model, api_key=api_key, base_url=base_url, timeout=timeout + ) + mode = "http" + latency_ms = (time.perf_counter() - started) * 1000.0 + return response, mode, latency_ms + + +def rank_candidates( + candidates: Sequence[Mapping[str, Any]], + answers: Mapping[str, Mapping[str, Any]], +) -> list[dict[str, Any]]: + """Compose Score answers in code (weights live here, not in the prompt).""" + rows: list[dict[str, Any]] = [] + for cand in candidates: + cid = str(cand["id"]) + constraint = _score_value(answers.get(_qid(cid, "constraint_satisfaction"))) + diversity = _score_value(answers.get(_qid(cid, "diversity_value"))) + exploit = _score_value(answers.get(_qid(cid, "exploit_vs_explore"))) + disposition = answers.get(_qid(cid, "disposition")) or {} + # Normalize each Score onto [0, 1] using its own max level. + c_norm = _normalize_score(constraint, len(CONSTRAINT_LEVELS) - 1) + d_norm = _normalize_score(diversity, len(DIVERSITY_LEVELS) - 1) + e_norm = _normalize_score(exploit, len(EXPLOIT_LEVELS) - 1) + composite = 0.45 * c_norm + 0.35 * d_norm + 0.20 * e_norm + rows.append( + { + "id": cid, + "objectives": list(cand.get("objectives") or []), + "feasible": bool(cand.get("feasible", True)), + "constraint_satisfaction": constraint, + "diversity_value": diversity, + "exploit_vs_explore": exploit, + "disposition": disposition.get("choice"), + "disposition_confidence": disposition.get("confidence"), + "composite": composite, + } + ) + rows.sort(key=lambda r: (-r["composite"], r["id"])) + return rows + + +def _score_value(answer: Mapping[str, Any] | None) -> float | None: + if not answer: + return None + value = answer.get("score") + return float(value) if isinstance(value, (int, float)) else None + + +def _normalize_score(value: float | None, max_level: int) -> float: + if value is None or max_level <= 0: + return 0.0 + return max(0.0, min(1.0, float(value) / max_level)) + + +def metrics_record( + *, + model: str, + latency_ms: float, + usage: Mapping[str, Any] | None, + candidate_count: int, + answers: Mapping[str, Any], + notes: str, + extra: Mapping[str, Any] | None = None, +) -> dict[str, Any]: + record: dict[str, Any] = { + "ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "experiment": EXPERIMENT, + "repo": REPO, + "model": model, + "latency_ms": round(latency_ms, 3), + "usage": dict(usage) if usage else {"input_tokens": None, "output_tokens": None}, + "candidate_count": candidate_count, + "answers": dict(answers), + "notes": notes, + } + if extra: + record.update(extra) + return record + + +def append_metrics(path: Path, record: Mapping[str, Any]) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + line = json.dumps(record, separators=(",", ":"), ensure_ascii=False) + with path.open("a", encoding="utf-8") as fh: + fh.write(line + "\n") + return path + + +def format_ranking_table(rows: Sequence[Mapping[str, Any]]) -> str: + headers = ( + "id", + "f1", + "f2", + "feas", + "constraint", + "diversity", + "exploit", + "disposition", + "composite", + ) + lines = [" ".join(headers)] + for row in rows: + objs = row.get("objectives") or [] + f1 = f"{float(objs[0]):.4f}" if len(objs) > 0 else "-" + f2 = f"{float(objs[1]):.4f}" if len(objs) > 1 else "-" + c = row.get("constraint_satisfaction") + d = row.get("diversity_value") + e = row.get("exploit_vs_explore") + lines.append( + " ".join( + [ + str(row.get("id")), + f1, + f2, + "Y" if row.get("feasible") else "N", + "-" if c is None else f"{c:.2f}", + "-" if d is None else f"{d:.2f}", + "-" if e is None else f"{e:.2f}", + str(row.get("disposition") or "-"), + f"{float(row.get('composite') or 0.0):.3f}", + ] + ) + ) + return "\n".join(lines) + + +def run_pass( + payload: Mapping[str, Any], + *, + limit: int = SMOKE_LIMIT, + model: str | None = None, + api_key: str | None = None, + base_url: str | None = None, + timeout: float = DEFAULT_TIMEOUT_S, + force_mock: bool = False, + metrics_path: Path | None = DEFAULT_METRICS, + notes: str | None = None, + http_post: Callable[..., dict[str, Any]] | None = None, +) -> dict[str, Any]: + candidates = sample_candidates(payload, limit=limit) + state = build_state(payload, candidates) + questions = build_questions(candidates) + resolved_model = resolve_model(model) + resolved_key = resolve_api_key(api_key) + resolved_base = resolve_base_url(base_url) + response, mode, latency_ms = call_system_one( + state, + questions, + model=resolved_model, + api_key=resolved_key, + base_url=resolved_base, + timeout=timeout, + force_mock=force_mock, + http_post=http_post, + ) + answers = response.get("answers") or {} + rows = rank_candidates(candidates, answers) + note_bits = [notes] if notes else [] + note_bits.append(f"mode={mode}") + if mode == "mock": + note_bits.append("CI/mock System One; live call requires TYPESAFE_API_KEY") + if len(payload["candidates"]) > len(candidates): + note_bits.append(f"sampled {len(candidates)} of {len(payload['candidates'])}") + record = metrics_record( + model=response.get("model", resolved_model), + latency_ms=latency_ms, + usage=response.get("usage"), + candidate_count=len(candidates), + answers=answers, + notes="; ".join(note_bits), + extra={"mode": mode, "question_count": len(questions)}, + ) + written = None + if metrics_path is not None: + written = append_metrics(metrics_path, record) + return { + "state": state, + "questions": questions, + "response": response, + "ranking": rows, + "metrics": record, + "metrics_path": str(written) if written else None, + "mode": mode, + "latency_ms": latency_ms, + } + + +def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: + p = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + p.add_argument( + "--candidates", + type=Path, + default=None, + help="JSON export of candidates (default: fixture when --smoke)", + ) + p.add_argument("--smoke", action="store_true", help=f"Run fixture sample (≤{SMOKE_LIMIT} candidates)") + p.add_argument("--force-mock", action="store_true", help="Skip live API even if TYPESAFE_API_KEY is set") + p.add_argument("--limit", type=int, default=SMOKE_LIMIT) + p.add_argument("--model", default=None, help=f"System One model (default {DEFAULT_MODEL})") + p.add_argument("--metrics", type=Path, default=DEFAULT_METRICS) + p.add_argument("--no-metrics", action="store_true") + p.add_argument("--timeout", type=float, default=DEFAULT_TIMEOUT_S) + p.add_argument("--json", action="store_true", help="Print full result JSON") + return p.parse_args(argv) + + +def main(argv: Sequence[str] | None = None) -> int: + args = parse_args(argv) + if args.candidates is None: + if not args.smoke: + print("Pass --candidates PATH or --smoke.", file=sys.stderr) + return 2 + path = DEFAULT_FIXTURE + else: + path = args.candidates + if not path.is_file(): + print(f"Candidate file not found: {path}", file=sys.stderr) + return 2 + + payload = load_export(path) + metrics_path = None if args.no_metrics else args.metrics + result = run_pass( + payload, + limit=args.limit, + model=args.model, + force_mock=args.force_mock, + metrics_path=metrics_path, + notes="smoke fixture" if args.smoke or path == DEFAULT_FIXTURE else f"export {path.name}", + timeout=args.timeout, + ) + + print(f"TypeSafe Pareto pass | mode={result['mode']} model={result['metrics']['model']} " + f"candidates={result['metrics']['candidate_count']} questions={len(result['questions'])} " + f"latency_ms={result['latency_ms']:.1f}") + print() + print(format_ranking_table(result["ranking"])) + if result["metrics_path"]: + print() + print(f"metrics: {result['metrics_path']}") + if args.json: + print() + print(json.dumps({ + "mode": result["mode"], + "ranking": result["ranking"], + "metrics": result["metrics"], + "answers": result["response"].get("answers"), + }, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/typesafe-pareto/test_score_pareto.py b/tools/typesafe-pareto/test_score_pareto.py new file mode 100644 index 0000000..3470dac --- /dev/null +++ b/tools/typesafe-pareto/test_score_pareto.py @@ -0,0 +1,224 @@ +#!/usr/bin/env python3 +"""Mock tests for the TypeSafe Pareto helper. No network, no API keys.""" +from __future__ import annotations + +import json +import os +import sys +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +HERE = Path(__file__).resolve().parent +if str(HERE) not in sys.path: + sys.path.insert(0, str(HERE)) + +import score_pareto as sp + +FIXTURE = Path(__file__).resolve().parent / "fixtures" / "zdt1_candidates.json" + + +class ExportTests(unittest.TestCase): + def test_fixture_loads_eight_zdt1_like_candidates(self) -> None: + payload = sp.load_export(FIXTURE) + self.assertEqual(payload["problem"], "zdt1") + self.assertEqual(len(payload["candidates"]), 8) + ids = [c["id"] for c in payload["candidates"]] + self.assertEqual(ids, [f"c{i}" for i in range(8)]) + self.assertFalse(payload["candidates"][-1]["feasible"]) + + def test_raw_list_and_nondominated_alias(self) -> None: + listed = sp.normalize_export([{"objectives": [0.2, 0.6]}]) + self.assertEqual(listed["candidates"][0]["id"], "c0") + aliased = sp.normalize_export( + {"NonDominatedSolutions": [{"id": "front0", "objectives": [0.1, 0.7]}]} + ) + self.assertEqual(aliased["candidates"][0]["id"], "front0") + + def test_empty_export_rejected(self) -> None: + with self.assertRaises(ValueError): + sp.normalize_export({"candidates": []}) + + def test_sample_caps_at_ten(self) -> None: + fat = {"candidates": [{"id": f"x{i}", "objectives": [i / 20, 1 - i / 20]} for i in range(15)]} + sampled = sp.sample_candidates(fat, limit=sp.SMOKE_LIMIT) + self.assertEqual(len(sampled), 10) + self.assertEqual(sampled[0]["id"], "x0") + self.assertEqual(sampled[-1]["id"], "x9") + + +class QuestionTests(unittest.TestCase): + def test_fanout_is_one_map_four_questions_per_candidate(self) -> None: + payload = sp.load_export(FIXTURE) + candidates = sp.sample_candidates(payload) + questions = sp.build_questions(candidates) + self.assertEqual(len(questions), 4 * len(candidates)) + self.assertLessEqual(len(candidates), 10) + expected = sp.expected_question_ids(candidates) + self.assertEqual(sorted(questions), sorted(expected)) + first = questions["c0_constraint_satisfaction"] + self.assertEqual(first["type"], "score") + self.assertGreaterEqual(len(first["criteria"]), 2) + self.assertIn("candidates[0]", first["instructions"]) + choice = questions["c0_disposition"] + self.assertEqual(choice["type"], "choice") + self.assertEqual(set(choice["criteria"]), {"keep", "drop", "review"}) + + def test_state_marks_layer_as_additive(self) -> None: + payload = sp.load_export(FIXTURE) + state = sp.build_state(payload, payload["candidates"][:2]) + self.assertIn("does not replace", state["layer"].lower()) + self.assertEqual(len(state["candidates"]), 2) + self.assertEqual(state["repo"], "AppSprout-dev/Unsga3") + + +class MockClientTests(unittest.TestCase): + def test_mock_response_matches_documented_shapes(self) -> None: + payload = sp.load_export(FIXTURE) + candidates = payload["candidates"] + questions = sp.build_questions(candidates) + state = sp.build_state(payload, candidates) + response = sp.mock_system_one(state, questions, model="jev-latest") + self.assertEqual(response["model"], "jev-latest") + self.assertEqual(set(response["answers"]), set(questions)) + score = response["answers"]["c0_constraint_satisfaction"] + self.assertEqual(score["type"], "score") + self.assertIn("score", score) + self.assertIn("legend", score) + self.assertIn("probabilities", score) + self.assertIn("confidence", score) + self.assertAlmostEqual(sum(score["probabilities"].values()), 1.0, places=6) + choice = response["answers"]["c7_disposition"] + self.assertEqual(choice["type"], "choice") + self.assertEqual(choice["choice"], "drop") + self.assertAlmostEqual(sum(choice["probabilities"].values()), 1.0, places=6) + + def test_mock_drops_infeasible_and_off_front(self) -> None: + payload = sp.load_export(FIXTURE) + result = sp.run_pass(payload, force_mock=True, metrics_path=None) + by_id = {row["id"]: row for row in result["ranking"]} + self.assertEqual(by_id["c7"]["disposition"], "drop") + self.assertEqual(by_id["c6"]["disposition"], "drop") + self.assertEqual(by_id["c0"]["disposition"], "keep") + self.assertGreater(by_id["c0"]["composite"], by_id["c7"]["composite"]) + + def test_call_system_one_uses_mock_without_api_key(self) -> None: + payload = sp.load_export(FIXTURE) + candidates = payload["candidates"][:2] + state = sp.build_state(payload, candidates) + questions = sp.build_questions(candidates) + with mock.patch.dict(os.environ, {}, clear=False): + os.environ.pop(sp.API_KEY_ENV, None) + response, mode, latency_ms = sp.call_system_one( + state, questions, model="jev-latest", api_key=None, base_url=sp.DEFAULT_BASE_URL + ) + self.assertEqual(mode, "mock") + self.assertGreaterEqual(latency_ms, 0.0) + self.assertEqual(len(response["answers"]), 8) + + +class MetricsTests(unittest.TestCase): + def test_jsonl_append_schema(self) -> None: + payload = sp.load_export(FIXTURE) + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "typesafe-runs.jsonl" + first = sp.run_pass(payload, force_mock=True, metrics_path=path, notes="unit") + second = sp.run_pass(payload, force_mock=True, metrics_path=path, notes="unit") + lines = path.read_text(encoding="utf-8").strip().splitlines() + self.assertEqual(len(lines), 2) + rec = json.loads(lines[0]) + for key in ( + "ts", + "experiment", + "repo", + "model", + "latency_ms", + "usage", + "candidate_count", + "answers", + "notes", + ): + self.assertIn(key, rec) + self.assertEqual(rec["experiment"], "unsga3_pareto_score") + self.assertEqual(rec["repo"], "AppSprout-dev/Unsga3") + self.assertEqual(rec["candidate_count"], 8) + self.assertEqual(rec["model"], "jev-latest") + self.assertIn("mode=mock", rec["notes"]) + self.assertTrue(rec["answers"]) + self.assertEqual(first["metrics"]["candidate_count"], 8) + self.assertEqual(second["metrics_path"], str(path)) + + def test_default_metrics_path(self) -> None: + self.assertEqual(sp.default_metrics_path(), sp.REPO_ROOT / "metrics" / "typesafe-runs.jsonl") + + +class HttpClientTests(unittest.TestCase): + def test_injected_http_post_used_when_key_present(self) -> None: + payload = sp.load_export(FIXTURE) + captured: dict[str, object] = {} + + def fake_post(state, questions, *, model, api_key, base_url, timeout): + captured["n_questions"] = len(questions) + captured["api_key"] = api_key + captured["url_base"] = base_url + captured["state_n"] = len(state["candidates"]) + return { + "model": model, + "answers": sp.mock_system_one(state, questions, model=model)["answers"], + "usage": {"input_tokens": 12, "output_tokens": 4}, + } + + result = sp.run_pass( + payload, + api_key="test-not-a-real-key", + force_mock=False, + metrics_path=None, + http_post=fake_post, + ) + self.assertEqual(result["mode"], "http") + self.assertEqual(captured["n_questions"], 32) + self.assertEqual(captured["api_key"], "test-not-a-real-key") + self.assertEqual(captured["url_base"], "https://api.typesafe.ai") + self.assertEqual(result["metrics"]["usage"]["input_tokens"], 12) + + def test_systemone_url_matches_docs(self) -> None: + self.assertEqual( + sp._systemone_url("https://api.typesafe.ai"), + "https://api.typesafe.ai/v1/systemone", + ) + + def test_force_mock_wins_over_key(self) -> None: + payload = sp.load_export(FIXTURE) + + def boom(*_a, **_k): + raise AssertionError("live client must not run under --force-mock") + + result = sp.run_pass( + payload, + api_key="test-not-a-real-key", + force_mock=True, + metrics_path=None, + http_post=boom, + ) + self.assertEqual(result["mode"], "mock") + + +class CliTests(unittest.TestCase): + def test_smoke_cli_writes_metrics(self) -> None: + with tempfile.TemporaryDirectory() as tmp: + metrics = Path(tmp) / "typesafe-runs.jsonl" + with mock.patch("sys.stdout", new=mock.Mock()): + rc = sp.main(["--smoke", "--force-mock", "--metrics", str(metrics)]) + self.assertEqual(rc, 0) + self.assertTrue(metrics.is_file()) + rec = json.loads(metrics.read_text(encoding="utf-8").splitlines()[0]) + self.assertEqual(rec["experiment"], "unsga3_pareto_score") + self.assertLessEqual(rec["candidate_count"], 10) + + def test_missing_args_exits_2(self) -> None: + self.assertEqual(sp.main([]), 2) + + +if __name__ == "__main__": + unittest.main()