From f2c7776d80caafed634490087e8539542c679ec1 Mon Sep 17 00:00:00 2001 From: adibarra <93070681+adibarra@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:42:28 -0500 Subject: [PATCH 1/2] refactor(bench): replace benchmark_lib.sh with the infx.bench package Delete benchmarks/benchmark_lib.sh and move its behavior into the container-side Python package infx/bench (stdlib only, Python 3.10): wait, fixed-seq, agentic and eval commands behind one `python3 -m infx.bench` entrypoint, with shared process, readiness and input helpers. srt entry scripts become thin shims; recipe paths are unchanged. benchmarks/check_env.sh keeps check_env_vars for workflows and recipe setup scripts. AgentX single-concurrency validation moves from the workflow into LaunchRequest. Also drops code only the retired AMD/TileRT lanes used (server_watch, agentic --server-pid, the monitor command, fixed-seq point mode), the PROFILE trace relay, and dead topology aliases from the meta_env.json mapping. --- .agents/skills/debug-agentx-runs/SKILL.md | 2 +- .../workflows/benchmark-multinode-tmpl.yml | 17 +- .github/workflows/benchmark-tmpl.yml | 6 +- .github/workflows/collectivex-sweep.yml | 2 +- .github/workflows/operatorx-sweep.yml | 12 +- .github/workflows/profile.yml | 4 +- AGENTS.md | 6 +- inferencex-e2e/benchmarks/benchmark_lib.sh | 2718 ----------------- inferencex-e2e/benchmarks/check_env.sh | 24 + .../benchmarks/multi_node/runtime_settings.sh | 2 +- .../configs/glm5.3-tilert-rocm.sh | 2 +- .../configs/minimaxm3-trtllm-agentx.sh | 2 +- .../benchmarks/multi_node/srt_eval.sh | 62 +- .../multi_node/srt_fixed_sequence.sh | 64 +- .../speedbench/dsv4_fp4_b300_vllm.sh | 10 +- .../speedbench/dsv4dspark_fp4_b300_vllm.sh | 12 +- .../speedbench/glm52_fp4_b300_vllm.sh | 10 +- .../speedbench/kimik3_fp4_b300_vllm.sh | 10 +- ...le_method_block_rejection_sample_method.sh | 10 +- .../speedbench/minimaxm3_fp4_b300_vllm.sh | 10 +- .../speedbench/qwen3.5_fp4_b300_vllm.sh | 10 +- .../speedbench/qwen3.8next_fp4_b300_vllm.sh | 10 +- .../benchmarks/single_node/srt_eval.sh | 28 +- .../single_node/srt_fixed_sequence.sh | 67 +- inferencex-e2e/benchmarks/srt_agentic.sh | 94 +- inferencex-e2e/docs/DOCUMENTATION_PLAN.md | 2 +- inferencex-e2e/docs/DOCUMENTATION_PLAN_zh.md | 4 +- inferencex-e2e/docs/architecture.md | 6 +- inferencex-e2e/docs/architecture_zh.md | 6 +- .../docs/configuration-procedures.md | 15 +- .../docs/configuration-procedures_zh.md | 14 +- inferencex-e2e/docs/eval-agentx-procedures.md | 82 +- .../docs/eval-agentx-procedures_zh.md | 84 +- .../docs/recovery-results-procedures.md | 26 +- .../docs/recovery-results-procedures_zh.md | 23 +- inferencex-e2e/docs/testing.md | 3 +- inferencex-e2e/docs/testing_zh.md | 3 +- inferencex-e2e/docs/troubleshooting.md | 8 +- inferencex-e2e/docs/troubleshooting_zh.md | 8 +- inferencex-e2e/infx/bench/__init__.py | 1 + inferencex-e2e/infx/bench/__main__.py | 35 + inferencex-e2e/infx/bench/agentic/__init__.py | 1 + inferencex-e2e/infx/bench/agentic/replay.py | 193 ++ inferencex-e2e/infx/bench/agentic/run.py | 296 ++ inferencex-e2e/infx/bench/agentic/traces.py | 28 + inferencex-e2e/infx/bench/agentic/venv.py | 124 + inferencex-e2e/infx/bench/env.py | 57 + inferencex-e2e/infx/bench/eval/__init__.py | 211 ++ inferencex-e2e/infx/bench/eval/context.py | 35 + inferencex-e2e/infx/bench/eval/lm_eval.py | 146 + inferencex-e2e/infx/bench/eval/meta.py | 127 + inferencex-e2e/infx/bench/eval/stage.py | 43 + inferencex-e2e/infx/bench/eval/vendor.py | 360 +++ inferencex-e2e/infx/bench/fixed_seq.py | 186 ++ inferencex-e2e/infx/bench/gpu_monitor.py | 189 ++ inferencex-e2e/infx/bench/proc.py | 133 + inferencex-e2e/infx/bench/server.py | 134 + .../infx/bench_serving/server_watch.py | 155 - inferencex-e2e/infx/evals/EVALS.md | 204 +- .../infx/evals/_kimi_verifier_archive.py | 2 +- inferencex-e2e/infx/evals/bfcl_adapter.py | 85 +- inferencex-e2e/infx/evals/kimi_vendor_eval.py | 5 +- .../infx/evals/minimax_m3_full_eval.py | 5 +- .../infx/evals/minimax_provider_eval.py | 5 +- .../infx/launch/drivers/srt/collect.py | 32 +- inferencex-e2e/infx/launch/request.py | 22 +- .../infx/results/power/single_node.py | 2 +- inferencex-e2e/infx/ruff.toml | 2 + inferencex-e2e/infx/tests/bench/conftest.py | 48 + inferencex-e2e/infx/tests/bench/stubs.py | 12 + .../infx/tests/bench/test_agentic_command.py | 432 +++ .../infx/tests/bench/test_agentic_replay.py | 196 ++ .../infx/tests/bench/test_agentic_runtime.py | 106 + .../infx/tests/bench/test_eval_command.py | 418 +++ .../infx/tests/bench/test_eval_meta.py | 41 + .../infx/tests/bench/test_fixed_seq.py | 312 ++ .../infx/tests/bench/test_gpu_monitor.py | 176 ++ .../infx/tests/bench/test_server.py | 86 + .../infx/tests/bench/test_vendor_eval.py | 419 +++ .../tests/bench_serving/test_server_watch.py | 123 - .../infx/tests/evals/test_batched_eval.py | 114 +- .../infx/tests/evals/test_bfcl_eval.py | 56 + .../tests/evals/test_kimi_verifier_archive.py | 164 + .../tests/evals/test_run_eval_dispatch.py | 2572 ---------------- .../infx/tests/launch/fake_slurm.py | 7 +- .../infx/tests/launch/test_launch_request.py | 35 + .../infx/tests/launch/test_srt_driver.py | 27 +- .../results/agentic/test_power_lifecycle.py | 347 --- .../results/power/test_aggregate_power.py | 2 +- .../results/power/test_process_result.py | 320 -- .../srt_slurm/test_fixed_sequence_client.py | 66 - .../tests/workflows/test_launch_layout.py | 50 - .../srt-slurm/hooks/mi355x-amds/check-rdma.sh | 2 +- 93 files changed, 5320 insertions(+), 7107 deletions(-) delete mode 100644 inferencex-e2e/benchmarks/benchmark_lib.sh create mode 100644 inferencex-e2e/benchmarks/check_env.sh create mode 100644 inferencex-e2e/infx/bench/__init__.py create mode 100644 inferencex-e2e/infx/bench/__main__.py create mode 100644 inferencex-e2e/infx/bench/agentic/__init__.py create mode 100644 inferencex-e2e/infx/bench/agentic/replay.py create mode 100644 inferencex-e2e/infx/bench/agentic/run.py create mode 100644 inferencex-e2e/infx/bench/agentic/traces.py create mode 100644 inferencex-e2e/infx/bench/agentic/venv.py create mode 100644 inferencex-e2e/infx/bench/env.py create mode 100644 inferencex-e2e/infx/bench/eval/__init__.py create mode 100644 inferencex-e2e/infx/bench/eval/context.py create mode 100644 inferencex-e2e/infx/bench/eval/lm_eval.py create mode 100644 inferencex-e2e/infx/bench/eval/meta.py create mode 100644 inferencex-e2e/infx/bench/eval/stage.py create mode 100644 inferencex-e2e/infx/bench/eval/vendor.py create mode 100644 inferencex-e2e/infx/bench/fixed_seq.py create mode 100644 inferencex-e2e/infx/bench/gpu_monitor.py create mode 100644 inferencex-e2e/infx/bench/proc.py create mode 100644 inferencex-e2e/infx/bench/server.py delete mode 100644 inferencex-e2e/infx/bench_serving/server_watch.py create mode 100644 inferencex-e2e/infx/tests/bench/conftest.py create mode 100644 inferencex-e2e/infx/tests/bench/stubs.py create mode 100644 inferencex-e2e/infx/tests/bench/test_agentic_command.py create mode 100644 inferencex-e2e/infx/tests/bench/test_agentic_replay.py create mode 100644 inferencex-e2e/infx/tests/bench/test_agentic_runtime.py create mode 100644 inferencex-e2e/infx/tests/bench/test_eval_command.py create mode 100644 inferencex-e2e/infx/tests/bench/test_eval_meta.py create mode 100644 inferencex-e2e/infx/tests/bench/test_fixed_seq.py create mode 100644 inferencex-e2e/infx/tests/bench/test_gpu_monitor.py create mode 100644 inferencex-e2e/infx/tests/bench/test_server.py create mode 100644 inferencex-e2e/infx/tests/bench/test_vendor_eval.py delete mode 100644 inferencex-e2e/infx/tests/bench_serving/test_server_watch.py create mode 100644 inferencex-e2e/infx/tests/evals/test_kimi_verifier_archive.py delete mode 100644 inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py create mode 100644 inferencex-e2e/infx/tests/launch/test_launch_request.py delete mode 100644 inferencex-e2e/infx/tests/results/agentic/test_power_lifecycle.py delete mode 100644 inferencex-e2e/infx/tests/srt_slurm/test_fixed_sequence_client.py diff --git a/.agents/skills/debug-agentx-runs/SKILL.md b/.agents/skills/debug-agentx-runs/SKILL.md index a198659e7e..661829fc73 100644 --- a/.agents/skills/debug-agentx-runs/SKILL.md +++ b/.agents/skills/debug-agentx-runs/SKILL.md @@ -154,7 +154,7 @@ Use AgentX phase markers, not total Slurm runtime: ```bash grep -E \ - "Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|replay_rc=" \ + "Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|process_agentic_result" \ "/benchmark.out" date -u ``` diff --git a/.github/workflows/benchmark-multinode-tmpl.yml b/.github/workflows/benchmark-multinode-tmpl.yml index a9acc6bc7a..e82ee262cf 100644 --- a/.github/workflows/benchmark-multinode-tmpl.yml +++ b/.github/workflows/benchmark-multinode-tmpl.yml @@ -304,7 +304,6 @@ jobs: - name: Launch multi-node job script env: - VALIDATION_BENCHMARK_LIB: ${{ github.workspace }}/.result-tooling/inferencex-e2e/benchmarks/benchmark_lib.sh PREFILL_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).prefill.additional-settings) }} DECODE_ADDITIONAL_SETTINGS: ${{ toJSON(fromJSON(inputs.config).decode.additional-settings) }} RUNNER_NAME: ${{ runner.name }} @@ -335,18 +334,6 @@ jobs: while IFS= read -r -d '' setting; do export "$setting" done < <(jq -j '.[] + "\u0000"' <<< "$settings_json") - # Each AgentX throughput point owns a fresh server deployment. - # Fixed-sequence and graded eval jobs retain their batching semantics. - if [[ "$IS_AGENTIC" == 1 && "$EVAL_ONLY" != true ]]; then - source "$VALIDATION_BENCHMARK_LIB" --validation-only - check_env_vars CONC CONC_LIST - validate_agentic_concurrency "$CONC" - validate_agentic_concurrency "$CONC_LIST" - if [[ "$CONC" != "$CONC_LIST" ]]; then - echo "ERROR: AgentX CONC must match the single CONC_LIST value" >&2 - exit 1 - fi - fi # Resolve the workflow's documented automatic eval-concurrency selection. if [[ -z "$EVAL_CONC" ]]; then EVAL_CONC=$(python3 -c 'import os; print(max(map(int, os.environ["CONC_LIST"].split())))') @@ -397,7 +384,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON # Launcher stamp supplies the producer SHA unless the # power-producer-sha input is set as a manual override @@ -481,7 +468,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON expected_concs="${EVAL_CONC}" if [[ -z "${expected_concs}" ]]; then diff --git a/.github/workflows/benchmark-tmpl.yml b/.github/workflows/benchmark-tmpl.yml index 666c8f28d9..d025a711c7 100644 --- a/.github/workflows/benchmark-tmpl.yml +++ b/.github/workflows/benchmark-tmpl.yml @@ -318,7 +318,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON AIPERF_FAILED_REQUEST_THRESHOLD "$INFERENCEX_RESULTS_PYTHON" -P -m infx.results.agentic.validate_agentic_result \ results/aiperf_artifacts \ @@ -331,7 +331,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON if [ ! -f "$RESULT_FILENAME.json" ]; then echo "no raw result to process: $RESULT_FILENAME.json" >&2 @@ -440,7 +440,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON "$INFERENCEX_RESULTS_PYTHON" -P -m infx.evals.validate_scores diff --git a/.github/workflows/collectivex-sweep.yml b/.github/workflows/collectivex-sweep.yml index 77789fe46f..89ca1c9304 100644 --- a/.github/workflows/collectivex-sweep.yml +++ b/.github/workflows/collectivex-sweep.yml @@ -84,7 +84,7 @@ jobs: RUN_ATTEMPT: ${{ github.run_attempt }} run: | set -eo pipefail - source ../inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source ../inferencex-e2e/benchmarks/check_env.sh check_env_vars INPUT_SUITES INPUT_SWAP_PROFILE INPUT_BACKEND RUN_ID RUN_ATTEMPT GITHUB_OUTPUT args=(--suites "$INPUT_SUITES" --swap-profile "$INPUT_SWAP_PROFILE" --backend "$INPUT_BACKEND") [ -n "$INPUT_ONLY_SKU" ] && args+=(--only-sku "$INPUT_ONLY_SKU") diff --git a/.github/workflows/operatorx-sweep.yml b/.github/workflows/operatorx-sweep.yml index e8529bf7d7..125993d9ac 100644 --- a/.github/workflows/operatorx-sweep.yml +++ b/.github/workflows/operatorx-sweep.yml @@ -69,7 +69,7 @@ jobs: PYTHONPATH: .:inferencex-e2e run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars OPERATORX_POOL OPERATORX_BACKENDS OPERATORX_TESTLISTS OPERATORX_WORLD_SIZES OPERATORX_CHUNK_SIZE GITHUB_RUN_ID GITHUB_RUN_ATTEMPT GITHUB_SHA GITHUB_OUTPUT python3 -m operatorx.ci plan \ --platform-config operatorx/platforms.json \ @@ -141,7 +141,7 @@ jobs: OPERATORX_POOL: ${{ matrix.pool }} run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars RUNNER_TEMP OPERATORX_RECOVERY_RUN OPERATORX_POOL OPERATORX_CLEANUP_SECONDS python3 -m operatorx.ci recover --artifacts operatorx-recovery \ --run-id "$OPERATORX_RECOVERY_RUN" --pool "$OPERATORX_POOL" \ @@ -150,7 +150,7 @@ jobs: - name: Run one Slurm allocation run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars RUNNER_TEMP OPERATORX_SHARD OPERATORX_OUTPUT OPERATORX_CLEANUP_SECONDS GITHUB_RUN_ID GITHUB_RUN_ATTEMPT GITHUB_SHA RUNNER_NAME python3 -m operatorx.ci execute \ --platform-config operatorx/platforms.json \ @@ -163,7 +163,7 @@ jobs: if: always() run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars OPERATORX_OUTPUT OPERATORX_CLEANUP_SECONDS python3 -m operatorx.ci finalize --output "$OPERATORX_OUTPUT" \ --cleanup-seconds "$OPERATORX_CLEANUP_SECONDS" @@ -171,7 +171,7 @@ jobs: if: failure() run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars OPERATORX_OUTPUT if [ -d "$OPERATORX_OUTPUT" ]; then timeout 15 sinfo --Node --noheader --format='%N|%P|%t' \ @@ -210,7 +210,7 @@ jobs: PYTHONPATH: .:inferencex-e2e run: | set -eo pipefail - source inferencex-e2e/benchmarks/benchmark_lib.sh --validation-only + source inferencex-e2e/benchmarks/check_env.sh check_env_vars GITHUB_STEP_SUMMARY python3 -m operatorx.ci summarize \ --manifest operatorx-control/operatorx-manifest.json \ diff --git a/.github/workflows/profile.yml b/.github/workflows/profile.yml index c506bc02c9..494afabbb9 100644 --- a/.github/workflows/profile.yml +++ b/.github/workflows/profile.yml @@ -91,7 +91,7 @@ jobs: PRIORITY_ROOT="$PRIORITY_ROOT/inferencex-e2e" fi env PYTHONPATH="$PRIORITY_ROOT" python3 -P -m infx.workflows.require_launcher "$GITHUB_WORKSPACE" - source "$PRIORITY_ROOT/benchmarks/benchmark_lib.sh" --validation-only + source "$PRIORITY_ROOT/benchmarks/check_env.sh" check_env_vars INPUTS_CONFIG_FILE INPUTS_CONFIG_KEY INPUTS_CONC GENERATOR_PATH="${MEASURED_ROOT}" if [ -f "${MEASURED_ROOT}/infx/matrix/generate.py" ]; then @@ -350,7 +350,7 @@ jobs: PYTHONPATH: ${{ github.workspace }}/.result-tooling/inferencex-e2e run: | if [[ -f inferencex-e2e/configs/runners.yaml ]]; then cd inferencex-e2e; fi - source "$PYTHONPATH/benchmarks/benchmark_lib.sh" --validation-only + source "$PYTHONPATH/benchmarks/check_env.sh" check_env_vars INFERENCEX_RESULTS_PYTHON "$INFERENCEX_RESULTS_PYTHON" -P -m infx.results.fixed_sequence diff --git a/AGENTS.md b/AGENTS.md index eed52086fb..afee287cfa 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -22,7 +22,7 @@ The end-to-end Python project owns `inferencex-e2e/pyproject.toml`, `inferencex- - **Klaud Cold reports:** Follow the compact body/comment templates in [`inferencex-e2e/docs/klaud-reporting.md`](inferencex-e2e/docs/klaud-reporting.md), including cleanup and completion reports. - Commit subjects use conventional English style, while commit bodies include the Chinese translation. Contributor-facing docs use English as the source version and ship with a synchronized `_zh.md` page and language switcher. - Python under `inferencex-e2e/infx/` uses all stable Ruff rules with reviewed exclusions in `inferencex-e2e/infx/ruff.toml`, line length 100, and the Ruff formatter. The Lint job in `.github/workflows/ci.yml` runs whenever Python files change and fails on any finding. Before pushing Python changes, run the [commands in the testing guide](inferencex-e2e/docs/testing.md#python-lint-and-formatting). Fix findings where practical; justified exceptions use inline `# noqa: CODE` rather than file-wide ignores. -- Follow the nearest existing pattern. Python uses typed signatures and strict Pydantic schemas. YAML uses kebab-case fields. Shared benchmark Bash behavior belongs in `benchmark_lib.sh`, with parameters passed through environment variables. +- Follow the nearest existing pattern. Python uses typed signatures and strict Pydantic schemas. YAML uses kebab-case fields. Shared benchmark behavior that runs inside serving containers belongs in `inferencex-e2e/infx/bench/` as stdlib-only, Python 3.10 compatible commands (`python3 -m infx.bench `), with parameters passed through environment variables or flags. Bash entrypoints stay thin shims. ## Bash conventions (mandatory) @@ -30,7 +30,7 @@ These rules apply to active Bash scripts and shell commands embedded in workflow - **Configuration flows from the caller.** Workflows, master configs, runtime profiles, and launchers explicitly supply configuration to the scripts they invoke. Receiving scripts consume and validate those inputs; they must not silently choose defaults. - **No fallback defaults for caller-supplied configuration.** Avoid `${VAR:-default}`, `${VAR:=default}`, their colon-free equivalents, and equivalent "if unset, assign a default" logic. A missing input is a caller error and must fail clearly. Pass values such as `false` and `0` explicitly too. -- **Validate every required environment input with `check_env_vars` before use.** Use the shared helper in `inferencex-e2e/benchmarks/benchmark_lib.sh`. Group required inputs near the beginning, after sourcing the helper; validate inputs used only by a particular execution path when entering that path. The helper rejects both missing and empty values. Do not duplicate it or remove its safe handling of unset variables. Callers needing validation without benchmark initialization can source the library with `--validation-only`. +- **Validate every required environment input with `check_env_vars` before use.** Source the shared helper from `inferencex-e2e/benchmarks/check_env.sh`; sourcing it only defines the function. Group required inputs near the beginning, after sourcing the helper; validate inputs used only by a particular execution path when entering that path. The helper rejects both missing and empty values and lists every missing name. Do not duplicate it or remove its safe handling of unset variables. - **Do not enable nounset.** No `set -u`, `set -o nounset`, combined flags such as `set -euo pipefail`, or `bash -u` invocation flags. Use explicit validation; preserve other intended shell options, for example `set -eo pipefail`. - **Preserve configuration precedence and forwarding.** Apply caller-owned settings before recipe-specific overrides, and explicitly forward required inputs across container or job boundaries. Do not replace a supported override with an unconditional assignment in the receiving script. - Preserve deliberate optional-input handling, runtime-derived values, and unset-safe internal-state probes. These are not permission to invent fallback configuration or replace a documented automatic selection with an arbitrary constant. @@ -125,7 +125,7 @@ Deleting a test that fails these questions needs no replacement. Do not preserve - Every change that can affect benchmark performance and every recipe addition or modification requires a new `inferencex-e2e/perf-changelog.yaml` entry. The file is append-only and byte-sensitive. Preserve all existing bytes and separator whitespace, and append only at the tail. - New `inferencex-e2e/perf-changelog.yaml` entries must be English-only. Do not add Chinese translations or bilingual descriptions; the bilingual documentation and GitHub-content rules do not apply to these entries. Leave historical entries unchanged. - Multi-node srt-slurm changes update the recipe YAML and matching master config together. For image bumps, `model.container` must equal `image`. -- Every speculative fixed-sequence benchmark renders prompts with the chat template: single-node srt-slurm recipes that speculate set `benchmark.env.USE_CHAT_TEMPLATE: "true"` (enforced by `inferencex-e2e/infx/srt_slurm/single_node.py::validate_recipe`), which `srt_fixed_sequence.sh` turns into `--use-chat-template` for `run_benchmark_serving`. +- Every speculative fixed-sequence benchmark renders prompts with the chat template: single-node srt-slurm recipes that speculate set `benchmark.env.USE_CHAT_TEMPLATE: "true"` (enforced by `inferencex-e2e/infx/srt_slurm/single_node.py::validate_recipe`), which `python3 -m infx.bench fixed-seq` (run by `srt_fixed_sequence.sh`) turns into `--use-chat-template` for the benchmark client. - Benchmarks create no new directories under `/workspace`. Root containers must not leave root-owned files in shared AMD runner workspaces. - Generated configuration is not runtime proof. Run the narrowest local check, then the applicable smoke, sweep, or eval procedure from [`inferencex-e2e/docs/procedures.md`](inferencex-e2e/docs/procedures.md). diff --git a/inferencex-e2e/benchmarks/benchmark_lib.sh b/inferencex-e2e/benchmarks/benchmark_lib.sh deleted file mode 100644 index 0a155a65cf..0000000000 --- a/inferencex-e2e/benchmarks/benchmark_lib.sh +++ /dev/null @@ -1,2718 +0,0 @@ -#!/usr/bin/env bash - -# Usage: check_env_vars VAR1 VAR2 ...; exits 1 if any is unset. -check_env_vars() { - local missing_vars=() - - for var_name in "$@"; do - if [[ -z "${!var_name:-}" ]]; then - missing_vars+=("$var_name") - fi - done - - if [[ ${#missing_vars[@]} -gt 0 ]]; then - echo "Error: The following required environment variables are not set:" - for var in "${missing_vars[@]}"; do - echo " - $var" - done - exit 1 - fi -} - -validate_agentic_concurrency() { - if [[ $# -ne 1 || ! "$1" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: AgentX requires exactly one positive concurrency per server deployment; launch a fresh server for each concurrency." >&2 - return 1 - fi -} - -# Launchers may load only input validation, without benchmark initialization. -if [[ "${1-}" == "--validation-only" ]]; then - return 0 -fi - -# Keep Python bytecode out of the mounted workspace. Benchmark jobs often run as -# root inside containers, and root-owned cache directories break future checkout -# cleanup on self-hosted runners. -export PYTHONDONTWRITEBYTECODE=1 -export PYTHONPYCACHEPREFIX="${PYTHONPYCACHEPREFIX:-/tmp/inferencex-pycache}" -mkdir -p "$PYTHONPYCACHEPREFIX" 2>/dev/null || true -INFERENCEX_BENCHMARK_LIB_DIR="$( - cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -)" -INFERENCEX_REPO_ROOT="$( - cd "$INFERENCEX_BENCHMARK_LIB_DIR/.." && pwd -)" - -# Workflows supply PORT; launchers may select a cluster-specific port. - -# Agentic replays must use the model's native context limit. Ignore inherited -# workflow or shell overrides so neither the server nor AIPerf applies a cap. -_benchmark_caller="${BASH_SOURCE[1]:-}" -if [[ "$_benchmark_caller" == */agentic/* || - "$_benchmark_caller" == */agentic_*.sh || - "${IS_AGENTIC-}" == "1" || - "${SCENARIO_TYPE:-}" == "agentic-coding" ]]; then - unset MAX_MODEL_LEN - if [[ -z "${KV_OFFLOADING+x}" || -z "$KV_OFFLOADING" ]]; then - echo "Error: KV_OFFLOADING must be set for agentic benchmarks" >&2 - exit 1 - fi - case "$KV_OFFLOADING" in - none) - if [[ -n "${KV_OFFLOAD_BACKEND:-}" ]]; then - echo "Error: KV_OFFLOAD_BACKEND must be empty when KV_OFFLOADING=none" >&2 - exit 1 - fi - ;; - dram) - if [[ -z "${KV_OFFLOAD_BACKEND:-}" || "${KV_OFFLOAD_BACKEND:-}" == "none" ]]; then - echo "Error: KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=dram" >&2 - exit 1 - fi - if [[ ! "${TOTAL_CPU_DRAM_GB:-}" =~ ^[1-9][0-9]*$ ]]; then - echo "Error: DRAM KV offloading requires a positive configured TOTAL_CPU_DRAM_GB capacity" >&2 - exit 1 - fi - ;; - *) - echo "Error: unsupported KV_OFFLOADING value '$KV_OFFLOADING' (expected one of: none, dram)" >&2 - exit 1 - ;; - esac -fi -unset _benchmark_caller - -# GPU monitoring helpers - -GPU_MONITOR_PID="" -GPU_MONITOR_VENDOR="" -GPU_MONITOR_INTERVAL=1 -GPU_METRICS_CSV="${GPU_METRICS_CSV:-gpu_metrics.csv}" -NVIDIA_GPU_MONITOR_QUERY="timestamp,index,power.draw,temperature.gpu,clocks.current.sm,clocks.current.memory,utilization.gpu,utilization.memory" -export GPU_METRICS_CSV - -# Keep one AMD CSV header and forward each complete row immediately. Some awk -# implementations buffer pipe input even with fflush(), losing the final ticks -# when the monitor stops. -_filter_amd_smi_metrics() { - local line header_seen=false - while IFS= read -r line; do - if [[ "$line" == timestamp,* ]]; then - if [[ "$header_seen" == true ]]; then - continue - fi - header_seen=true - fi - if [[ "$header_seen" == true ]]; then - printf '%s\n' "$line" - fi - done -} - -# Background nvidia-smi/amd-smi sampler writing CSV. -# Usage: start_gpu_monitor [--output /path/to/output.csv] [--interval 1] -start_gpu_monitor() { - local output="$GPU_METRICS_CSV" - local interval=1 - - while [[ $# -gt 0 ]]; do - case $1 in - --output) output="$2"; shift 2 ;; - --interval) interval="$2"; shift 2 ;; - *) shift ;; - esac - done - - GPU_METRICS_CSV="$output" - GPU_MONITOR_INTERVAL="$interval" - export GPU_METRICS_CSV - - if command -v nvidia-smi &>/dev/null; then - GPU_MONITOR_VENDOR="nvidia" - if ! nvidia-smi --query-gpu=index,uuid,pci.bus_id,name,driver_version \ - --format=csv > "${output%.csv}_identity.csv" 2>/dev/null; then - rm -f "${output%.csv}_identity.csv" - echo "[GPU Monitor] Warning: NVIDIA identity sidecar failed" >&2 - fi - nvidia-smi --query-gpu="$NVIDIA_GPU_MONITOR_QUERY" \ - --format=csv -l "$interval" > "$output" 2>/dev/null & - GPU_MONITOR_PID=$! - echo "[GPU Monitor] Started NVIDIA (PID=$GPU_MONITOR_PID, interval=${interval}s, output=$output)" - elif command -v amd-smi &>/dev/null; then - GPU_MONITOR_VENDOR="amd" - # amd-smi is Python and block-buffers stdout; without PYTHONUNBUFFERED the - # trailing ticks were lost at kill (measured on MI355X). - PYTHONUNBUFFERED=1 amd-smi metric -p -c -t -u -w "$interval" --csv 2>/dev/null \ - | _filter_amd_smi_metrics > "$output" & - GPU_MONITOR_PID=$! - # Hardware energy-accumulator + identity snapshots; the end-side twin in - # stop_gpu_monitor lets auditors cross-check the integrated energy - # against the accumulator delta. - _write_amd_smi_sidecar "${output%.csv}_energy_start.csv" metric -E --csv - _write_amd_smi_sidecar "${output%.csv}_identity.json" static --json - echo "[GPU Monitor] Started AMD (PID=$GPU_MONITOR_PID, interval=${interval}s, output=$output)" - else - GPU_MONITOR_VENDOR="" - echo "[GPU Monitor] No GPU monitoring tool found (nvidia-smi or amd-smi), skipping" - return 0 - fi -} - -stop_gpu_monitor() { - if [[ -n "$GPU_MONITOR_PID" ]] && kill -0 "$GPU_MONITOR_PID" 2>/dev/null; then - # The stream must cover one sample past benchmark_end_time_unix for - # boundary interpolation. NVIDIA appends a one-shot sample below; amd-smi - # one-shot CSV has no timestamp column, so the AMD watch stream must emit - # final ticks before the kill. Two extra intervals because amd-smi stamps - # integer seconds: a tick in the same second as the window end still - # fails bracketing (MI355X: end=...153.325 vs last sample ...153.0). - if [[ "$GPU_MONITOR_VENDOR" == "amd" ]]; then - sleep $(( ${GPU_MONITOR_INTERVAL} + 2 )) - fi - kill "$GPU_MONITOR_PID" 2>/dev/null - wait "$GPU_MONITOR_PID" 2>/dev/null || true - case "$GPU_MONITOR_VENDOR" in - nvidia) - if _repair_truncated_gpu_metrics_tail; then - nvidia-smi --query-gpu="$NVIDIA_GPU_MONITOR_QUERY" \ - --format=csv,noheader >> "$GPU_METRICS_CSV" 2>/dev/null || - echo "[GPU Monitor] Warning: final NVIDIA sample failed" >&2 - fi - ;; - amd) - _repair_truncated_gpu_metrics_tail || true - _write_amd_smi_sidecar "${GPU_METRICS_CSV%.csv}_energy_end.csv" metric -E --csv - ;; - esac - echo "[GPU Monitor] Stopped (PID=$GPU_MONITOR_PID)" - if [[ -f "$GPU_METRICS_CSV" ]]; then - local lines - lines=$(wc -l < "$GPU_METRICS_CSV") - echo "[GPU Monitor] Collected $lines rows -> $GPU_METRICS_CSV" - fi - fi - GPU_MONITOR_PID="" - GPU_MONITOR_VENDOR="" -} - -# Drop a partial trailing row left behind when the monitor dies mid-write. -# Returns non-zero when a truncated row was detected but could not be removed. -_repair_truncated_gpu_metrics_tail() { - local repaired_metrics="${GPU_METRICS_CSV}.repair.$$" - if [[ -s "$GPU_METRICS_CSV" ]] && - ! tail -c 1 "$GPU_METRICS_CSV" | grep -q '^$'; then - if sed '$d' "$GPU_METRICS_CSV" > "$repaired_metrics" && - mv "$repaired_metrics" "$GPU_METRICS_CSV"; then - echo "[GPU Monitor] Dropped truncated trailing sample" - else - rm -f "$repaired_metrics" - echo "[GPU Monitor] Warning: could not repair truncated trailing sample" >&2 - return 1 - fi - fi - return 0 -} - -# Write one best-effort amd-smi snapshot; remove the file rather than keep a -# partial one when the invocation fails. -_write_amd_smi_sidecar() { - local out="$1" - shift - if ! amd-smi "$@" > "$out" 2>/dev/null; then - rm -f "$out" - echo "[GPU Monitor] Warning: amd-smi $1 sidecar failed" >&2 - fi -} - -# shellcheck source=runners/srt-slurm/hooks/common.sh -source "$(dirname "${BASH_SOURCE[0]}")/../runners/srt-slurm/hooks/common.sh" || return 1 - - -# Poll an HTTP endpoint while streaming the owning process log. -# Required: --endpoint, --log, --pid. A zero timeout waits indefinitely. -wait_for_ready() { - set +x - local endpoint="" - local process_log="" - local process_pid="" - local sleep_interval=5 - local timeout=0 - - while [[ $# -gt 0 ]]; do - case $1 in - --endpoint) - endpoint="$2" - shift 2 - ;; - --log) - process_log="$2" - shift 2 - ;; - --pid) - process_pid="$2" - shift 2 - ;; - --sleep-interval) - sleep_interval="$2" - shift 2 - ;; - --timeout) - timeout="$2" - shift 2 - ;; - *) - echo "Unknown parameter: $1" - return 1 - ;; - esac - done - - if [[ -z "$endpoint" ]]; then - echo "Error: --endpoint is required" - return 1 - fi - if [[ -z "$process_log" ]]; then - echo "Error: --log is required" - return 1 - fi - if [[ -z "$process_pid" ]]; then - echo "Error: --pid is required" - return 1 - fi - if [[ ! "$sleep_interval" =~ ^[1-9][0-9]*$ ]]; then - echo "Error: --sleep-interval must be a positive integer" - return 1 - fi - if [[ ! "$timeout" =~ ^[0-9]+$ ]]; then - echo "Error: --timeout must be a non-negative integer" - return 1 - fi - - local deadline=0 - if [[ "$timeout" -gt 0 ]]; then - deadline=$((SECONDS + timeout)) - fi - - while [[ ! -f "$process_log" ]]; do - if ! kill -0 "$process_pid" 2>/dev/null; then - echo "Process died before creating $process_log." >&2 - exit 1 - fi - if [[ "$deadline" -gt 0 && "$SECONDS" -ge "$deadline" ]]; then - echo "Timed out waiting for $endpoint." >&2 - exit 1 - fi - sleep 1 - done - - tail -f -n +1 "$process_log" & - local tail_pid=$! - until curl --output /dev/null --silent --fail "$endpoint"; do - if ! kill -0 "$process_pid" 2>/dev/null; then - echo "Process died before $endpoint became ready." >&2 - kill "$tail_pid" 2>/dev/null || true - exit 1 - fi - if [[ "$deadline" -gt 0 && "$SECONDS" -ge "$deadline" ]]; then - echo "Timed out waiting for $endpoint." >&2 - kill "$tail_pid" 2>/dev/null || true - exit 1 - fi - sleep "$sleep_interval" - done - kill "$tail_pid" 2>/dev/null || true - wait "$tail_pid" 2>/dev/null || true -} - -wait_for_server_ready() { - local port="" - local server_log="" - local server_pid="" - local sleep_interval=5 - - while [[ $# -gt 0 ]]; do - case $1 in - --port) port="$2"; shift 2 ;; - --server-log) server_log="$2"; shift 2 ;; - --server-pid) server_pid="$2"; shift 2 ;; - --sleep-interval) sleep_interval="$2"; shift 2 ;; - *) echo "Unknown parameter: $1"; return 1 ;; - esac - done - - if [[ -z "$port" || -z "$server_log" || -z "$server_pid" ]]; then - echo "Error: --port, --server-log, and --server-pid are required" - return 1 - fi - - wait_for_ready \ - --endpoint "http://0.0.0.0:${port}/health" \ - --log "$server_log" \ - --pid "$server_pid" \ - --sleep-interval "$sleep_interval" || return $? - INFERENCEX_SERVER_STATE=$(mktemp /tmp/inferencex-server-state.XXXXXX) || return 1 - PYTHONPATH="$INFERENCEX_REPO_ROOT${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.server_watch capture --pid "$server_pid" \ - > "$INFERENCEX_SERVER_STATE" || return 1 - INFERENCEX_SERVER_PID="$server_pid" -} - -# Keep client process groups separate; on confirmed server/worker death only -# this client's descendants are stopped. There is no elapsed-time cutoff. -run_server_client() { - if [[ -n "${INFERENCEX_SERVER_STATE:-}" ]]; then - PYTHONPATH="$INFERENCEX_REPO_ROOT${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.server_watch run --state "$INFERENCEX_SERVER_STATE" -- "$@" - else - "$@" - fi -} - -# --dsv4 renders prompts with the DeepSeek-V4 template (encoding_dsv4.py) instead -# of the tokenizer's jinja template and implies --use-chat-template. -run_benchmark_serving() { - if [ "${EVAL_ONLY}" = "true" ]; then - echo "EVAL_ONLY mode: skipping throughput benchmark" - return 0 - fi - - set +x - local model="" - local port="" - local backend="" - local base_url="" - local endpoint="" - local input_len="" - local output_len="" - local random_range_ratio="" - local num_prompts="" - local max_concurrency="" - local result_filename="" - local result_dir="" - local workspace_dir="" - local use_chat_template=false - local dsv4=false - local trust_remote_code=false - local server_pid="" - local tokenizer="" - local tokenizer_mode="" - - while [[ $# -gt 0 ]]; do - case $1 in - --model) - model="$2" - shift 2 - ;; - --port) - port="$2" - shift 2 - ;; - --backend) - backend="$2" - shift 2 - ;; - --endpoint) - endpoint="$2" - shift 2 - ;; - --base-url) - base_url="$2" - shift 2 - ;; - --input-len) - input_len="$2" - shift 2 - ;; - --output-len) - output_len="$2" - shift 2 - ;; - --random-range-ratio) - random_range_ratio="$2" - shift 2 - ;; - --num-prompts) - num_prompts="$2" - shift 2 - ;; - --max-concurrency) - max_concurrency="$2" - shift 2 - ;; - --result-filename) - result_filename="$2" - shift 2 - ;; - --result-dir) - result_dir="$2" - shift 2 - ;; - --bench-serving-dir) - workspace_dir="$2" - shift 2 - ;; - --use-chat-template) - use_chat_template=true - shift - ;; - --dsv4) - dsv4=true - use_chat_template=true - shift - ;; - --trust-remote-code) - trust_remote_code=true - shift - ;; - --server-pid) - server_pid="$2" - shift 2 - ;; - --tokenizer) - tokenizer="$2" - shift 2 - ;; - --tokenizer-mode) - tokenizer_mode="$2" - shift 2 - ;; - *) - echo "Unknown parameter: $1" - return 1 - ;; - esac - done - - if [[ -z "$model" ]]; then - echo "Error: --model is required" - return 1 - fi - if [[ -z "$port" ]]; then - echo "Error: --port is required" - return 1 - fi - if [[ -z "$backend" ]]; then - echo "Error: --backend is required" - return 1 - fi - if [[ -z "$input_len" ]]; then - echo "Error: --input-len is required" - return 1 - fi - if [[ -z "$output_len" ]]; then - echo "Error: --output-len is required" - return 1 - fi - if [[ -z "$random_range_ratio" ]]; then - echo "Error: --random-range-ratio is required" - return 1 - fi - if [[ -z "$num_prompts" ]]; then - echo "Error: --num-prompts is required" - return 1 - fi - if [[ -z "$max_concurrency" ]]; then - echo "Error: --max-concurrency is required" - return 1 - fi - if [[ -z "$result_filename" ]]; then - echo "Error: --result-filename is required" - return 1 - fi - if [[ -z "$result_dir" ]]; then - echo "Error: --result-dir is required" - return 1 - fi - - if [[ -z "$workspace_dir" ]]; then - workspace_dir=$(pwd) - fi - - # PROFILE=1 caps num_prompts at max_concurrency to keep traces small. - local profile_flag=() - if [[ "${PROFILE:-}" == "1" ]]; then - local _prof_dir="${SGLANG_TORCH_PROFILER_DIR:-${VLLM_TORCH_PROFILER_DIR:-}}" - if [[ -n "$_prof_dir" ]]; then - mkdir -p "$_prof_dir" - fi - profile_flag+=(--profile) - num_prompts="$max_concurrency" - fi - - if [[ -z "$base_url" ]]; then - base_url="http://0.0.0.0:$port" - fi - - local benchmark_cmd=( - env PYTHONPATH="$workspace_dir${PYTHONPATH:+:$PYTHONPATH}" - python3 -m infx.bench_serving.benchmark_serving - --model "$model" - --backend "$backend" - --base-url "$base_url" - --dataset-name random - --random-input-len "$input_len" - --random-output-len "$output_len" - --random-range-ratio "$random_range_ratio" - --num-prompts "$num_prompts" - --max-concurrency "$max_concurrency" - --request-rate inf - --ignore-eos - "${profile_flag[@]}" - --save-result - --num-warmups "$((2 * max_concurrency))" \ - --percentile-metrics 'ttft,tpot,itl,e2el' - --result-dir "$result_dir" - --result-filename "$result_filename.json" - ) - - if [[ -n "$endpoint" ]]; then - benchmark_cmd+=(--endpoint "$endpoint") - fi - - if [[ "$use_chat_template" == true ]]; then - benchmark_cmd+=(--use-chat-template) - fi - - if [[ "$dsv4" == true ]]; then - benchmark_cmd+=(--dsv4) - fi - - if [[ "$trust_remote_code" == true ]]; then - benchmark_cmd+=(--trust-remote-code) - fi - - if [[ -n "$tokenizer" ]]; then - benchmark_cmd+=(--tokenizer "$tokenizer") - fi - - if [[ -n "$tokenizer_mode" ]]; then - benchmark_cmd+=(--tokenizer-mode "$tokenizer_mode") - fi - - if [[ -n "$server_pid" && "$server_pid" != "${INFERENCEX_SERVER_PID:-}" ]]; then - INFERENCEX_SERVER_STATE=$(mktemp /tmp/inferencex-server-state.XXXXXX) || return 1 - PYTHONPATH="$INFERENCEX_REPO_ROOT${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.server_watch capture --pid "$server_pid" \ - > "$INFERENCEX_SERVER_STATE" || return 1 - INFERENCEX_SERVER_PID="$server_pid" - fi - local benchmark_exit_code=0 - set -x - run_server_client "${benchmark_cmd[@]}" || benchmark_exit_code=$? - set +x - - if [[ "${PROFILE:-}" == "1" ]]; then - move_profile_trace_for_relay - fi - - return $benchmark_exit_code -} - - -# Profiling trace helpers - -_find_latest_profile_trace() { - local latest="" - local dir="" candidate="" base="" - local -a search_roots=() - - for dir in "$@"; do - search_roots=() - if [[ -d "$dir" ]]; then - search_roots+=("$dir") - fi - if [[ -d "$dir/profiles" ]]; then - search_roots+=("$dir/profiles") - fi - if [[ ${#search_roots[@]} -eq 0 ]]; then - continue - fi - - while IFS= read -r -d '' candidate; do - base="$(basename "$candidate")" - if [[ "$base" == profile_*.trace.json.gz ]]; then - continue - fi - if [[ -z "$latest" || "$candidate" -nt "$latest" ]]; then - latest="$candidate" - fi - done < <( - find "${search_roots[@]}" -maxdepth 1 -type f \ - \( -name "*.trace.json" -o -name "*.trace.json.gz" -o -name "*trace*.json" -o -name "*trace*.json.gz" -o -name "*profile*.json" -o -name "*profile*.json.gz" \) \ - -print0 2>/dev/null - ) - done - - printf '%s' "$latest" -} - -# Move profiler trace into a stable workspace path for workflow relay/upload. -move_profile_trace_for_relay() { - if [[ "${PROFILE:-}" != "1" ]]; then - return 0 - fi - - if [[ -z "${RESULT_FILENAME:-}" ]]; then - echo "[PROFILE] RESULT_FILENAME is not set; skipping relay trace staging." >&2 - return 0 - fi - - check_env_vars SGLANG_TORCH_PROFILER_DIR VLLM_TORCH_PROFILER_DIR - local sglang_dir="${SGLANG_TORCH_PROFILER_DIR}" - local vllm_dir="${VLLM_TORCH_PROFILER_DIR}" - local -a search_dirs=() - local dir="" existing="" - local seen=0 - - for dir in "$sglang_dir" "$vllm_dir" "/workspace"; do - if [[ -z "$dir" ]]; then - continue - fi - seen=0 - for existing in "${search_dirs[@]}"; do - if [[ "$existing" == "$dir" ]]; then - seen=1 - break - fi - done - if [[ "$seen" -eq 0 ]]; then - search_dirs+=("$dir") - fi - done - - local trace_file="" - local wait_attempts=10 - for (( i=1; i<=wait_attempts; i++ )); do - trace_file="$(_find_latest_profile_trace "${search_dirs[@]}")" - if [[ -n "$trace_file" ]]; then - break - fi - sleep 10 - done - - if [[ -z "$trace_file" ]]; then - echo "[PROFILE] No trace found for relay under: ${search_dirs[*]}" >&2 - return 0 - fi - - local dest_trace="/workspace/profile_${RESULT_FILENAME}.trace.json.gz" - if [[ "$trace_file" == *.gz ]]; then - cp -f "$trace_file" "$dest_trace" - else - gzip -c "$trace_file" > "$dest_trace" - fi - - echo "[PROFILE] Relay trace prepared: $dest_trace (source: $trace_file)" -} - - -# Eval (lm-eval-harness) helpers - -_install_lm_eval_deps() { - # torchvision causes circular imports in ATOM; TRT-LLM/SGLang need it at module level. - if [[ "${IMAGE:-}" == *atom* ]]; then - python3 -m pip uninstall -y torchvision 2>/dev/null || true - fi - python3 -m pip install -q --no-cache-dir --break-system-packages "lm-eval[api]" || true - local lm_eval_ref="b315ef3b05176acc9732bb7fdec116abe1ecc476" - if command -v git >/dev/null 2>&1; then - if ! python3 -m pip install -q --no-cache-dir --no-deps --force-reinstall --break-system-packages \ - "git+https://github.com/EleutherAI/lm-evaluation-harness.git@${lm_eval_ref}"; then - python3 -m pip install -q --no-cache-dir --no-deps --force-reinstall --break-system-packages \ - "https://github.com/EleutherAI/lm-evaluation-harness/archive/${lm_eval_ref}.tar.gz" || true - fi - else - python3 -m pip install -q --no-cache-dir --no-deps --force-reinstall --break-system-packages \ - "https://github.com/EleutherAI/lm-evaluation-harness/archive/${lm_eval_ref}.tar.gz" || true - fi -} - -_prepare_vendor_verifier_python() { - local verifier_name="$1" - local runtime_prefix="$2" - local use_system_site_packages="${3:-false}" - local minimum_python_minor="${4:-12}" - - VENDOR_VERIFIER_PYTHON=python3 - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR - - local system_python_is_compatible=false - if python3 -c \ - 'import sys; raise SystemExit(sys.version_info < (3, int(sys.argv[1])))' \ - "$minimum_python_minor"; then - system_python_is_compatible=true - if [ "$use_system_site_packages" != "true" ]; then - return 0 - fi - fi - - local python_dir uv_prefix uv_bin venv_dir prepare_rc=0 - python_dir="$(mktemp -d "/tmp/${runtime_prefix}-XXXXXX")" || { - echo "ERROR: could not create a temporary Python directory for ${verifier_name}" >&2 - return 1 - } - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$python_dir" - export VENDOR_VERIFIER_PYTHON_CLEANUP_DIR - - venv_dir="${python_dir}/venv" - if [ "$system_python_is_compatible" = "true" ]; then - python3 -m venv --system-site-packages "$venv_dir" || prepare_rc=$? - else - uv_prefix="${python_dir}/uv" - uv_bin="${uv_prefix}/bin/uv" - python3 -m pip install -q --no-cache-dir --break-system-packages \ - --prefix "$uv_prefix" "uv==0.11.33" || prepare_rc=$? - if [ "$prepare_rc" -eq 0 ] && [ ! -x "$uv_bin" ]; then - echo "ERROR: pinned uv installation did not create ${uv_bin}" >&2 - prepare_rc=1 - fi - if [ "$prepare_rc" -eq 0 ]; then - local system_site_packages_args=() - if [ "$use_system_site_packages" = "true" ]; then - system_site_packages_args+=(--system-site-packages) - fi - UV_CACHE_DIR="${python_dir}/uv-cache" \ - UV_PYTHON_INSTALL_DIR="${python_dir}/python" \ - "$uv_bin" venv --python "3.${minimum_python_minor}" --seed \ - "${system_site_packages_args[@]}" "$venv_dir" \ - || prepare_rc=$? - fi - fi - if [ "$prepare_rc" -eq 0 ] && [ ! -x "${venv_dir}/bin/python" ]; then - echo "ERROR: pinned Python setup did not create the ${verifier_name} interpreter" >&2 - prepare_rc=1 - fi - if [ "$prepare_rc" -ne 0 ]; then - rm -rf "$python_dir" || true - VENDOR_VERIFIER_PYTHON=python3 - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR - return "$prepare_rc" - fi - - VENDOR_VERIFIER_PYTHON="${venv_dir}/bin/python" - export VENDOR_VERIFIER_PYTHON -} - -_install_kimi_vendor_eval_deps() { - check_env_vars VENDOR_VERIFIER_PYTHON - local target_dir="$1" - local eval_suite="${2:-kimi_tool_call_schema}" - local -a packages=( - "httpx[http2]==0.28.1" - "openai==2.14.0" - "jsonschema==4.25.1" - "pytest==8.4.2" - ) - if [ "$eval_suite" = "kimi_tool_call_schema_full" ]; then - packages+=("pytest-xdist==3.8.0") - fi - "${VENDOR_VERIFIER_PYTHON}" -m pip install -q --no-cache-dir \ - --target "$target_dir" "${packages[@]}" -} - -_prepare_kimi_vendor_runtime() { - local eval_suite="${1:-kimi_tool_call_schema}" - local runtime_dir install_rc=0 - runtime_dir="$(mktemp -d /tmp/kimi-vendor-runtime-XXXXXX)" || return $? - _install_kimi_vendor_eval_deps "$runtime_dir" "$eval_suite" >&2 || install_rc=$? - if [ "$install_rc" -ne 0 ]; then - rm -rf "$runtime_dir" - return "$install_rc" - fi - printf '%s\n' "$runtime_dir" -} - -_prepare_kimi_vendor_verifier() { - check_env_vars VENDOR_VERIFIER_PYTHON - local repo_url="$1" - local verifier_ref="$2" - local expected_archive_sha256="$3" - local checkout_dir prepare_rc=0 - - checkout_dir="$(mktemp -d /tmp/kimi-vendor-verifier-XXXXXX)" || { - echo "ERROR: could not create a temporary directory for Kimi-Vendor-Verifier" >&2 - return 1 - } - - "${VENDOR_VERIFIER_PYTHON}" - \ - "$repo_url" "$verifier_ref" "$expected_archive_sha256" "$checkout_dir" \ - < "${INFERENCEX_REPO_ROOT}/infx/evals/_kimi_verifier_archive.py" || prepare_rc=$? - - if [ "$prepare_rc" -ne 0 ]; then - if ! rm -rf "$checkout_dir"; then - echo "ERROR: failed to remove partial Kimi-Vendor-Verifier directory ${checkout_dir}" >&2 - fi - return "$prepare_rc" - fi - - printf '%s\n' "$checkout_dir" -} - -_cleanup_vendor_eval() { - local path - for path in "$@"; do - [ -z "$path" ] || rm -rf "$path" || true - done -} - -_has_eval_result() { - local results_dir="$1" - local filename_prefix="$2" - local matches=("${results_dir}/${filename_prefix}"*.json) - [ -f "${matches[0]}" ] -} - -_prepare_eval_artifact_family() { - local results_dir="$1" - local family="$2" - local artifact rm_rc=0 - local artifacts=() - - export EVAL_RESULT_DIR="" - case "$family" in - kimi) - artifacts=( - "${results_dir}"/results_kimi_vendor_*.json - "${results_dir}/kimi_vendor_report.json" - ) - ;; - minimax) - artifacts=( - "${results_dir}"/results_minimax_vendor_*.json - "${results_dir}/minimax_vendor_report.json" - "${results_dir}/minimax_vendor_results.jsonl" - ) - ;; - bfcl) - artifacts=( - "${results_dir}"/results_bfcl*.json - "${results_dir}/bfcl_report.json" - "${results_dir}/bfcl_upstream_artifacts.tar.gz" - ) - ;; - *) - echo "ERROR: unsupported eval artifact family '${family}'" >&2 - return 2 - ;; - esac - - for artifact in "${artifacts[@]}"; do - if [ -e "$artifact" ] || [ -L "$artifact" ]; then - rm -f -- "$artifact" || rm_rc=$? - if [ "$rm_rc" -ne 0 ]; then - echo "ERROR: failed to remove stale eval artifact ${artifact}" >&2 - return "$rm_rc" - fi - fi - done - export EVAL_RESULT_DIR="$results_dir" -} - -_write_kimi_vendor_integration_error() { - check_env_vars VENDOR_VERIFIER_PYTHON - local adapter_path="$1" - local model_name="$2" - local results_dir="$3" - local task_name="$4" - local message="$5" - - "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" \ - --model "$model_name" \ - --output-dir "$results_dir" \ - --task-name "$task_name" \ - --integration-error "$message" -} - -_run_kimi_tool_call_schema_eval() { - check_env_vars PORT - local port="${PORT}" - local results_dir="${EVAL_RESULT_DIR:-$(mktemp -d /tmp/eval_out-XXXXXX)}" - local verifier_repo="https://github.com/MoonshotAI/Kimi-Vendor-Verifier.git" - local verifier_ref="3dad65a760a8867cda72f6dd8848d876a4e851b4" - local verifier_archive_sha256="ede9ea300c72ccfde9d8975ea4b1b54e423c7625690f6631ab1e65a715821e01" - local eval_suite="${EVAL_SUITE:-kimi_tool_call_schema}" - - while [[ $# -gt 0 ]]; do - case "$1" in - --port|--results-dir) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: $1 requires a value" >&2 - return 2 - fi - case "$1" in - --port) port="$2" ;; - --results-dir) results_dir="$2" ;; - esac - shift 2 - ;; - *) - echo "Unknown parameter: $1" >&2 - return 2 - ;; - esac - done - - - local model_name="${MODEL_NAME:-${MODEL:-}}" - local adapter_path="${INFERENCEX_REPO_ROOT}/infx/evals/kimi_vendor_eval.py" - local runtime_dir="" - local checkout_dir="" - - mkdir -p "$results_dir" || return $? - results_dir="$(cd "$results_dir" && pwd)" || return $? - _prepare_eval_artifact_family "$results_dir" kimi || return $? - - local setup_rc=0 integration_error="" - _prepare_vendor_verifier_python "Kimi-Vendor-Verifier" "kimi-vendor-python" || { - setup_rc=$? - integration_error="Kimi Vendor Verifier Python runtime preparation failed with exit code ${setup_rc}" - } - if [ "$setup_rc" -eq 0 ]; then - runtime_dir=$(_prepare_kimi_vendor_runtime "$eval_suite") || { - setup_rc=$? - integration_error="Kimi Vendor Verifier dependency installation failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -eq 0 ]; then - checkout_dir=$( - _prepare_kimi_vendor_verifier \ - "$verifier_repo" "$verifier_ref" "$verifier_archive_sha256" - ) || { - setup_rc=$? - integration_error="Kimi Vendor Verifier checkout failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -ne 0 ]; then - echo "ERROR: ${integration_error}" >&2 - local artifact_rc=0 - _write_kimi_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$eval_suite" \ - "$integration_error" || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write Kimi verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - _cleanup_vendor_eval \ - "$runtime_dir" "$checkout_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$setup_rc" - fi - - local eval_rc=0 - PYTHONPATH="${runtime_dir}${PYTHONPATH:+:${PYTHONPATH}}" \ - run_server_client "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" \ - --verifier-dir "$checkout_dir" \ - --base-url "http://127.0.0.1:${port}/v1" \ - --api-key EMPTY \ - --model "$model_name" \ - --model-prefix "${MODEL_PREFIX:-}" \ - --output-dir "$results_dir" \ - --task-name "$eval_suite" \ - || eval_rc=$? - if [ "$eval_rc" -ne 0 ] \ - && ! _has_eval_result "$results_dir" "results_kimi_vendor_"; then - integration_error="Kimi Vendor Verifier failed with exit code ${eval_rc}" - local artifact_rc=0 - _write_kimi_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$eval_suite" \ - "$integration_error" || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write Kimi verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - fi - _cleanup_vendor_eval \ - "$runtime_dir" "$checkout_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$eval_rc" -} - -run_kimi_vendor_eval() { - local eval_suite="${EVAL_SUITE:-kimi_tool_call_schema}" - export EVAL_SUITE="$eval_suite" - - case "$eval_suite" in - kimi_tool_call_schema|kimi_tool_call_schema_full) - _run_kimi_tool_call_schema_eval "$@" - ;; - *) - echo "ERROR: unsupported Kimi Vendor Verifier suite '${eval_suite}'" >&2 - export EVAL_RESULT_DIR="" - return 2 - ;; - esac -} - -_install_bfcl_eval_deps() { - check_env_vars VENDOR_VERIFIER_PYTHON - local download_dir="$1" - local wheel_url="https://files.pythonhosted.org/packages/ba/41/ed458527c770c50225b60bae3b0c3444b26804ee455fa2d8f187018d2cb2/bfcl_eval-2026.3.23-py3-none-any.whl" - local wheel_sha256="3bb6dfa5f0c68ad403c9ec50b00db2bb3b4cc9b38ab1ff33f48fe30d853d3a0a" - local wheel_path="${download_dir}/bfcl_eval-2026.3.23-py3-none-any.whl" - - "${VENDOR_VERIFIER_PYTHON}" - \ - "$wheel_url" "$wheel_sha256" "$wheel_path" <<'PY' || return $? -from hashlib import sha256 -from pathlib import Path -import sys -from urllib.request import Request, urlopen - -wheel_url, expected_sha256, wheel_path_arg = sys.argv[1:] -wheel_path = Path(wheel_path_arg) -digest = sha256() -downloaded = 0 -request = Request( - wheel_url, - headers={"User-Agent": "InferenceX-BFCL-Smoke"}, -) - -try: - with urlopen(request, timeout=180) as response, wheel_path.open("xb") as output: - while chunk := response.read(1024 * 1024): - downloaded += len(chunk) - if downloaded > 512 * 1024 * 1024: - raise ValueError("BFCL wheel exceeds the 512 MiB safety limit") - digest.update(chunk) - output.write(chunk) - if downloaded == 0: - raise ValueError("downloaded BFCL wheel is empty") - actual_sha256 = digest.hexdigest() - if actual_sha256 != expected_sha256: - raise ValueError( - f"BFCL wheel SHA256 mismatch: expected {expected_sha256}, " - f"got {actual_sha256}" - ) -except Exception as error: - wheel_path.unlink(missing_ok=True) - print(f"ERROR: failed to download and verify the pinned BFCL wheel: {error}", file=sys.stderr) - raise SystemExit(1) -PY - - timeout 600 "${VENDOR_VERIFIER_PYTHON}" -m pip install \ - -q --no-cache-dir "$wheel_path" "soundfile==0.13.1" -} - -_prepare_bfcl_runtime() { - local runtime_dir install_rc=0 - runtime_dir="$(mktemp -d /tmp/bfcl-runtime-XXXXXX)" || return $? - _install_bfcl_eval_deps "$runtime_dir" >&2 || install_rc=$? - if [ "$install_rc" -ne 0 ]; then - rm -rf "$runtime_dir" - return "$install_rc" - fi - printf '%s\n' "$runtime_dir" -} - -_archive_bfcl_upstream_artifacts() { - check_env_vars VENDOR_VERIFIER_PYTHON - local project_root="$1" - local archive_path="$2" - - "${VENDOR_VERIFIER_PYTHON}" - "$project_root" "$archive_path" <<'PY' -import gzip -import os -from pathlib import Path -import tarfile -import sys - -project_root = Path(sys.argv[1]) -archive_path = Path(sys.argv[2]) -temporary_path = archive_path.with_name(f".{archive_path.name}.tmp") -temporary_path.unlink(missing_ok=True) - -try: - with ( - temporary_path.open("xb") as raw_archive, - gzip.GzipFile(filename="", mode="wb", fileobj=raw_archive, mtime=0) as compressed, - tarfile.open(fileobj=compressed, mode="w", format=tarfile.PAX_FORMAT) as archive, - ): - for path in sorted( - project_root.rglob("*"), - key=lambda candidate: candidate.relative_to(project_root).as_posix(), - ): - relative_path = path.relative_to(project_root).as_posix() - if path.is_symlink(): - raise ValueError(f"refusing to archive symbolic link: {relative_path}") - info = archive.gettarinfo(str(path), arcname=relative_path) - info.uid = 0 - info.gid = 0 - info.uname = "" - info.gname = "" - info.mtime = 0 - if info.isdir(): - archive.addfile(info) - elif info.isfile(): - with path.open("rb") as source: - archive.addfile(info, source) - else: - raise ValueError(f"refusing to archive special file: {relative_path}") - os.replace(temporary_path, archive_path) -except BaseException: - temporary_path.unlink(missing_ok=True) - raise -PY -} - -_write_bfcl_integration_error() { - check_env_vars VENDOR_VERIFIER_PYTHON - local adapter_path="$1" - local model_name="$2" - local results_dir="$3" - local message="$4" - local suite="$5" - local adapter_rc=0 - - # Integration errors deliberately make the adapter exit nonzero after - # publishing both score artifacts. Treat those artifacts, not that expected - # status, as proof that failure reporting succeeded. - "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" \ - --model "$model_name" \ - --output-dir "$results_dir" \ - --suite "$suite" \ - --integration-error "$message" \ - || adapter_rc=$? - if [ -f "${results_dir}/bfcl_report.json" ] \ - && [ -f "${results_dir}/results_bfcl.json" ]; then - return 0 - fi - if [ "$adapter_rc" -eq 0 ]; then - return 1 - fi - return "$adapter_rc" -} - -_run_bfcl_suite_eval() { - check_env_vars PORT - local eval_suite="$1" - local num_threads="$2" - local process_timeout_seconds="$3" - local archive_upstream="$4" - shift 4 - - local port="${PORT}" - local results_dir="${EVAL_RESULT_DIR:-}" - - while [[ $# -gt 0 ]]; do - case "$1" in - --port|--results-dir) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: $1 requires a value" >&2 - return 2 - fi - case "$1" in - --port) port="$2" ;; - --results-dir) results_dir="$2" ;; - esac - shift 2 - ;; - *) - echo "Unknown parameter: $1" >&2 - return 2 - ;; - esac - done - - if [ -z "$results_dir" ]; then - results_dir="$(mktemp -d /tmp/eval_out-XXXXXX)" || return $? - fi - - local model_name="${MODEL_NAME:-${MODEL:-}}" - local adapter_path="${INFERENCEX_REPO_ROOT}/infx/evals/bfcl_adapter.py" - local runtime_dir="" - local project_root="" - - mkdir -p "$results_dir" || return $? - results_dir="$(cd "$results_dir" && pwd)" || return $? - _prepare_eval_artifact_family "$results_dir" bfcl || return $? - - local setup_rc=0 integration_error="" - _prepare_vendor_verifier_python "BFCL" "bfcl-python" true 10 || { - setup_rc=$? - integration_error="BFCL Python runtime preparation failed with exit code ${setup_rc}" - } - if [ "$setup_rc" -eq 0 ]; then - runtime_dir=$(_prepare_bfcl_runtime) || { - setup_rc=$? - integration_error="BFCL dependency installation failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -eq 0 ]; then - project_root="$(mktemp -d /tmp/bfcl-project-root-XXXXXX)" || { - setup_rc=$? - integration_error="BFCL project root preparation failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -ne 0 ]; then - echo "ERROR: ${integration_error}" >&2 - local artifact_rc=0 - _write_bfcl_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - "$eval_suite" || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write BFCL failure artifact (exit code ${artifact_rc})" >&2 - fi - _cleanup_vendor_eval \ - "$runtime_dir" "$project_root" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$setup_rc" - fi - - local eval_rc=0 - local -a suite_args=() - if [ "$eval_suite" != "bfcl_smoke" ]; then - suite_args=(--suite "$eval_suite") - fi - run_server_client timeout "$process_timeout_seconds" \ - "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" \ - --base-url "http://127.0.0.1:${port}/v1" \ - --api-key EMPTY \ - --model "$model_name" \ - --output-dir "$results_dir" \ - --bfcl-project-root "$project_root" \ - "${suite_args[@]}" \ - --num-threads "$num_threads" \ - || eval_rc=$? - local archive_rc=0 - if [ "$archive_upstream" = true ]; then - _archive_bfcl_upstream_artifacts \ - "$project_root" "${results_dir}/bfcl_upstream_artifacts.tar.gz" \ - || archive_rc=$? - if [ "$archive_rc" -ne 0 ]; then - echo "ERROR: failed to archive BFCL upstream artifacts (exit code ${archive_rc})" >&2 - fi - fi - if [ "$eval_rc" -ne 0 ] \ - && { [ ! -f "${results_dir}/bfcl_report.json" ] \ - || [ ! -f "${results_dir}/results_bfcl.json" ]; }; then - local integration_error="BFCL evaluation failed with exit code ${eval_rc}" - local artifact_rc=0 - _write_bfcl_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - "$eval_suite" || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write BFCL failure artifact (exit code ${artifact_rc})" >&2 - fi - fi - _cleanup_vendor_eval \ - "$runtime_dir" "$project_root" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - if [ "$eval_rc" -ne 0 ]; then - return "$eval_rc" - fi - return "$archive_rc" -} - - -_run_bfcl_smoke_eval() { - _run_bfcl_suite_eval bfcl_smoke 4 900 false "$@" -} - -run_bfcl_eval() { - local eval_suite="${EVAL_SUITE:-bfcl_smoke}" - export EVAL_SUITE="$eval_suite" - - case "$eval_suite" in - bfcl_smoke) - _run_bfcl_smoke_eval "$@" - ;; - bfcl_vllm_minimax_m3) - _run_bfcl_suite_eval "$eval_suite" 8 7200 true "$@" - ;; - bfcl_vllm_kimi) - _run_bfcl_suite_eval "$eval_suite" 16 14400 true "$@" - ;; - *) - echo "ERROR: unsupported BFCL suite '${eval_suite}'" >&2 - export EVAL_RESULT_DIR="" - return 2 - ;; - esac -} - - -_write_minimax_vendor_integration_error() { - check_env_vars VENDOR_VERIFIER_PYTHON - local adapter_path="$1" - local model_name="$2" - local results_dir="$3" - local message="$4" - - # The failure path is stdlib-only, so it remains usable when runtime - # provisioning or dependency installation is what failed. - "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" failure \ - --model "$model_name" \ - --output-dir "$results_dir" \ - --message "$message" -} - -_run_minimax_m3_smoke_eval() { - check_env_vars PORT - local port="${PORT}" - local results_dir="${EVAL_RESULT_DIR:-}" - - while [[ $# -gt 0 ]]; do - case "$1" in - --port|--results-dir) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: $1 requires a value" >&2 - return 2 - fi - case "$1" in - --port) port="$2" ;; - --results-dir) results_dir="$2" ;; - esac - shift 2 - ;; - *) - echo "Unknown parameter: $1" >&2 - return 2 - ;; - esac - done - - if [ -z "$results_dir" ]; then - results_dir="$(mktemp -d /tmp/eval_out-XXXXXX)" || return $? - fi - - local model_name="${MODEL_NAME:-${MODEL:-}}" - local adapter_path="${INFERENCEX_REPO_ROOT}/infx/evals/minimax_provider_eval.py" - local fixture_path="${INFERENCEX_REPO_ROOT}/infx/evals/minimax_m3_smoke.json" - local runtime_dir="" - - mkdir -p "$results_dir" || return $? - results_dir="$(cd "$results_dir" && pwd)" || return $? - _prepare_eval_artifact_family "$results_dir" minimax || return $? - - local setup_rc=0 integration_error="" - _prepare_vendor_verifier_python "MiniMax Provider Verifier" "minimax-vendor-python" || { - setup_rc=$? - integration_error="MiniMax Provider Verifier Python runtime preparation failed with exit code ${setup_rc}" - } - if [ "$setup_rc" -eq 0 ]; then - runtime_dir=$(_prepare_minimax_m3_full_runtime) || { - setup_rc=$? - integration_error="MiniMax Provider Verifier pinned runtime preparation failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -ne 0 ]; then - echo "ERROR: ${integration_error}" >&2 - local artifact_rc=0 - _write_minimax_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write MiniMax verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - _cleanup_vendor_eval \ - "$runtime_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$setup_rc" - fi - - local eval_rc=0 - run_server_client "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" run \ - --python "${VENDOR_VERIFIER_PYTHON}" \ - --source-dir "${runtime_dir}/source" \ - --dependency-dir "${runtime_dir}/deps" \ - --base-url "http://127.0.0.1:${port}/v1" \ - --model "$model_name" \ - --output-dir "$results_dir" \ - --fixture "$fixture_path" \ - || eval_rc=$? - if [ "$eval_rc" -ne 0 ] \ - && ! _has_eval_result "$results_dir" "results_minimax_vendor_"; then - integration_error="MiniMax Provider Verifier failed with exit code ${eval_rc}" - local artifact_rc=0 - _write_minimax_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write MiniMax verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - fi - _cleanup_vendor_eval \ - "$runtime_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$eval_rc" -} - -_install_minimax_m3_full_deps() { - check_env_vars VENDOR_VERIFIER_PYTHON - local target_dir="$1" - "${VENDOR_VERIFIER_PYTHON}" -m pip install -q --no-cache-dir --target "$target_dir" \ - "jsonschema==4.25.1" \ - "loguru==0.7.3" \ - "megfile==4.2.5" \ - "numpy==2.3.4" \ - "openai==2.7.1" \ - "tqdm==4.67.1" -} - -_prepare_minimax_m3_full_runtime() { - check_env_vars VENDOR_VERIFIER_PYTHON - local source_adapter_path="${INFERENCEX_REPO_ROOT}/infx/evals/minimax_m3_full_eval.py" - local runtime_dir prepare_rc=0 - runtime_dir="$(mktemp -d /tmp/minimax-m3-full-runtime-XXXXXX)" || return $? - "${VENDOR_VERIFIER_PYTHON}" "$source_adapter_path" prepare-source \ - --source-dir "${runtime_dir}/source" >&2 || prepare_rc=$? - if [ "$prepare_rc" -eq 0 ]; then - _install_minimax_m3_full_deps "${runtime_dir}/deps" >&2 || prepare_rc=$? - fi - if [ "$prepare_rc" -ne 0 ]; then - rm -rf "$runtime_dir" - return "$prepare_rc" - fi - printf '%s\n' "$runtime_dir" -} - -_run_minimax_m3_full_eval() { - check_env_vars PORT - local port="${PORT}" - local results_dir="${EVAL_RESULT_DIR:-}" - - while [[ $# -gt 0 ]]; do - case "$1" in - --port|--results-dir) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: $1 requires a value" >&2 - return 2 - fi - case "$1" in - --port) port="$2" ;; - --results-dir) results_dir="$2" ;; - esac - shift 2 - ;; - *) - echo "Unknown parameter: $1" >&2 - return 2 - ;; - esac - done - - if [ -z "$results_dir" ]; then - results_dir="$(mktemp -d /tmp/eval_out-XXXXXX)" || return $? - fi - - local model_name="${MODEL_NAME:-${MODEL:-}}" - local adapter_path="${INFERENCEX_REPO_ROOT}/infx/evals/minimax_m3_full_eval.py" - local runtime_dir="" - - mkdir -p "$results_dir" || return $? - results_dir="$(cd "$results_dir" && pwd)" || return $? - _prepare_eval_artifact_family "$results_dir" minimax || return $? - - local setup_rc=0 integration_error="" - _prepare_vendor_verifier_python "MiniMax M3 full verifier" "minimax-m3-full-python" || { - setup_rc=$? - integration_error="MiniMax M3 full Python runtime preparation failed with exit code ${setup_rc}" - } - if [ "$setup_rc" -eq 0 ]; then - runtime_dir=$(_prepare_minimax_m3_full_runtime) || { - setup_rc=$? - integration_error="MiniMax M3 full pinned runtime preparation failed with exit code ${setup_rc}" - } - fi - if [ "$setup_rc" -ne 0 ]; then - echo "ERROR: ${integration_error}" >&2 - local artifact_rc=0 - _write_minimax_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write MiniMax full verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - _cleanup_vendor_eval \ - "$runtime_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$setup_rc" - fi - - local eval_rc=0 - run_server_client "${VENDOR_VERIFIER_PYTHON}" "$adapter_path" run \ - --python "${VENDOR_VERIFIER_PYTHON}" \ - --source-dir "${runtime_dir}/source" \ - --dependency-dir "${runtime_dir}/deps" \ - --base-url "http://127.0.0.1:${port}/v1" \ - --model "$model_name" \ - --output-dir "$results_dir" \ - || eval_rc=$? - if [ "$eval_rc" -ne 0 ] \ - && ! _has_eval_result "$results_dir" "results_minimax_vendor_full_"; then - integration_error="MiniMax M3 full verifier failed with exit code ${eval_rc}" - local artifact_rc=0 - _write_minimax_vendor_integration_error \ - "$adapter_path" "$model_name" "$results_dir" "$integration_error" \ - || artifact_rc=$? - if [ "$artifact_rc" -ne 0 ]; then - echo "ERROR: failed to write MiniMax full verifier failure artifact (exit code ${artifact_rc})" >&2 - fi - fi - _cleanup_vendor_eval \ - "$runtime_dir" "${VENDOR_VERIFIER_PYTHON_CLEANUP_DIR:-}" - return "$eval_rc" -} - - -run_minimax_vendor_eval() { - local eval_suite="${EVAL_SUITE:-minimax_m3_smoke}" - export EVAL_SUITE="$eval_suite" - - case "$eval_suite" in - minimax_m3_smoke) - _run_minimax_m3_smoke_eval "$@" - ;; - minimax_m3_full) - _run_minimax_m3_full_eval "$@" - ;; - *) - echo "ERROR: unsupported MiniMax Provider Verifier suite '${eval_suite}'" >&2 - export EVAL_RESULT_DIR="" - return 2 - ;; - esac -} - -_eval_patches_dir() { - printf '%s\n' "${INFERENCEX_REPO_ROOT}/infx/evals/patches" -} - -_patch_lm_eval() { - local patch_dir - patch_dir="$(mktemp -d)" - cp "$(_eval_patches_dir)/lm_eval_sitecustomize.py" "$patch_dir/sitecustomize.py" - export PYTHONPATH="${patch_dir}${PYTHONPATH:+:${PYTHONPATH}}" -} - -get_native_max_context_length() { - local model_path="$1" - # Prefer MODEL_PATH (local model directory) when available, since the - # argument may be a served-model name that is neither a valid HF repo - # ID nor a local path (e.g. "deepseek-r1-fp4" on the B300 cluster). - if [ -n "${MODEL_PATH:-}" ] && [ -d "${MODEL_PATH}" ]; then - model_path="${MODEL_PATH}" - fi - python3 - "$model_path" <<'PY' -import json -import sys -from pathlib import Path - -fields = ['max_position_embeddings', 'max_sequence_length', 'seq_length', 'n_positions'] -try: - config = json.loads((Path(sys.argv[1]) / 'config.json').read_text()) - for field in fields: - value = config.get(field) - if type(value) is int and value > 0: - print(value) - sys.exit(0) -except (OSError, ValueError, AttributeError): - pass - -try: - from transformers import AutoConfig - config = AutoConfig.from_pretrained(sys.argv[1], trust_remote_code=True) - for attr in fields: - if hasattr(config, attr): - print(getattr(config, attr)) - break - else: - print(0) -except Exception: - print(0) -PY -} - -# Requested benchmark context capped at the model's native max. Sets -# EVAL_MAX_MODEL_LEN (read by run_lm_eval) and echoes the value. -compute_eval_context_length() { - local model="$1" - local benchmark_ctx="${2:-0}" - local native_max - native_max=$(get_native_max_context_length "$model") - native_max="${native_max:-0}" - - if [ "$benchmark_ctx" -eq 0 ] 2>/dev/null; then - benchmark_ctx="${native_max:-0}" - fi - local eval_ctx=$(( benchmark_ctx * 1 )) - if [ "$native_max" -gt 0 ] 2>/dev/null && [ "$eval_ctx" -gt "$native_max" ]; then - eval_ctx="$native_max" - fi - if [ "$eval_ctx" -le 0 ] 2>/dev/null; then - echo "WARN: compute_eval_context_length could not determine context length for $model" >&2 - eval_ctx="${MAX_MODEL_LEN:-16384}" - fi - EVAL_MAX_MODEL_LEN="$eval_ctx" - echo "$eval_ctx" -} - -run_lm_eval() { - check_env_vars OPENAI_API_KEY PORT - local port="${PORT}" - local tasks_dir="${EVAL_TASKS_DIR:-infx/evals/gsm8k.yaml}" - local results_dir="${EVAL_RESULT_DIR:-$(mktemp -d /tmp/eval_out-XXXXXX)}" - local eval_context_len="${EVAL_MAX_MODEL_LEN}" - local temperature=0 - local top_p=1 - local concurrent_requests="${EVAL_CONCURRENT_REQUESTS:-${CONC}}" - check_env_vars concurrent_requests - # --limit is passed only when EVAL_LIMIT requests a smoke-test slice. - local eval_limit="${EVAL_LIMIT:-}" - local include_path="${EVAL_INCLUDE_PATH:-}" - - while [[ $# -gt 0 ]]; do - case "$1" in - --port|--task|--results-dir|--gen-max-tokens|--temperature|--top-p) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: $1 requires a value" >&2 - return 2 - fi - case "$1" in - --port) port="$2" ;; - --task) tasks_dir="$2" ;; - --results-dir) results_dir="$2" ;; - --gen-max-tokens) eval_context_len="$2" ;; - --temperature) temperature="$2" ;; - --top-p) top_p="$2" ;; - esac - shift 2 - ;; - *) - echo "Unknown parameter: $1" >&2 - return 2 - ;; - esac - done - - check_env_vars eval_context_len - - # Serving images may use a different WORKDIR. - local _repo_root="$INFERENCEX_REPO_ROOT" - if [[ "$tasks_dir" == *.yaml && "$tasks_dir" != /* \ - && ! -f "$tasks_dir" && -f "$_repo_root/$tasks_dir" ]]; then - echo "run_lm_eval: anchoring relative task '$tasks_dir' to repo root -> $_repo_root/$tasks_dir" - tasks_dir="$_repo_root/$tasks_dir" - fi - - export EVAL_TASKS_DIR="$tasks_dir" - - if [ "${INFERENCEX_LM_EVAL_RUNTIME_READY:-false}" != "true" ]; then - _install_lm_eval_deps - _patch_lm_eval - export INFERENCEX_LM_EVAL_RUNTIME_READY=true - fi - - local openai_server_base="http://0.0.0.0:${port}" - local openai_chat_base="${openai_server_base}/v1/chat/completions" - export OPENAI_API_KEY=${OPENAI_API_KEY} - MODEL_NAME=${MODEL_NAME:-$MODEL} # Prefer MODEL_NAME, else MODEL - - # Leave room for input within the context window and avoid excessive - # per-request KV cache reservation on TRT. - local max_output_tokens=$(( eval_context_len > 4096 ? eval_context_len - 4096 : eval_context_len / 2 )) - if [ "$max_output_tokens" -gt 16384 ]; then - max_output_tokens=16384 - fi - echo "Eval budget: eval_context_len=${eval_context_len}, max_output_tokens=${max_output_tokens}" - - # Read by append_lm_eval_summary. - export EVAL_RESULT_DIR="$results_dir" - set -x - run_server_client python3 -m lm_eval --model local-chat-completions --apply_chat_template \ - ${include_path:+--include_path "$include_path"} \ - --tasks "${tasks_dir}" \ - --output_path "${results_dir}" \ - --log_samples \ - --model_args "model=${MODEL_NAME},base_url=${openai_chat_base},api_key=${OPENAI_API_KEY},eos_string=,max_retries=5,num_concurrent=${concurrent_requests},timeout=1800,tokenized_requests=False,max_length=${eval_context_len}" \ - --gen_kwargs "max_tokens=${max_output_tokens},temperature=${temperature},top_p=${top_p}" \ - ${eval_limit:+--limit "$eval_limit"} - local eval_exit=$? - set +x - return $eval_exit -} - -_stage_lm_eval_artifacts() { - local results_dir="$1" - local eval_conc="$2" - local moved=0 - local failed=0 - local jf base stem extension target suffix - - if [ ! -d "${results_dir}" ]; then - echo "WARN: eval result directory '${results_dir}' does not exist" >&2 - return 1 - fi - - while IFS= read -r -d '' jf; do - base=$(basename "$jf") - case "$base" in - meta_env.json) - continue - ;; - *.jsonl) - stem="${base%.jsonl}" - extension=".jsonl" - ;; - *.json) - stem="${base%.json}" - extension=".json" - ;; - *) - continue - ;; - esac - - target="./${stem}_conc${eval_conc}${extension}" - suffix=2 - while [ -e "$target" ]; do - target="./${stem}_conc${eval_conc}_${suffix}${extension}" - suffix=$((suffix + 1)) - done - - if mv -f "$jf" "$target"; then - moved=1 - else - echo "WARN: failed to stage eval artifact ${jf}" >&2 - failed=1 - fi - done < <( - find "${results_dir}" -type f \ - \( -name "*.json" -o -name "*.jsonl" \) -print0 2>/dev/null - ) - - rm -rf --one-file-system "${results_dir}" 2>/dev/null \ - || rm -rf "${results_dir}" \ - || true - - if [ "$moved" -eq 0 ]; then - echo "WARN: no eval artifacts were produced for concurrency ${eval_conc}" >&2 - return 1 - fi - return "$failed" -} - -_eval_concs_to_json() { - local values="$1" - local value - local joined="" - - for value in $values; do - if [[ ! "$value" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: invalid eval concurrency '${value}'" >&2 - return 1 - fi - if [ -n "$joined" ]; then - joined="${joined}, " - fi - joined="${joined}${value}" - done - - printf '[%s]' "$joined" -} - -_env_is_true() { - case "${1:-}" in - 1|[Tt][Rr][Uu][Ee]|[Yy][Ee][Ss]|[Oo][Nn]) return 0 ;; - *) return 1 ;; - esac -} - -_resolve_disagg_ep() { - local ep="${1:-1}" - local enable_flag="${2:-false}" - local tp_size="${3:-1}" - if [[ "$ep" == "1" ]] && _env_is_true "$enable_flag"; then - echo "$tp_size" - else - echo "$ep" - fi -} - -_normalize_bool_json() { - if _env_is_true "${1:-false}"; then - echo "true" - else - echo "false" - fi -} - -# Export TP/EP/DP metadata for append_lm_eval_summary / meta_env.json. -# Prefer workflow PREFILL_EP/DECODE_EP and *_DP_ATTN (from job.slurm) over -# ENABLE_* launch booleans so DEP8/DPA arms record the correct topology. -bridge_disagg_eval_metadata() { - export TP="${PREFILL_TP:-${PREFILL_TP_SIZE:-${TP:-1}}}" - export PREFILL_TP="${PREFILL_TP:-${PREFILL_TP_SIZE:-${TP:-1}}}" - export PREFILL_EP="$(_resolve_disagg_ep "${PREFILL_EP:-${EP_SIZE:-${EP:-1}}}" "${PREFILL_ENABLE_EP:-false}" "${PREFILL_TP_SIZE:-${PREFILL_TP:-1}}")" - export EP_SIZE="${PREFILL_EP}" - export PREFILL_NUM_WORKERS="${PREFILL_NUM_WORKERS:-${xP:-1}}" - export DECODE_TP="${DECODE_TP:-${DECODE_TP_SIZE:-${TP:-1}}}" - export DECODE_EP="$(_resolve_disagg_ep "${DECODE_EP:-${EP_SIZE:-${EP:-1}}}" "${DECODE_ENABLE_EP:-false}" "${DECODE_TP_SIZE:-${DECODE_TP:-1}}")" - export DECODE_NUM_WORKERS="${DECODE_NUM_WORKERS:-${yD:-1}}" - - local prefill_dp="${PREFILL_DP_ATTN:-${PREFILL_DP_ATTENTION:-${PREFILL_ENABLE_DP:-false}}}" - local decode_dp="${DECODE_DP_ATTN:-${DECODE_DP_ATTENTION:-${DECODE_ENABLE_DP:-false}}}" - export DP_ATTENTION="$(_normalize_bool_json "$prefill_dp")" - export PREFILL_DP_ATTENTION="$(_normalize_bool_json "$prefill_dp")" - export DECODE_DP_ATTENTION="$(_normalize_bool_json "$decode_dp")" -} - -_write_lm_eval_meta_json() { - check_env_vars IS_MULTINODE - local meta_json="$1" - local batch_metadata="${2:-}" - local metadata_conc="${3:-${CONC:-1}}" - - # Single-node jobs already export TP/EP/DP_ATTENTION. The disaggregated - # bridge defaults missing per-phase DP flags to false, so applying it to - # single-node jobs would overwrite their actual DP-attention setting. - if [ "${IS_MULTINODE}" = "true" ]; then - bridge_disagg_eval_metadata - fi - - local model_name="${MODEL_NAME:-$MODEL}" - local is_multinode_json="false" - if [ "${IS_MULTINODE}" = "true" ]; then - is_multinode_json="true" - fi - - local prefill_tp="${PREFILL_TP:-${TP:-1}}" - local prefill_pp="${PREFILL_PP_SIZE:-${PP_SIZE:-1}}" - local prefill_dcp_size="${PREFILL_DCP_SIZE:-${DCP_SIZE:-1}}" - local prefill_pcp_size="${PREFILL_PCP_SIZE:-${PCP_SIZE:-1}}" - local prefill_ep="${PREFILL_EP:-${EP_SIZE:-1}}" - local prefill_num_workers="${PREFILL_NUM_WORKERS:-1}" - local decode_tp="${DECODE_TP:-${TP:-1}}" - local decode_pp="${DECODE_PP_SIZE:-${PP_SIZE:-1}}" - local decode_dcp_size="${DECODE_DCP_SIZE:-${DCP_SIZE:-1}}" - local decode_pcp_size="${DECODE_PCP_SIZE:-${PCP_SIZE:-1}}" - local decode_ep="${DECODE_EP:-${EP_SIZE:-1}}" - local decode_num_workers="${DECODE_NUM_WORKERS:-1}" - - local dp_json - dp_json="$(_normalize_bool_json "${DP_ATTENTION:-false}")" - local prefill_dp_json - prefill_dp_json="$(_normalize_bool_json "${PREFILL_DP_ATTENTION:-${DP_ATTENTION:-false}}")" - local decode_dp_json - decode_dp_json="$(_normalize_bool_json "${DECODE_DP_ATTENTION:-${DP_ATTENTION:-false}}")" - - local fw="${FRAMEWORK:-}" - local prec="${PRECISION:-}" - if [[ -z "$fw" || -z "$prec" ]]; then - if [[ -n "${RESULT_FILENAME:-}" ]]; then - local parsed - parsed=$(echo "${RESULT_FILENAME}" | sed -n 's/.*_\([^_][^_]*\)_\([^_][^_]*\)_tp.*/\1 \2/p') - local p1="${parsed%% *}" - local p2="${parsed#* }" - if [[ -z "$prec" && -n "$p1" && "$p1" != "$parsed" ]]; then - prec="$p1" - fi - if [[ -z "$fw" && -n "$p2" && "$p2" != "$parsed" ]]; then - fw="$p2" - fi - fi - fi - local eval_suite="${EVAL_COMPLETED_SUITE:-${EVAL_SUITE:-}}" - if [ -z "$eval_suite" ] && [ -n "${EVAL_TASKS_DIR:-}" ]; then - eval_suite="$(basename "${EVAL_TASKS_DIR}")" - eval_suite="${eval_suite%.yaml}" - eval_suite="${eval_suite%.yml}" - fi - eval_suite="${eval_suite:-gsm8k}" - - cat > "${meta_json}" <&2 - return 1 - fi - if [ ! -d "${out_dir}" ]; then - echo "WARN: EVAL_RESULT_DIR='${out_dir}' does not exist; skipping artifact collection" >&2 - return 1 - fi - meta_json="${out_dir}/meta_env.json" - fi - - _write_lm_eval_meta_json "$meta_json" "$batch_metadata" "$metadata_conc" - - if [ -n "$batch_concs" ]; then - echo "Prepared batched eval artifacts in: $(pwd)" - return 0 - fi - - stage_eval_artifacts "$(pwd)" "$out_dir" || return $? - - if [ -n "${out_dir}" ] && [ -d "${out_dir}" ]; then - rm -rf --one-file-system "${out_dir}" || rm -rf "${out_dir}" || true - fi - - echo "Staged eval artifacts in: $(pwd)" -} - -stage_eval_artifacts() { - local destination="$1" - shift - - mkdir -p "$destination" || return $? - local source_dir artifact - local copied=0 - local artifacts=() - for source_dir in "$@"; do - [ -d "$source_dir" ] || continue - artifacts=( - "$source_dir"/meta_env.json - "$source_dir"/results*.json - "$source_dir"/*_report.json - "$source_dir"/*_results.jsonl - "$source_dir"/*_artifacts.tar.gz - "$source_dir"/sample*.jsonl - ) - for artifact in "${artifacts[@]}"; do - [ -f "$artifact" ] || continue - cp -f "$artifact" "$destination/" || return $? - copied=$((copied + 1)) - done - done - if [ "$copied" -eq 0 ]; then - echo "ERROR: no eval artifacts found to stage" >&2 - return 1 - fi -} - - -_wait_for_openai_chat_route() { - check_env_vars EVAL_ENDPOINT_READY_TIMEOUT_SECONDS EVAL_MODEL_STABILIZATION_SECONDS PORT - local port="${PORT}" - local timeout_seconds="${EVAL_ENDPOINT_READY_TIMEOUT_SECONDS}" - local poll_seconds=5 - local stabilization_seconds="${EVAL_MODEL_STABILIZATION_SECONDS}" - local start_seconds=$SECONDS - local server_ready_since=-1 - local next_report=0 - local elapsed percent chat_status - local served_model="${SERVED_MODEL_NAME:-${MODEL:-}}" - local health_url models_url chat_url - - while [[ $# -gt 0 ]]; do - case "$1" in - --port) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: --port requires a value" >&2 - return 2 - fi - port="$2" - shift 2 - ;; - *) shift ;; - esac - done - if ! [[ "$timeout_seconds" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: EVAL_ENDPOINT_READY_TIMEOUT_SECONDS must be a positive integer" >&2 - return 2 - fi - if ! [[ "$stabilization_seconds" =~ ^[0-9]+$ ]]; then - echo "ERROR: EVAL_MODEL_STABILIZATION_SECONDS must be a non-negative integer" >&2 - return 2 - fi - if [ -z "$served_model" ]; then - echo "ERROR: MODEL or SERVED_MODEL_NAME is required for chat endpoint readiness" >&2 - return 2 - fi - health_url="http://localhost:${port}/health" - models_url="http://localhost:${port}/v1/models" - chat_url="http://localhost:${port}/v1/chat/completions" - - while true; do - local model_ready=false - local server_ready=false - if curl -fsS --max-time 10 "$health_url" >/dev/null 2>&1; then - server_ready=true - fi - if curl -fsS --max-time 10 "$models_url" 2>/dev/null \ - | python3 -c ' -import json -import sys - -expected = sys.argv[1] -payload = json.load(sys.stdin) -models = payload.get("data", []) -raise SystemExit(0 if any(model.get("id") == expected for model in models) else 1) -' "$served_model" >/dev/null 2>&1; then - model_ready=true - fi - - chat_status="" - if [ "$model_ready" = true ]; then - chat_status="$(curl -sS -o /dev/null -w '%{http_code}' --max-time 10 \ - "$chat_url" 2>/dev/null)" || true - case "$chat_status" in - 401|403|405) break ;; - esac - fi - if [ "$server_ready" = true ]; then - if [ "$server_ready_since" -lt 0 ]; then - server_ready_since=$SECONDS - fi - if [ $((SECONDS - server_ready_since)) -ge "$stabilization_seconds" ]; then - break - fi - else - server_ready_since=-1 - fi - - elapsed=$((SECONDS - start_seconds)) - if [ "$elapsed" -ge "$timeout_seconds" ]; then - echo "ERROR: chat endpoint for model '$served_model' did not become ready within ${timeout_seconds}s: $chat_url" >&2 - return 1 - fi - if [ "$elapsed" -ge "$next_report" ]; then - percent=$((elapsed * 100 / timeout_seconds)) - echo "Waiting for chat endpoint for model '$served_model': ${elapsed}/${timeout_seconds}s (${percent}%)" - next_report=$((next_report + 60)) - fi - sleep "$poll_seconds" - done - echo "OpenAI chat endpoint ready for model '$served_model': $chat_url" -} - - -# Unified eval entrypoint - -run_eval() { - check_env_vars EVAL_ONLY IS_AGENTIC - local cli_framework="" - local forwarded=() - # Keep runner-selected suite identity scoped to this invocation. - local EVAL_SUITE="${EVAL_SUITE:-}" - unset EVAL_COMPLETED_SUITE - - while [[ $# -gt 0 ]]; do - case "$1" in - --framework) - if [[ $# -lt 2 || -z "${2:-}" || "${2:-}" == --* ]]; then - echo "ERROR: --framework requires a value" >&2 - return 2 - fi - cli_framework="$2" - shift 2 - ;; - *) - forwarded+=("$1") - shift - ;; - esac - done - - local scenario_default="lm-eval" - local scenario_is_agentic=0 - if [ "${IS_AGENTIC}" = "1" ] || [ "${SCENARIO_TYPE:-}" = "agentic-coding" ]; then - scenario_is_agentic=1 - fi - - local framework="${EVAL_FRAMEWORK:-${cli_framework:-$scenario_default}}" - case "$framework" in - kimi-vendor) - [ -n "${EVAL_SUITE:-}" ] || EVAL_SUITE="kimi_tool_call_schema" - ;; - minimax-vendor) - [ -n "${EVAL_SUITE:-}" ] || EVAL_SUITE="minimax_m3_smoke" - ;; - bfcl) - [ -n "${EVAL_SUITE:-}" ] || EVAL_SUITE="bfcl_smoke" - ;; - esac - - case "${EVAL_SUITE:-}" in - "") ;; - *[!A-Za-z0-9_.-]*) - echo "ERROR: EVAL_SUITE may contain only letters, digits, '.', '_', and '-'" >&2 - return 2 - ;; - esac - - if [ -n "${EVAL_SUITE:-}" ] \ - && [ "$framework" != "kimi-vendor" ] \ - && [ "$framework" != "minimax-vendor" ] \ - && [ "$framework" != "bfcl" ]; then - echo "ERROR: EVAL_SUITE is only supported with kimi-vendor, minimax-vendor, or bfcl" >&2 - return 2 - fi - - if [ "${EVAL_ONLY}" = "true" ]; then - case "$framework" in - kimi-vendor|minimax-vendor|bfcl) - _wait_for_openai_chat_route "${forwarded[@]}" || return $? - ;; - esac - fi - - # Explicit verifier suites use fixed request budgets and do not consume - # EVAL_MAX_MODEL_LEN, so avoid loading model configuration for those paths. - if [ "$framework" != "kimi-vendor" ] \ - && [ "$framework" != "minimax-vendor" ] \ - && [ "$framework" != "bfcl" ] \ - && [ -z "${EVAL_MAX_MODEL_LEN:-}" ]; then - compute_eval_context_length "$MODEL" "${MAX_MODEL_LEN:-0}" > /dev/null - fi - - unset EVAL_BATCHED_CONCS - unset EVAL_BATCHED_COMPLETED_CONCS - unset EVAL_BATCHED_FAILED_CONCS - - local requested_concs="${EVAL_CONCURRENT_REQUESTS:-}" - local eval_concs=() - if [ -n "$requested_concs" ]; then - read -r -a eval_concs <<< "$requested_concs" - fi - - if [ "${#eval_concs[@]}" -gt 1 ]; then - if [[ "$framework" != "lm-eval" && "$framework" != "lm_eval" ]]; then - echo "ERROR: batched eval concurrency is only supported for lm-eval" >&2 - return 1 - fi - - local eval_conc results_dir eval_rc stage_rc - local completed_concs=() - local failed_concs=() - - for eval_conc in "${eval_concs[@]}"; do - if [[ ! "$eval_conc" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: invalid eval concurrency '${eval_conc}'" >&2 - return 1 - fi - - if ! results_dir=$(mktemp -d /tmp/eval_out-conc"${eval_conc}"-XXXXXX); then - echo "ERROR: failed to create eval output directory for concurrency ${eval_conc}" >&2 - failed_concs+=("$eval_conc") - continue - fi - - echo "Running lm-eval at concurrency ${eval_conc} using the existing engine" - export EVAL_CONCURRENT_REQUESTS="$eval_conc" - export CONC="$eval_conc" - eval_rc=0 - stage_rc=0 - run_lm_eval "${forwarded[@]}" --results-dir "$results_dir" \ - || eval_rc=$? - _stage_lm_eval_artifacts "$results_dir" "$eval_conc" \ - || stage_rc=$? - - if [ "$eval_rc" -eq 0 ] && [ "$stage_rc" -eq 0 ]; then - completed_concs+=("$eval_conc") - else - echo "ERROR: lm-eval failed at concurrency ${eval_conc} (eval_rc=${eval_rc}, stage_rc=${stage_rc})" >&2 - failed_concs+=("$eval_conc") - fi - done - - export EVAL_CONCURRENT_REQUESTS="$requested_concs" - export EVAL_RESULT_DIR="" - export EVAL_BATCHED_CONCS="${eval_concs[*]}" - export EVAL_BATCHED_COMPLETED_CONCS="${completed_concs[*]}" - export EVAL_BATCHED_FAILED_CONCS="${failed_concs[*]}" - - if [ "${#failed_concs[@]}" -gt 0 ]; then - echo "ERROR: batched eval failed for concurrency: ${failed_concs[*]}" >&2 - echo "Deferring failure until post-upload score validation preserves all artifacts" >&2 - fi - return 0 - fi - - if [ -n "${EVAL_CONCURRENT_REQUESTS:-}" ]; then - export CONC="$EVAL_CONCURRENT_REQUESTS" - fi - - local eval_rc=0 - case "$framework" in - lm-eval|lm_eval) run_lm_eval "${forwarded[@]}" || eval_rc=$? ;; - kimi-vendor) run_kimi_vendor_eval "${forwarded[@]}" || eval_rc=$? ;; - minimax-vendor) run_minimax_vendor_eval "${forwarded[@]}" || eval_rc=$? ;; - bfcl) run_bfcl_eval "${forwarded[@]}" || eval_rc=$? ;; - *) echo "Unknown framework '${framework}'"; eval_rc=1 ;; - esac - - if [ -n "${EVAL_SUITE:-}" ]; then - export EVAL_COMPLETED_SUITE="$EVAL_SUITE" - fi - - local stage_rc=0 - # Agentic eval-only recipes have no separate staging step. Provider - # failures are staged before returning so diagnostic artifacts survive. - if { [ "${EVAL_ONLY}" = "true" ] && [ "$scenario_is_agentic" = "1" ]; } \ - || { { [ "$framework" = "kimi-vendor" ] \ - || [ "$framework" = "minimax-vendor" ] \ - || [ "$framework" = "bfcl" ]; } \ - && [ "$eval_rc" -ne 0 ]; }; then - append_lm_eval_summary || stage_rc=$? - fi - if [ "$eval_rc" -ne 0 ]; then - echo "ERROR: run_eval failed with exit code $eval_rc" >&2 - if [ "${EVAL_ONLY}" = "true" ]; then - echo "Eval-only mode: failing after artifact collection" >&2 - fi - return "$eval_rc" - fi - if [ "$stage_rc" -ne 0 ]; then - echo "ERROR: eval artifact staging failed with exit code $stage_rc" >&2 - return "$stage_rc" - fi - return 0 -} - - -# Agentic trace replay helpers (aiperf driver) - -AIPERF_DIR="${INFMAX_CONTAINER_WORKSPACE}/utils/aiperf" -AIPERF_RUNTIME_DIR="${AIPERF_RUNTIME_DIR:-${TMPDIR:-/tmp}/inferencex-agentic-${SLURM_JOB_ID:-$$}}" -AIPERF_VENV="${AIPERF_RUNTIME_DIR}/venv" -AIPERF_UV_INSTALL_DIR="${AIPERF_RUNTIME_DIR}/uv/bin" -AIPERF_UV_CACHE_DIR="${AIPERF_RUNTIME_DIR}/uv-cache" -AIPERF_PYTHON="${AIPERF_VENV}/bin/python" -AIPERF_CLI="${AIPERF_VENV}/bin/aiperf" -AIPERF_HF_CLI="${AIPERF_VENV}/bin/hf" -AIPERF_DEPS_READY=0 - -ensure_agentic_uv() { - if command -v uv >/dev/null 2>&1; then - AIPERF_UV_BIN="$(command -v uv)" - return - fi - - AIPERF_UV_BIN="${AIPERF_UV_INSTALL_DIR}/uv" - if [ ! -x "$AIPERF_UV_BIN" ]; then - mkdir -p "$AIPERF_UV_INSTALL_DIR" - curl -LsSf https://astral.sh/uv/install.sh | - UV_INSTALL_DIR="$AIPERF_UV_INSTALL_DIR" sh - fi - - if [ ! -x "$AIPERF_UV_BIN" ]; then - echo "ERROR: uv installation did not create $AIPERF_UV_BIN" >&2 - return 1 - fi -} - -install_agentic_deps() { - check_env_vars AIPERF_PYTHON_VERSION INFMAX_CONTAINER_WORKSPACE - if [ "$AIPERF_DEPS_READY" = "1" ]; then - return - fi - - # uv install from the checked-out aiperf source: needs no git, and rootless - # Enroot containers cannot mutate dpkg. - - ensure_agentic_uv || return $? - rm -rf "$AIPERF_VENV" - mkdir -p "$AIPERF_UV_CACHE_DIR" - - # Pin the interpreter version instead of the container's python3: aiperf - # dropped Python 3.10 (SemiAnalysisAI/aiperf#1107) while sglang-rocm/vllm-rocm - # images still default to 3.10.12, which left the venv without aiperf/hf. - # uv downloads a standalone build when the system lacks one. - UV_CACHE_DIR="$AIPERF_UV_CACHE_DIR" \ - "$AIPERF_UV_BIN" venv --python "${AIPERF_PYTHON_VERSION}" "$AIPERF_VENV" || return $? - UV_CACHE_DIR="$AIPERF_UV_CACHE_DIR" UV_HTTP_TIMEOUT=120 UV_HTTP_RETRIES=3 \ - "$AIPERF_UV_BIN" pip install --python "$AIPERF_PYTHON" \ - -e "$AIPERF_DIR" \ - "numpy>=1.24" \ - "pandas>=2.0.0" \ - "aiohttp>=3.10" \ - "transformers>=4.46" \ - "xlsxwriter>=3.2.1" \ - "tqdm>=4.66" \ - "datasets>=4.7.0" \ - tiktoken \ - "huggingface_hub[cli]>=0.25.0" \ - urllib3 \ - requests || { - echo "ERROR: benchmark client dependency bootstrap failed; inspect network/package resolution before recipe repairs" >&2 - return 1 - } - - if [ ! -x "$AIPERF_CLI" ] || [ ! -x "$AIPERF_HF_CLI" ]; then - echo "ERROR: isolated AIPerf environment is incomplete at $AIPERF_VENV" >&2 - return 1 - fi - AIPERF_DEPS_READY=1 -} - -ensure_hf_cli() { - install_agentic_deps -} - -resolve_trace_source() { - # WEKA_LOADER_OVERRIDE picks an aiperf public-dataset loader for recipes - # with non-default context caps (minimaxm2.5 at ~256k cannot replay the - # unfiltered corpus) or to pin an older corpus. Default: the 062126 v7 - # corpus; 1M-context families take the unfiltered variant, others 256k. - local default_loader - case "${MODEL_PREFIX:-}" in - dsv4*|glm5.2*|glm5.3*|minimaxm3*|kimik3*) - default_loader="semianalysis_cc_traces_weka_062126" - ;; - *) - default_loader="semianalysis_cc_traces_weka_062126_256k" - ;; - esac - local loader="${WEKA_LOADER_OVERRIDE:-$default_loader}" - local dataset - case "$loader" in - semianalysis_cc_traces_weka_with_subagents) - dataset="semianalysisai/cc-traces-weka-061526" - ;; - semianalysis_cc_traces_weka_with_subagents_256k) - dataset="semianalysisai/cc-traces-weka-061526-256k" - ;; - semianalysis_cc_traces_weka_with_subagents_060226) - dataset="semianalysisai/cc-traces-weka-with-subagents-060226" - ;; - semianalysis_cc_traces_weka_with_subagents_060226_256k) - dataset="semianalysisai/cc-traces-weka-with-subagents-060226-256k" - ;; - semianalysis_cc_traces_weka_with_subagents_060526) - dataset="semianalysisai/cc-traces-weka-with-subagents-060526" - ;; - semianalysis_cc_traces_weka_with_subagents_060526_256k) - dataset="semianalysisai/cc-traces-weka-with-subagents-060526-256k" - ;; - semianalysis_cc_traces_weka_with_subagents_060826) - dataset="semianalysisai/cc-traces-weka-with-subagents-060826" - ;; - semianalysis_cc_traces_weka_with_subagents_060826_256k) - dataset="semianalysisai/cc-traces-weka-with-subagents-060826-256k" - ;; - semianalysis_cc_traces_weka_061326) - dataset="semianalysisai/cc-traces-weka-061326" - ;; - semianalysis_cc_traces_weka_061326_256k) - dataset="semianalysisai/cc-traces-weka-061326-256k" - ;; - semianalysis_cc_traces_weka_061526) - dataset="semianalysisai/cc-traces-weka-061526" - ;; - semianalysis_cc_traces_weka_061526_256k) - dataset="semianalysisai/cc-traces-weka-061526-256k" - ;; - semianalysis_cc_traces_weka_062126) - dataset="semianalysisai/cc-traces-weka-062126" - ;; - semianalysis_cc_traces_weka_062126_256k) - dataset="semianalysisai/cc-traces-weka-062126-256k" - ;; - *) - echo "Error: unknown WEKA_LOADER_OVERRIDE='$loader'. Allowed: semianalysis_cc_traces_weka_with_subagents, semianalysis_cc_traces_weka_with_subagents_256k, semianalysis_cc_traces_weka_with_subagents_060226, semianalysis_cc_traces_weka_with_subagents_060226_256k, semianalysis_cc_traces_weka_with_subagents_060526, semianalysis_cc_traces_weka_with_subagents_060526_256k, semianalysis_cc_traces_weka_with_subagents_060826, semianalysis_cc_traces_weka_with_subagents_060826_256k, semianalysis_cc_traces_weka_061326, semianalysis_cc_traces_weka_061326_256k, semianalysis_cc_traces_weka_061526, semianalysis_cc_traces_weka_061526_256k, semianalysis_cc_traces_weka_062126, semianalysis_cc_traces_weka_062126_256k" >&2 - exit 1 - ;; - esac - TRACE_SOURCE_FLAG="--public-dataset $loader" - echo "Loading traces via aiperf public-dataset: $loader ($dataset) [MODEL_PREFIX=${MODEL_PREFIX:-unset}]" - # Pre-download into the shared HF_HUB_CACHE so later jobs hit cache. - ensure_hf_cli - "$AIPERF_HF_CLI" download --repo-type dataset "$dataset" -} - -build_replay_cmd() { - check_env_vars INFMAX_CONTAINER_WORKSPACE MODEL PORT CONC DURATION - validate_agentic_concurrency "$CONC" || return 1 - check_env_vars \ - AIPERF_FAILED_REQUEST_THRESHOLD AIPERF_LIVE_FAILED_REQUEST_THRESHOLD \ - AIPERF_TRACE_IDLE_GAP_CAP_SECONDS - check_env_vars \ - AGENTIC_WARMUP_GRACE_PERIOD \ - AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS AIPERF_EXPERIMENTAL_FAST \ - AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID AIPERF_UNSAFE_OVERRIDE \ - AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING AIPERF_WARMUP_REQUESTS_PER_LANE - # The agentx preset supplies shared replay and runtime defaults. Recipe-owned - # duration, warmup, failure threshold, and trace cap remain explicit below. - local result_dir="$1" - local duration="$DURATION" - local warmup_requests_per_lane="${AIPERF_WARMUP_REQUESTS_PER_LANE}" - - # Fast mode: one advance per lane and a 20-minute profile. - if [[ "${AIPERF_EXPERIMENTAL_FAST}" == "1" ]]; then - duration=1200 - warmup_requests_per_lane=1 - fi - - REPLAY_CMD="$AIPERF_CLI profile --scenario agentx" - REPLAY_CMD+=" --url ${AIPERF_SERVER_URL:-http://localhost:$PORT}" - REPLAY_CMD+=" --endpoint /v1/chat/completions" - # SERVED_MODEL_NAME covers frontends that register the model under a wire - # name (dynamo-trt serves "DeepSeek-V4-Pro" while $MODEL is the HF id); - # a mismatch 404s at warmup. - REPLAY_CMD+=" --model ${SERVED_MODEL_NAME:-$MODEL}" - # The tokenizer defaults to --model, and a wire name is not necessarily a - # valid HF repo id, so pass the real id explicitly. - REPLAY_CMD+=" --tokenizer $MODEL" - REPLAY_CMD+=" --concurrency $CONC" - REPLAY_CMD+=" --benchmark-duration $duration" - # Live abort threshold; recipes with correlated low-concurrency trajectories - # may loosen it while AIPERF_FAILED_REQUEST_THRESHOLD stays the post-run gate. - REPLAY_CMD+=" --failed-request-threshold $AIPERF_LIVE_FAILED_REQUEST_THRESHOLD" - # Extra one-token requests per lane after the t* snapshot primers; profiling - # resumes from the resulting live state. Do not pass --burst-phase-starts: - # the spread default preserves each lane's recorded phase-start offset. - REPLAY_CMD+=" --warmup-requests-per-lane $warmup_requests_per_lane" - # Caps end-to-start idle time per trajectory tree (root plus subagents) - # without reordering requests or bypassing spawn/join dependencies. - REPLAY_CMD+=" --trace-idle-gap-cap-seconds $AIPERF_TRACE_IDLE_GAP_CAP_SECONDS" - # Maximum wait for warmup to drain, not a fixed sleep; saturation arms with a - # larger in-flight set can raise AGENTIC_WARMUP_GRACE_PERIOD. - REPLAY_CMD+=" --warmup-grace-period ${AGENTIC_WARMUP_GRACE_PERIOD}" - if [ -n "${AIPERF_EXTRA_INPUTS:-}" ]; then - REPLAY_CMD+=" --extra-inputs $AIPERF_EXTRA_INPUTS" - fi - # Dynamo's KV router needs an explicit session binding to keep later turns on - # the prefill worker owning their prefix blocks; X-Correlation-ID alone does - # not establish it. This flag emits nvext.session_control, which dynamo builds - # after #9920 (v1.3.0-dev) reject with 400; recipes on those builds set - # AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID=true (header routing) - # or AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING=0. - if [[ "${FRAMEWORK:-}" == dynamo-* \ - && "${AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING}" != "0" \ - && "${AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID}" != "true" ]]; then - REPLAY_CMD+=" --use-dynamo-conv-aware-routing" - # The upstream 300s affinity TTL is shorter than an overloaded agentic - # request; this is the router's inactivity lease, not an HTTP timeout. - REPLAY_CMD+=" --dynamo-session-timeout-seconds ${AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS}" - fi - # The dataset manager loads the tokenizer regardless of - # --use-server-token-count, and Kimi checkpoints ship a custom tokenizer - # that needs trust_remote_code. Benign for other models. - REPLAY_CMD+=" --tokenizer-trust-remote-code" - # The WEKA corpus has a few traces longer than smaller-context servers - # accept; replayed unfiltered they become deterministic 4xxs that still - # pressure the engine while queued. - if [ -n "${MAX_MODEL_LEN:-}" ] && [ "$MAX_MODEL_LEN" != "0" ]; then - REPLAY_CMD+=" --max-context-length $MAX_MODEL_LEN" - fi - # Multi-node launchers pass every worker's Prometheus endpoint; AIPerf takes - # several values after one --server-metrics flag and keeps endpoint_url per series. - if [ -n "${AIPERF_SERVER_METRICS_URLS:-}" ]; then - local metrics_url - local -a metrics_urls - IFS=',' read -r -a metrics_urls <<< "$AIPERF_SERVER_METRICS_URLS" - REPLAY_CMD+=" --server-metrics" - for metrics_url in "${metrics_urls[@]}"; do - if [ -z "$metrics_url" ] || [[ "$metrics_url" == *[[:space:]]* ]]; then - echo "ERROR: AIPERF_SERVER_METRICS_URLS must be a comma-separated list of non-empty URLs" >&2 - return 1 - fi - REPLAY_CMD+=" $metrics_url" - done - fi - REPLAY_CMD+=" --output-artifact-dir $result_dir/aiperf_artifacts" - # The scenario enforces a 900s minimum duration; shorter smoke tests need - # --unsafe-override and are flagged submission_valid=false. - if [ "$duration" -lt 900 ] || [ "${AIPERF_UNSAFE_OVERRIDE}" = "true" ]; then - REPLAY_CMD+=" --unsafe-override" - fi - REPLAY_CMD+=" $TRACE_SOURCE_FLAG" -} - -run_agentic_replay_and_write_outputs() ( - check_env_vars ENABLE_AGENTX_POWER IS_MULTINODE REQUIRE_POWER - local result_dir="$1" - local replay_rc - local validation_rc - local power_rc=0 - local agentx_power_enabled=0 - local agentx_multinode_power_enabled=0 - local agentx_multinode_contract_missing=0 - local agentx_monitor_stopped=1 - - case "${ENABLE_AGENTX_POWER}" in - 1|true|TRUE|yes|YES) - if [ "${IS_MULTINODE}" = "true" ]; then - if [ -n "${SRT_MEASUREMENT_WINDOW_DIR:-}" ]; then - agentx_multinode_power_enabled=1 - else - agentx_multinode_contract_missing=1 - fi - else - check_env_vars TP PP_SIZE PCP_SIZE - agentx_power_enabled=1 - fi - ;; - esac - - _stop_agentx_power_monitor() { - if [ "$agentx_monitor_stopped" = "0" ]; then - agentx_monitor_stopped=1 - stop_gpu_monitor - fi - } - - _write_agentx_multinode_window() { - local state="$1" - local -a power_args - power_args=( - --result-dir "$result_dir" - --concurrency "${CONC:?CONC must be set for multinode AgentX power}" - --write-multinode-window "$state" - ) - case "${REQUIRE_POWER}" in - 1|true|TRUE|yes|YES) power_args+=(--require-power) ;; - esac - ( - cd "$INFMAX_CONTAINER_WORKSPACE" - "$AIPERF_PYTHON" -m infx.results.agentic.power_adapter "${power_args[@]}" - ) - } - - if [ "$agentx_power_enabled" = "1" ] || [ "$agentx_multinode_power_enabled" = "1" ]; then - # AIPerf exports naive local datetimes and SMI the same host wall clock; - # the adapter needs the offset to normalize the profiling window. - date +%z > "$result_dir/agentic_power_timezone_offset.txt" - fi - - if [ "$agentx_multinode_power_enabled" = "1" ]; then - set +e - _write_agentx_multinode_window running - power_rc=$? - set -e - if [ "$power_rc" -ne 0 ]; then - echo "ERROR: failed to publish the AgentX formal running power window" >&2 - return "$power_rc" - fi - fi - - if [ "$agentx_power_enabled" = "1" ]; then - start_gpu_monitor --output "$result_dir/gpu_metrics.csv" - agentx_monitor_stopped=0 - # This function runs in a subshell, so these traps cannot clobber - # launcher-owned ones; the stopped flag keeps cleanup idempotent. - trap '_stop_agentx_power_monitor' EXIT - trap '_stop_agentx_power_monitor; exit 130' INT - trap '_stop_agentx_power_monitor; exit 143' TERM - fi - - echo "$REPLAY_CMD" > "$result_dir/benchmark_command.txt" - - set +e - set -x - run_server_client $REPLAY_CMD 2>&1 | tee "$result_dir/benchmark.log" - replay_rc=${PIPESTATUS[0]} - set +x - set -e - - if [ "$agentx_power_enabled" = "1" ]; then - _stop_agentx_power_monitor - trap - EXIT INT TERM - fi - - write_agentic_result_json "$result_dir" - - if [ "$agentx_multinode_power_enabled" = "1" ] && [ "$replay_rc" -eq 0 ]; then - set +e - _write_agentx_multinode_window completed - power_rc=$? - set -e - fi - - if [ "$agentx_power_enabled" = "1" ] || [ "$agentx_multinode_contract_missing" = "1" ]; then - local expected_num_gpus - local -a power_args - power_args=( - --result-dir "$result_dir" - --agg-result "${AGENTIC_OUTPUT_DIR:-$INFMAX_CONTAINER_WORKSPACE}/$RESULT_FILENAME.json" - ) - if [ "$agentx_multinode_contract_missing" = "1" ]; then - power_args+=(--multinode-contract-missing) - else - expected_num_gpus=$((TP * PP_SIZE * PCP_SIZE)) - power_args+=(--expected-num-gpus "$expected_num_gpus") - fi - case "${REQUIRE_POWER}" in - 1|true|TRUE|yes|YES) power_args+=(--require-power) ;; - esac - set +e - ( - cd "$INFMAX_CONTAINER_WORKSPACE" - "$AIPERF_PYTHON" -m infx.results.agentic.power_adapter "${power_args[@]}" - ) - power_rc=$? - set -e - fi - - PYTHONPATH="$INFMAX_CONTAINER_WORKSPACE${PYTHONPATH:+:$PYTHONPATH}" "$AIPERF_PYTHON" -m infx.results.agentic.analyze_benchmark_distributions \ - "$result_dir/aiperf_artifacts" -o "$result_dir" 2>&1 || true - - set +e - ( - cd "$INFMAX_CONTAINER_WORKSPACE" - "$AIPERF_PYTHON" -m infx.results.agentic.validate_agentic_result \ - "$result_dir/aiperf_artifacts" \ - --failed-request-threshold "$AIPERF_FAILED_REQUEST_THRESHOLD" - ) - validation_rc=$? - set -e - - if [ "$replay_rc" -ne 0 ]; then - echo "ERROR: agentic trace replay exited with code $replay_rc after writing available results" >&2 - return "$replay_rc" - fi - - if [ "$validation_rc" -ne 0 ]; then - echo "ERROR: agentic trace replay produced invalid results after writing available artifacts" >&2 - return "$validation_rc" - fi - - if [ "$power_rc" -ne 0 ]; then - echo "ERROR: AgentX power validation failed after writing audit artifacts" >&2 - return "$power_rc" - fi - - validate_required_agentic_server_metrics "$result_dir" -) diff --git a/inferencex-e2e/benchmarks/check_env.sh b/inferencex-e2e/benchmarks/check_env.sh new file mode 100644 index 0000000000..27df76f2bb --- /dev/null +++ b/inferencex-e2e/benchmarks/check_env.sh @@ -0,0 +1,24 @@ +#!/usr/bin/env bash + +# Required-input validation for Bash callers. Sourcing this file only defines +# check_env_vars; it has no other side effects. + +# Usage: check_env_vars VAR1 VAR2 ...; exits 1 listing every name that is unset or empty. +check_env_vars() { + local missing_vars=() + local var_name + + for var_name in "$@"; do + if [[ -z "${!var_name:-}" ]]; then + missing_vars+=("$var_name") + fi + done + + if [[ ${#missing_vars[@]} -gt 0 ]]; then + echo "Error: The following required environment variables are not set:" + for var_name in "${missing_vars[@]}"; do + echo " - $var_name" + done + exit 1 + fi +} diff --git a/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh b/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh index 7a2cce9397..87ea9cb7fc 100644 --- a/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh +++ b/inferencex-e2e/benchmarks/multi_node/runtime_settings.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # Source at the workflow boundary before master-config additional-settings. -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only +source "$(dirname "${BASH_SOURCE[0]}")/../check_env.sh" check_env_vars FRAMEWORK case "$FRAMEWORK" in diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh index e9906ebb2c..87e890ad2e 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/glm5.3-tilert-rocm.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash set -eo pipefail -source /infmax-workspace/benchmarks/benchmark_lib.sh --validation-only +source /infmax-workspace/benchmarks/check_env.sh check_env_vars TILERT_VERSION TILERT_ROLE case "$TILERT_ROLE" in prefill|decode|router) ;; diff --git a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/minimaxm3-trtllm-agentx.sh b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/minimaxm3-trtllm-agentx.sh index 8e06d185d5..340de7a151 100755 --- a/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/minimaxm3-trtllm-agentx.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt-slurm-recipes/configs/minimaxm3-trtllm-agentx.sh @@ -2,7 +2,7 @@ # Prepare a TensorRT-LLM 1.3 worker for MiniMax-M3 AgentX: accept OpenAI's # store=false chat field. Every rank runs this; the lock serializes ranks that # share a container, and the patch is a no-op once applied. -set -euo pipefail +set -eo pipefail exec 9>/tmp/minimaxm3-trtllm-agentx.lock flock 9 python3 /infmax-workspace/runners/patch_trtllm_chat_store.py diff --git a/inferencex-e2e/benchmarks/multi_node/srt_eval.sh b/inferencex-e2e/benchmarks/multi_node/srt_eval.sh index 898401030d..69a76297a4 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt_eval.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt_eval.sh @@ -3,59 +3,21 @@ # SPDX-FileCopyrightText: Copyright (c) 2026 SemiAnalysis LLC. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -# Accuracy evaluation using InferenceX benchmark_lib. -# Requires: endpoint infmax_workspace and explicit workflow metadata. - -set -e - +# srt-slurm multi-node post_eval: srt_eval.sh ENDPOINT INFMAX_WORKSPACE. +# The eval artifacts land in /logs/eval_results, which the launcher collects. +set -eo pipefail if [[ $# -ne 2 || -z "$1" || -z "$2" ]]; then echo "Usage: $0 endpoint infmax_workspace" >&2 exit 1 fi - -ENDPOINT=$1 -INFMAX_WORKSPACE=$2 - -HOST=$(echo "$ENDPOINT" | sed -E 's|https?://||; s|:.*||') -PORT=$(echo "$ENDPOINT" | sed -E 's|.*:([0-9]+).*|\1|') - -echo "Eval Config: endpoint=${ENDPOINT}; host=${HOST}; port=${PORT}; workspace=${INFMAX_WORKSPACE}" - -# cd to workspace so that relative paths (e.g., infx/evals/*.yaml) resolve -cd "${INFMAX_WORKSPACE}" - -source "${INFMAX_WORKSPACE}/benchmarks/benchmark_lib.sh" - -# The workflow supplies topology and concurrency; srt-slurm supplies MODEL_NAME -# from the recipe's served model name. Missing inputs are configuration errors. +source "$2/benchmarks/check_env.sh" +# The workflow supplies the topology; srt-slurm supplies MODEL_NAME (served) and EVAL_CONC. check_env_vars \ - IS_MULTINODE MODEL_NAME EVAL_CONC PREFILL_TP PREFILL_EP \ - PREFILL_DP_ATTN DECODE_DP_ATTN - -# Translate the explicit workflow names to benchmark_lib's metadata names. -export EVAL_CONCURRENT_REQUESTS="$EVAL_CONC" -export TP="$PREFILL_TP" -export CONC="$EVAL_CONC" -export EP_SIZE="$PREFILL_EP" -export DP_ATTENTION="$PREFILL_DP_ATTN" -export PREFILL_DP_ATTENTION="$PREFILL_DP_ATTN" -export DECODE_DP_ATTENTION="$DECODE_DP_ATTN" - -echo "Running evaluation for ${MODEL_NAME} with concurrent-requests=${EVAL_CONCURRENT_REQUESTS}..." -eval_rc=0 -run_eval --port "$PORT" || eval_rc=$? - -echo "Generating lm-eval summary..." -append_lm_eval_summary || true - -mkdir -p /logs/eval_results -echo "Copying eval artifacts to /logs/eval_results/..." -cp -v meta_env.json /logs/eval_results/ 2>/dev/null || true -stage_eval_artifacts /logs/eval_results "$PWD" || true - -if [[ "$eval_rc" -ne 0 ]]; then - echo "Evaluation failed with exit code ${eval_rc}" - exit "$eval_rc" + IS_MULTINODE MODEL_NAME EVAL_CONC PREFILL_TP PREFILL_EP PREFILL_DP_ATTN DECODE_DP_ATTN +if [[ "$IS_MULTINODE" != true ]]; then + echo "ERROR: multi-node eval requires IS_MULTINODE=true" >&2 + exit 1 fi - -echo "Evaluation complete" +cd "$2" +PYTHONSAFEPATH=1 PYTHONPATH="$2${PYTHONPATH:+:$PYTHONPATH}" exec python3 -m infx.bench eval \ + --endpoint "$1" --concurrency "$EVAL_CONC" --stage-to /logs/eval_results diff --git a/inferencex-e2e/benchmarks/multi_node/srt_fixed_sequence.sh b/inferencex-e2e/benchmarks/multi_node/srt_fixed_sequence.sh index 6353141621..58e8aefebf 100644 --- a/inferencex-e2e/benchmarks/multi_node/srt_fixed_sequence.sh +++ b/inferencex-e2e/benchmarks/multi_node/srt_fixed_sequence.sh @@ -1,61 +1,5 @@ #!/usr/bin/env bash - -# Fixed-sequence client for multi-node srt-slurm recipes: SRT owns the servers; -# this runs the InferenceX client once per concurrency and writes the result -# layout that copy_fixed_sequence_results collects. -set -eo pipefail -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only -check_env_vars ISL OSL SRT_FRONTEND_HOST SRT_FRONTEND_PORT CONC_LIST \ - PREFILL_NUM_WORKERS PREFILL_TP DECODE_NUM_WORKERS DECODE_TP -CLIENT_ARGS=(--trust-remote-code) -case "${CLIENT_BACKEND:=openai}" in - openai) endpoint=/v1/completions ;; - openai-chat) endpoint=/v1/chat/completions ;; - *) echo "ERROR: unsupported CLIENT_BACKEND: $CLIENT_BACKEND" >&2; exit 1 ;; -esac -case "${USE_CHAT_TEMPLATE:=true}" in - true) CLIENT_ARGS+=(--use-chat-template) ;; - false) ;; - *) echo "ERROR: USE_CHAT_TEMPLATE must be true or false" >&2; exit 1 ;; -esac - -repo_root="$(dirname "${BASH_SOURCE[0]}")/../.." -# Request the name the workers registered; the workflow's MODEL is the HF id, which can differ. -if [[ -n "${BENCHMARK_SERVED_MODEL_NAME:-}" ]]; then - model="$BENCHMARK_SERVED_MODEL_NAME" -else - model=$(curl -sf "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}/v1/models" | - python3 -c 'import json, sys; print(json.load(sys.stdin)["data"][0]["id"])') -fi -result_dir="/logs/sa-bench_isl_${ISL}_osl_${OSL}" -mkdir -p "$result_dir" -ctx=$((PREFILL_NUM_WORKERS * PREFILL_TP)) -gen=$((DECODE_NUM_WORKERS * DECODE_TP)) -for concurrency in $CONC_LIST; do - result="results_concurrency_${concurrency}_gpus_$((ctx + gen))_ctx_${ctx}_gen_${gen}.json" - PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench_serving.benchmark_serving \ - --backend "$CLIENT_BACKEND" \ - --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ - --endpoint "$endpoint" \ - --model "$model" \ - --tokenizer "${TOKENIZER:-$model}" \ - --dataset-name random \ - --random-input-len "$ISL" \ - --random-output-len "$OSL" \ - --random-range-ratio "${RANDOM_RANGE_RATIO:-0.8}" \ - --random-num-workers 1 \ - --num-warmups "$((concurrency * 2))" \ - --num-prompts "${NUM_PROMPTS:-$((concurrency * 10))}" \ - --max-concurrency "$concurrency" \ - --request-rate inf \ - --ignore-eos \ - --disable-tqdm \ - --save-result \ - --result-dir "$result_dir" \ - --result-filename "$result" \ - "${CLIENT_ARGS[@]}" - # Power lanes: tell srt-slurm which interval this concurrency's result measured. - if [[ -n "${SRT_MEASUREMENT_WINDOW_DIR:-}" ]]; then - PYTHONPATH="$repo_root" python3 -m infx.results.power.window "$result_dir/$result" "$concurrency" - fi -done +# srt-slurm multi-node fixed-sequence client, one point per CONC_LIST value; see infx/bench/fixed_seq.py. +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" || exit 1 +exec env PYTHONSAFEPATH=1 PYTHONDONTWRITEBYTECODE=1 PYTHONPATH="$root${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.bench fixed-seq srt-sweep --logs-dir /logs "$@" diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh index c033b89901..d01cbc2ad5 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/dsv4_fp4_b300_vllm.sh @@ -11,7 +11,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -101,8 +102,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -142,7 +141,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -194,8 +194,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh index 8cd4255c67..2ef26539f0 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/dsv4dspark_fp4_b300_vllm.sh @@ -15,7 +15,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DRAFT_SAMPLE_METHOD MODEL MODEL_PATH MTP_LIST \ OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -146,8 +147,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -190,9 +189,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - # wait_for_server_ready exits the shell (rather than returning) when the server - # dies; the subshell keeps that exit local so one bad cell does not abort the matrix. - if ! (wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"); then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -244,8 +242,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh index b00044dbc2..62e209309e 100644 --- a/inferencex-e2e/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/glm52_fp4_b300_vllm.sh @@ -17,7 +17,8 @@ # below. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -107,8 +108,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -146,7 +145,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -198,8 +198,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh index 2fb2c0865b..6a31fcd1b4 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm.sh @@ -16,7 +16,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -170,8 +171,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -220,7 +219,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -276,8 +276,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh b/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh index 997ba45506..1aa5b5124f 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/kimik3_fp4_b300_vllm_probabilistic_sample_method_block_rejection_sample_method.sh @@ -20,7 +20,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -174,8 +175,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -224,7 +223,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode dspark=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -280,8 +280,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh index fc5ab0c968..82b7df39b8 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/minimaxm3_fp4_b300_vllm.sh @@ -23,7 +23,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -118,8 +119,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -156,7 +155,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode eagle3=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -209,8 +209,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh index 70ce4e342c..f62b279992 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.5_fp4_b300_vllm.sh @@ -16,7 +16,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -106,8 +107,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -152,7 +151,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -208,8 +208,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh b/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh index 4200d21d26..e1fe0c7a3f 100755 --- a/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh +++ b/inferencex-e2e/benchmarks/single_node/speedbench/qwen3.8next_fp4_b300_vllm.sh @@ -36,7 +36,8 @@ # Required collection settings come from speedbench-al.yml. set -o pipefail -source "$(dirname "$0")/../../benchmark_lib.sh" +repo_root="$(cd "$(dirname "$0")/../../.." && pwd)" +source "$repo_root/benchmarks/check_env.sh" check_env_vars \ CATEGORY CHAT_TEMPLATE_KWARGS_ON DP_ATTENTION EP_SIZE MODEL MODEL_PATH \ MTP_LIST OUT_YAML PORT SPEEDBENCH_OUTPUT_LEN THINKING_MODES TP @@ -150,8 +151,6 @@ cleanup_server() { } trap 'cleanup_server' EXIT -start_gpu_monitor - declare -A AL_RESULT run_cell() { @@ -198,7 +197,8 @@ run_cell() { vllm serve "$SERVE_MODEL" "${serve_args[@]}" > "$server_log" 2>&1 & SERVER_PID=$! - if ! wait_for_server_ready --port "$PORT" --server-log "$server_log" --server-pid "$SERVER_PID"; then + if ! PYTHONSAFEPATH=1 PYTHONPATH="$repo_root${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench wait \ + --url "http://0.0.0.0:${PORT}/health" --pid "$SERVER_PID" --log "$server_log"; then echo " -> server failed to start (thinking=$mode mtp=$mtp), recording N/A" AL_RESULT["${mode}_${mtp}"]="N/A" cleanup_server @@ -254,8 +254,6 @@ for mode in $THINKING_MODES; do done done -stop_gpu_monitor - emit_mode_block() { local mode="$1" for mtp in $MTP_LIST; do diff --git a/inferencex-e2e/benchmarks/single_node/srt_eval.sh b/inferencex-e2e/benchmarks/single_node/srt_eval.sh index 33591203ed..8e61a3798d 100644 --- a/inferencex-e2e/benchmarks/single_node/srt_eval.sh +++ b/inferencex-e2e/benchmarks/single_node/srt_eval.sh @@ -1,6 +1,7 @@ #!/usr/bin/env bash -# SRT owns readiness and lifecycle; InferenceX owns evaluation and its artifacts. +# srt-slurm single-node post_eval: srt_eval.sh ENDPOINT STATUS_FILE. The artifacts land in this +# checkout; srt-slurm ignores a failed eval, so its exit code goes to STATUS_FILE for the launcher. set -eo pipefail if [[ $# != 2 || -z "$1" || -z "$2" ]]; then echo "Usage: $0 endpoint status-file" >&2 @@ -9,28 +10,21 @@ fi SRT_EVAL_STATUS_FILE="$2" trap 'rc=$?; printf "%s\n" "$rc" > "$SRT_EVAL_STATUS_FILE"' EXIT -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +source "$root/benchmarks/check_env.sh" check_env_vars MODEL MODEL_NAME CONC TP EP_SIZE DP_ATTENTION IS_MULTINODE IS_AGENTIC -# AgentX evaluates at the native context with the workflow's eval framework. -eval_args=() +# Fixed-sequence evals fit the benchmark context; AgentX evaluates at the native context. if [[ "$IS_AGENTIC" != 1 ]]; then check_env_vars MAX_MODEL_LEN - eval_args=(--framework lm-eval) fi -export PORT="${1##*:}" -if [[ ! "$PORT" =~ ^[1-9][0-9]*$ || "$IS_MULTINODE" != false ]]; then - echo "ERROR: single-node eval requires a local endpoint and single-node metadata" >&2 +if [[ "$IS_MULTINODE" != false ]]; then + echo "ERROR: single-node eval requires IS_MULTINODE=false" >&2 exit 1 fi -cd "$INFERENCEX_REPO_ROOT" +# srt-slurm mounts the served checkpoint here; the workflow's MODEL_PATH is a host path. if [[ -d /model ]]; then export MODEL_PATH=/model fi - -eval_rc=0 -run_eval "${eval_args[@]}" --port "$PORT" || eval_rc=$? -# AgentX eval-only run_eval already staged and removed its results. -if [[ "$IS_AGENTIC" != 1 ]]; then - append_lm_eval_summary || eval_rc=1 -fi -exit "$eval_rc" +cd "$root" +PYTHONSAFEPATH=1 PYTHONPATH="$root${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.bench eval --endpoint "$1" --concurrency "$CONC" --stage-to "$root" diff --git a/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh b/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh index 0bc79b9ab4..b50a7112ca 100644 --- a/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh +++ b/inferencex-e2e/benchmarks/single_node/srt_fixed_sequence.sh @@ -1,64 +1,5 @@ #!/usr/bin/env bash - -# SRT owns the server lifecycle; retain the existing InferenceX client and sampler. -set -eo pipefail -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" --validation-only -check_env_vars MODEL CONC ISL OSL RANDOM_RANGE_RATIO RESULT_FILENAME RESULT_DIR \ - SRT_FRONTEND_HOST SRT_FRONTEND_PORT RUN_EVAL EVAL_ONLY GPU_MONITOR_INTERVAL USE_CHAT_TEMPLATE FRAMEWORK -for name in RUN_EVAL EVAL_ONLY; do - if [[ "${!name}" != true && "${!name}" != false ]]; then - echo "ERROR: $name must be true or false" >&2 - exit 1 - fi -done -case "$FRAMEWORK" in - sglang|atom) CLIENT_BACKEND=vllm ;; - trt) CLIENT_BACKEND=openai ;; - *) echo "ERROR: unsupported fixed-sequence FRAMEWORK: $FRAMEWORK" >&2; exit 1 ;; -esac -SRT_MONITOR_INTERVAL="$GPU_MONITOR_INTERVAL" -CLIENT_ARGS=() -for argument in "$@"; do - case "$argument" in - --trust-remote-code) CLIENT_ARGS+=("$argument") ;; - *) echo "ERROR: unsupported fixed-sequence argument: $argument" >&2; exit 1 ;; - esac -done -case "$USE_CHAT_TEMPLATE" in - true) CLIENT_ARGS+=(--use-chat-template) ;; - false) ;; - *) echo "ERROR: USE_CHAT_TEMPLATE must be true or false" >&2; exit 1 ;; -esac - -for name in CONC ISL OSL SRT_FRONTEND_PORT GPU_MONITOR_INTERVAL; do - if [[ ! "${!name}" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: $name must be a positive integer" >&2 - exit 1 - fi -done - -if [[ ! -d "$RESULT_DIR" ]]; then - echo "ERROR: RESULT_DIR must be an existing runtime-provided directory" >&2 - exit 1 -fi - -source "$(dirname "${BASH_SOURCE[0]}")/../benchmark_lib.sh" -cd "$INFERENCEX_REPO_ROOT" -pip3 install --break-system-packages sentencepiece datasets pandas - -start_gpu_monitor --output "$RESULT_DIR/gpu_metrics.csv" --interval "$SRT_MONITOR_INTERVAL" -trap 'rc=$?; stop_gpu_monitor; exit "$rc"' EXIT - -run_benchmark_serving \ - --model "$MODEL" \ - --port "$SRT_FRONTEND_PORT" \ - --base-url "http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" \ - --backend "$CLIENT_BACKEND" \ - --input-len "$ISL" \ - --output-len "$OSL" \ - --random-range-ratio "$RANDOM_RANGE_RATIO" \ - --num-prompts "$((CONC * 10))" \ - --max-concurrency "$CONC" \ - --result-filename "$RESULT_FILENAME" \ - --result-dir "$RESULT_DIR" \ - "${CLIENT_ARGS[@]}" +# srt-slurm single-node fixed-sequence client; see infx/bench/fixed_seq.py (srt-single). +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" || exit 1 +exec env PYTHONSAFEPATH=1 PYTHONDONTWRITEBYTECODE=1 PYTHONPATH="$root${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.bench fixed-seq srt-single "$@" diff --git a/inferencex-e2e/benchmarks/srt_agentic.sh b/inferencex-e2e/benchmarks/srt_agentic.sh index e4298e64d6..73cb97b277 100644 --- a/inferencex-e2e/benchmarks/srt_agentic.sh +++ b/inferencex-e2e/benchmarks/srt_agentic.sh @@ -1,91 +1,5 @@ #!/usr/bin/env bash -set -eo pipefail -set -x - -# Client-only AgentX trace replay for single- and multi-node srt-slurm jobs. -# srt-slurm owns server startup; this script runs as benchmark.type=custom -# against a fresh, already-ready frontend for exactly one concurrency. - -# Jobs inherit the legacy scripts' /workspace, which srt-slurm does not mount; -# fall back to the repo mount this client runs from. -if [[ ! -f "${INFMAX_CONTAINER_WORKSPACE:-}/benchmarks/benchmark_lib.sh" ]]; then - INFMAX_CONTAINER_WORKSPACE="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" -fi -: "${IS_MULTINODE:=false}" "${PORT:=8000}" -export INFMAX_CONTAINER_WORKSPACE IS_MULTINODE PORT -source "$INFMAX_CONTAINER_WORKSPACE/benchmarks/benchmark_lib.sh" --validation-only -check_env_vars RESULT_DIR EVAL_ONLY CONC -validate_agentic_concurrency "$CONC" -if [[ "${CONC_LIST+x}" ]]; then - validate_agentic_concurrency "$CONC_LIST" - if [[ "$CONC_LIST" != "$CONC" ]]; then - echo "ERROR: CONC must match the single CONC_LIST value" >&2 - exit 1 - fi -fi -source "$INFMAX_CONTAINER_WORKSPACE/benchmarks/benchmark_lib.sh" - -if [[ -n "${SRT_FRONTEND_HOST:-}" ]]; then - check_env_vars SRT_FRONTEND_PORT - export AIPERF_SERVER_URL="http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" -fi - -# benchmark_lib deliberately clears inherited MAX_MODEL_LEN for AgentX so a -# workflow default cannot silently truncate a model's native context. Native -# srt-slurm topologies may still expose a smaller, explicit service limit (for -# example when both P/D roles are configured identically below model-native -# context). Restore that limit only through this dedicated opt-in. -if [[ -n "${AIPERF_MAX_CONTEXT_LENGTH:-}" ]]; then - if ! [[ "$AIPERF_MAX_CONTEXT_LENGTH" =~ ^[1-9][0-9]*$ ]]; then - echo "ERROR: AIPERF_MAX_CONTEXT_LENGTH must be a positive integer" >&2 - exit 1 - fi - export MAX_MODEL_LEN="$AIPERF_MAX_CONTEXT_LENGTH" -fi - -check_env_vars \ - MODEL MODEL_PREFIX FRAMEWORK PRECISION CONC \ - RESULT_FILENAME DURATION - -if [[ -z "${AIPERF_SERVER_URL:-}" ]]; then - if [[ -n "${SRT_FRONTEND_HOST:-}" ]]; then - export AIPERF_SERVER_URL="http://${SRT_FRONTEND_HOST}:${SRT_FRONTEND_PORT}" - else - export AIPERF_SERVER_URL="http://localhost:${PORT}" - fi -fi -echo "Using srt-slurm frontend endpoint: $AIPERF_SERVER_URL" - -# A router frontend does not re-export engine metrics; read them from each worker. -if [[ -z "${AIPERF_SERVER_METRICS_URLS:-}" && "${SRTCTL_FRONTEND_TYPE:-}" != dynamo ]]; then - endpoints="${SRT_AGG_ENDPOINTS:-${SRT_PREFILL_ENDPOINTS:+$SRT_PREFILL_ENDPOINTS,}${SRT_DECODE_ENDPOINTS:-}}" - if [[ -n "${endpoints%,}" ]]; then - AIPERF_SERVER_METRICS_URLS=$(sed -E 's#([^,]+)#http://\1/metrics#g' <<< "${endpoints%,}") - export AIPERF_SERVER_METRICS_URLS - fi -fi - -resolve_trace_source -install_agentic_deps -if [[ "${EVAL_ONLY}" == "true" ]]; then - _wait_for_openai_chat_route --port "$PORT" -fi - -# Keep the multi-node artifact suffix; single-node collection uses the caller's name. -if [[ -n "${CONC_LIST:-}" ]]; then - export RESULT_FILENAME="${RESULT_FILENAME}_conc${CONC}" - RESULT_DIR="${RESULT_DIR}/conc_${CONC}" -fi -mkdir -p "$RESULT_DIR" - -echo "Running agentic concurrency $CONC on this server deployment" -build_replay_cmd "$RESULT_DIR" -# Recipes whose legacy launch rendered prompts client-side opt in here. -if [[ "${AIPERF_APPLY_CHAT_TEMPLATE:-}" == true ]]; then - REPLAY_CMD+=" --apply-chat-template" -fi -# Let this point's admitted responses finish before the deployment is torn down. -if [[ -n "${AIPERF_BENCHMARK_GRACE_PERIOD:-}" ]]; then - REPLAY_CMD+=" --benchmark-grace-period $AIPERF_BENCHMARK_GRACE_PERIOD" -fi -run_agentic_replay_and_write_outputs "$RESULT_DIR" +# srt-slurm AgentX client; infx/srt_slurm/single_node.py recognizes AgentX recipes by this name. +root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" || exit 1 +exec env PYTHONSAFEPATH=1 PYTHONDONTWRITEBYTECODE=1 PYTHONPATH="$root${PYTHONPATH:+:$PYTHONPATH}" \ + python3 -m infx.bench agentic "$@" diff --git a/inferencex-e2e/docs/DOCUMENTATION_PLAN.md b/inferencex-e2e/docs/DOCUMENTATION_PLAN.md index 3ea4a1c307..30d8da9298 100644 --- a/inferencex-e2e/docs/DOCUMENTATION_PLAN.md +++ b/inferencex-e2e/docs/DOCUMENTATION_PLAN.md @@ -86,7 +86,7 @@ Existing detailed references remain in place during migration. A new page should - Add an architecture page with the config-to-result data flow. - Add a configuration page that joins schema, generator, changelog, and validation steps. -- Add a benchmark-development page covering shared Bash helpers, environment propagation, single-node and multi-node paths, and MTP chat-template requirements. +- Add a benchmark-development page covering the shared `infx.bench` container commands, environment propagation, single-node and multi-node paths, and MTP chat-template requirements. - Add diagrams only where they clarify ownership or transitions. ### Phase 3, operations and recovery diff --git a/inferencex-e2e/docs/DOCUMENTATION_PLAN_zh.md b/inferencex-e2e/docs/DOCUMENTATION_PLAN_zh.md index 73d4f3130c..e1a9b79073 100644 --- a/inferencex-e2e/docs/DOCUMENTATION_PLAN_zh.md +++ b/inferencex-e2e/docs/DOCUMENTATION_PLAN_zh.md @@ -60,7 +60,7 @@ | `docs/documentation-procedures.md` | 本次变更完成 | 如何新增、索引、审阅与维护双语文档? | | `docs/architecture.md` | 本次变更完成 | 配置如何变成基准测试结果并发布为一行数据? | | `docs/configuration.md` | 计划中 | 如何修改配置而不破坏 Schema、拓扑与变更日志契约? | -| `docs/benchmark-development.md` | 计划中 | 基准测试脚本、共享 Bash 工具、启动器与运行时环境变量如何协作? | +| `docs/benchmark-development.md` | 计划中 | 基准测试脚本、共享辅助工具、启动器与运行时环境变量如何协作? | | `docs/agentx.md` | 计划中 | AgentX 当前状态、Trace 契约、执行路径与发布边界是什么? | | `docs/workflows-and-sweeps.md` | 计划中 | 如何生成、派发、监控、复用并收集一次扫描? | | `docs/evals.md` | 计划中 | 评估如何选择、运行、打分、校验与收集? | @@ -86,7 +86,7 @@ - 添加介绍配置到结果数据流的架构页面。 - 添加把 Schema、生成器、变更日志与校验步骤串起来的配置页面。 -- 添加基准测试开发页面,说明共享 Bash 工具、环境变量传递、单节点与多节点路径,以及 MTP 的聊天模板要求。 +- 添加基准测试开发页面,说明共享的 `infx.bench` 容器内命令、环境变量传递、单节点与多节点路径,以及 MTP 的聊天模板要求。 - 只在能够澄清所有权或状态转换时使用图示。 ### 第三阶段,运维与恢复 diff --git a/inferencex-e2e/docs/architecture.md b/inferencex-e2e/docs/architecture.md index cc41100340..080c6cd021 100644 --- a/inferencex-e2e/docs/architecture.md +++ b/inferencex-e2e/docs/architecture.md @@ -48,7 +48,7 @@ The repository separates `inferencex-e2e/`, `collectivex/`, `operatorx/`, `share | [`infx/launch/`](../infx/launch) | `python -m infx.launch run`: cluster resolution from the runner name, launch-path (driver) selection, workload policy, signal-safe cleanup, and artifact staging | | [`infx/clusters/`](../infx/clusters), [`infx/launch/backends/`](../infx/launch/backends) | Typed cluster records with one settings model per scheduler, and the scheduler backends that run containers and follow jobs (Slurm with Pyxis squash images today) | | [`runners/srt-slurm/`](../runners/srt-slurm) | srt-slurm host-setup hooks and temporary upstream patches | -| [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh) | Shared server readiness, benchmark client, eval, AgentX replay, and output behavior | +| [`infx/bench/`](../infx/bench) | Container-side `python3 -m infx.bench` commands (`wait`, `fixed-seq`, `agentic`, `eval`) for server readiness, the benchmark client, AgentX replay, and evals | | [`benchmarks/`](../benchmarks) | Framework and topology-specific server and client commands | | [`infx/github.py`](../infx/github.py) | GitHub REST, pagination, and comment reactions shared by workflow operations | | [`infx/workflows/`](../infx/workflows) | Reuse command parsing, authorization lookup, source-run validation, and reaction feedback; the existing reuse CLI remains compatible | @@ -85,7 +85,7 @@ flowchart LR E --> F[run-sweep.yml fan-out] F --> G[Reusable benchmark workflow] G --> H[infx.launch driver] - H --> I[Benchmark script and benchmark_lib] + H --> I[Recipe or script and infx.bench commands] I --> J[Benchmark, eval, logs, metrics, traces] J --> K[Per-job GitHub artifacts] K --> L[Run-level aggregate artifacts] @@ -232,7 +232,7 @@ Depending on the driver, the launcher may: - pass the workflow environment into the runtime container or allocation. - stream the job log, verify the allocation's terminal state, and stage results. -Benchmark scripts under [`benchmarks/`](../benchmarks) own the actual engine and client commands. Most source [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh), which centralizes server readiness, the serving benchmark client, GPU monitoring, lm-eval, SWE-bench, AgentX replay, and stable output helpers. +srt-slurm recipes and the scripts under [`benchmarks/`](../benchmarks) own the actual engine commands. The client side that every lane shares runs inside the serving container as `python3 -m infx.bench ` from [`infx/bench/`](../infx/bench). Its commands are `wait` (server readiness), `fixed-seq` (the serving benchmark client), `agentic` (AgentX replay), and `eval` (lm-eval and the vendor eval runners). They are stdlib-only and Python 3.10 compatible, take their inputs from environment variables or flags, and write the artifact names the collectors read. Recipes reach them through thin shims such as [`benchmarks/srt_agentic.sh`](../benchmarks/srt_agentic.sh) and the `srt_fixed_sequence.sh` and `srt_eval.sh` scripts under `benchmarks/single_node/` and `benchmarks/multi_node/`. Bash callers validate required inputs with `check_env_vars` from [`benchmarks/check_env.sh`](../benchmarks/check_env.sh). The boundary is intentional: a master config stays portable and reviewable, launch mechanics stay in cluster records (see [below](#launch-mechanics-stay-in-cluster-records)), and framework flags stay close to the benchmark recipe, where they can be tested against that engine. On `SIGINT`, `SIGTERM` or `SIGHUP` the launcher runs its registered cleanups, such as cancelling the allocation, and exits with 128 plus the signal number. The first nonzero workload exit code wins over cleanup failures. diff --git a/inferencex-e2e/docs/architecture_zh.md b/inferencex-e2e/docs/architecture_zh.md index 527ef7c3e4..c8310795b2 100644 --- a/inferencex-e2e/docs/architecture_zh.md +++ b/inferencex-e2e/docs/architecture_zh.md @@ -48,7 +48,7 @@ | [`infx/launch/`](../infx/launch) | `python -m infx.launch run`:根据运行器名称解析集群、选择启动路径(驱动)、工作负载策略、信号安全的清理以及工件暂存 | | [`infx/clusters/`](../infx/clusters)、[`infx/launch/backends/`](../infx/launch/backends) | 类型化集群记录(每个调度器一个设置模型),以及运行容器、跟踪作业的调度器后端(目前为使用 Pyxis squash 镜像的 Slurm) | | [`runners/srt-slurm/`](../runners/srt-slurm) | srt-slurm 主机设置 hook 和临时上游补丁 | -| [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh) | 共享的服务器就绪检查、基准测试客户端、评测、AgentX 重放和输出行为 | +| [`infx/bench/`](../infx/bench) | 在容器内运行的 `python3 -m infx.bench` 命令(`wait`、`fixed-seq`、`agentic`、`eval`),负责服务器就绪检查、基准测试客户端、AgentX 重放和评估 | | [`benchmarks/`](../benchmarks) | 特定于框架和拓扑的服务器与客户端命令 | | [`infx/github.py`](../infx/github.py) | 工作流操作共用的 GitHub REST、分页和评论表态基础操作 | | [`infx/workflows/`](../infx/workflows) | 复用命令解析、授权查找、源 Run 验证及表态反馈;现有复用 CLI 保持兼容 | @@ -85,7 +85,7 @@ flowchart LR E --> F[run-sweep.yml 扇出] F --> G[可复用基准测试工作流] G --> H[infx.launch 驱动] - H --> I[基准测试脚本和 benchmark_lib] + H --> I[配方或脚本与 infx.bench 命令] I --> J[基准测试、评测、日志、指标、追踪] J --> K[单作业 GitHub 工件] K --> L[运行级聚合工件] @@ -232,7 +232,7 @@ flowchart LR - 将工作流环境传入运行时容器或分配环境; - 跟踪作业日志、核验分配的最终状态并暂存结果。 -[`benchmarks/`](../benchmarks) 下的基准测试脚本负责实际的引擎和客户端命令。大多数脚本会引入 [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh),后者集中处理服务器就绪检查、服务基准测试客户端、GPU 监控、lm-eval、SWE-bench、AgentX 重放和稳定输出辅助函数。 +srt-slurm 配方和 [`benchmarks/`](../benchmarks) 下的脚本负责实际的引擎命令。各通道共用的客户端逻辑以 `python3 -m infx.bench ` 的形式在服务容器内运行,代码位于 [`infx/bench/`](../infx/bench)。其命令包括 `wait`(服务器就绪检查)、`fixed-seq`(服务基准测试客户端)、`agentic`(AgentX 重放)和 `eval`(lm-eval 与厂商评估运行器)。这些命令只依赖标准库并兼容 Python 3.10,从环境变量或命令行参数读取输入,并写出收集器读取的产物文件名。配方通过薄封装脚本调用它们,例如 [`benchmarks/srt_agentic.sh`](../benchmarks/srt_agentic.sh),以及 `benchmarks/single_node/` 和 `benchmarks/multi_node/` 下的 `srt_fixed_sequence.sh` 与 `srt_eval.sh`。Bash 调用方使用 [`benchmarks/check_env.sh`](../benchmarks/check_env.sh) 中的 `check_env_vars` 校验必需输入。 这一边界是有意设计的:主配置保持可移植且便于审查,启动机制保存在集群记录中(见[下文](#启动机制保存在集群记录中)),框架标志保持靠近基准测试方案,以便针对相应引擎进行测试。收到 `SIGINT`、`SIGTERM` 或 `SIGHUP` 时,启动器会先运行已注册的清理(例如取消分配),再以 128 加信号编号退出;第一个非零的工作负载退出码优先于清理失败。 diff --git a/inferencex-e2e/docs/configuration-procedures.md b/inferencex-e2e/docs/configuration-procedures.md index 89a710f825..c9d289eed5 100644 --- a/inferencex-e2e/docs/configuration-procedures.md +++ b/inferencex-e2e/docs/configuration-procedures.md @@ -226,7 +226,7 @@ Sources: [`AGENTS.md#non-negotiable-benchmark-invariants`](../../AGENTS.md#non-n 1. Confirm native MTP modules versus an external draft. For a draft, verify exact model ID, method (for example `eagle3`), and recommended speculative-token count from the model/upstream recipe. 2. Copy a working sibling for the same model and backend. Preserve its speculative config, attention backend, token count, model patches, and dependency setup. -3. Every speculative fixed-sequence recipe variant must set `benchmark.env.USE_CHAT_TEMPLATE: "true"`; `select_recipe` rejects a speculative variant without it, and [`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh) turns it into `--use-chat-template` for `run_benchmark_serving`. Raw prompts silently depress acceptance. +3. Every speculative fixed-sequence recipe variant must set `benchmark.env.USE_CHAT_TEMPLATE: "true"`; `select_recipe` rejects a speculative variant without it, and [`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh) runs `python3 -m infx.bench fixed-seq srt-single`, which turns it into the benchmark client's `--use-chat-template`. Raw prompts silently depress acceptance. 4. Size graph capture for at least `CONC * (1 + NUM_SPEC_TOKENS)`, rounded as the sibling does and capped at the framework limit (the current vLLM playbook caps at 2048). 5. Keep backend differences: do not copy CUDA-only drafter attention pins or patches into ROCm recipes. 6. Set `spec-decoding: mtp` in the relevant search-space entries and point their `srt-recipe:` at the `-mtp` recipe; `select_recipe` checks the recipe's speculative config against it. For a draft-model mode supported by the schema, use the matching generated value deliberately. Do not infer it from a filename. @@ -368,10 +368,13 @@ out of 141 GiB. Trace corpus: the arm replays the uncapped `semianalysis_cc_traces_weka_062126` corpus, not the 256k-capped `..._062126_256k` variant, because the model serves 1M context. The -recipe never names a corpus — `resolve_trace_source` picks the uncapped default only -because its `dsv4*` case arm also matches the `dsv41flash` prefix. That is load-bearing -and invisible at the call site, and no test pins it (the former `runners/test_dsv41flash_h200.py` -was removed in #3141); narrowing the arm would silently downgrade this recipe's traces. +recipe never names a corpus. `infx.bench.agentic.traces.resolve` picks the uncapped +default only because its `dsv4` family prefix also matches `dsv41flash`. That is +load-bearing and invisible at the call site. +`test_default_corpus_follows_the_model_family_context` in +[`infx/tests/bench/test_agentic_replay.py`](../infx/tests/bench/test_agentic_replay.py) +covers `dsv41flash`, so narrowing the prefix fails that test instead of silently +downgrading this recipe's traces. **The H100 arm is separate.** H100 is not in the upstream hardware table, and the blocker is not the weights. At 1M context the sparse attention indexer allocates a @@ -667,7 +670,7 @@ A configuration is ready for sweep only when the executable files agree, the exa The `dsv41flash-fp4-mi355x-vllm-agentic-dspark` recipe extends [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) to MI355X AgentX: TP4 and TP2, concurrency 1–64, native five-token DSpark. Throughput uses the [committed golden AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml) of 3.51 for thinking on and five draft tokens, with synthetic rejection sampling and adaptive verification disabled. Accuracy evals retain real block rejection but, unlike the CUDA arms, also keep adaptive verification disabled: it trims verification requests on device, which the ROCm `DeepseekV4IndexerBackend` does not support, and the engine refused to start with it enabled ([run 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283)). FP4 describes the MXFP4 experts; the checkpoint also contains MXFP8 weights. -Follow the AMD overrides in the merged [upstream recipe #968](https://github.com/vllm-project/recipes/pull/968): `VLLM_ROCM_USE_AITER=1`, `VLLM_ROCM_USE_AITER_MOE=1`, and `--moe-backend aiter`. The generic AITER selector lets vLLM pick the CK a8w4 experts, matching the DSV4-Pro MI355X recipe. The recipe pins `semianalysis_cc_traces_weka_062126` (the unfiltered corpus) via `WEKA_LOADER_OVERRIDE`. KV stays GPU-resident. Engram stayed on GPU under the upstream AMD defaults until [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) widened the two `is_cuda()` gates to `is_cuda_alike()`. From that commit on, ROCm resolves an `EngramConfig` and `cpu_offload` defaults to on through `VLLM_PLE_CPU_OFFLOAD`, so the recipe sets `--engram-config` explicitly rather than leaning on that default. TP=2 always offloads, since the tables need 94.4 GiB per rank there; TP=4 keeps them resident, since its KV pool is not the constraint through concurrency 64. The recipe likewise trims `--max-num-batched-tokens` only above concurrency 32, to 8192 at TP=2 c64, because the sparse-attention indexer and its companion per-rank buffers grow at roughly 4.4 MiB per batched token. Where that chunk falls below six times the API-server default of 1024 sequences, `--max-num-seqs` is capped at the graph-capture shape: DSpark verifies 1+5 tokens per sequence, and at 4096 against 1024 sequences the engram projection faults during profiling. The rule in every case is to spend device memory on KV only at the concurrencies that ran short of it, leaving the validated low-concurrency settings alone. Images built before that merge still reject the option on ROCm. The MI355X launcher uses the shared HF cache and mounts this model's repository at `/ix`, and exports `INFMAX_CONTAINER_WORKSPACE=/ix` so AgentX dependencies and outputs resolve inside that mount. +Follow the AMD overrides in the merged [upstream recipe #968](https://github.com/vllm-project/recipes/pull/968): `VLLM_ROCM_USE_AITER=1`, `VLLM_ROCM_USE_AITER_MOE=1`, and `--moe-backend aiter`. The generic AITER selector lets vLLM pick the CK a8w4 experts, matching the DSV4-Pro MI355X recipe. The recipe pins `semianalysis_cc_traces_weka_062126` (the unfiltered corpus) via `WEKA_LOADER_OVERRIDE`. KV stays GPU-resident. Engram stayed on GPU under the upstream AMD defaults until [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) widened the two `is_cuda()` gates to `is_cuda_alike()`. From that commit on, ROCm resolves an `EngramConfig` and `cpu_offload` defaults to on through `VLLM_PLE_CPU_OFFLOAD`, so the recipe sets `--engram-config` explicitly rather than leaning on that default. TP=2 always offloads, since the tables need 94.4 GiB per rank there; TP=4 keeps them resident, since its KV pool is not the constraint through concurrency 64. The recipe likewise trims `--max-num-batched-tokens` only above concurrency 32, to 8192 at TP=2 c64, because the sparse-attention indexer and its companion per-rank buffers grow at roughly 4.4 MiB per batched token. Where that chunk falls below six times the API-server default of 1024 sequences, `--max-num-seqs` is capped at the graph-capture shape: DSpark verifies 1+5 tokens per sequence, and at 4096 against 1024 sequences the engram projection faults during profiling. The rule in every case is to spend device memory on KV only at the concurrencies that ran short of it, leaving the validated low-concurrency settings alone. Images built before that merge still reject the option on ROCm. The MI355X launcher uses the shared HF cache, and `srt_agentic.sh` resolves the AgentX client from its own checkout at `/infmax-workspace`. **GPU validation:** The recipe uses `vllm/vllm-openai-rocm:nightly-rocm100-ac9126e58aa7bbab1856ba6593ba4d5003fea516`, a ROCm 10.0 nightly carrying vllm#58671 (paged MXFP4 sparse indexer), vllm#58655 (fused mHC Triton seams) and vllm#53492 (Gluon sparse-MLA kernel). [Run 36824408313](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36824408313) from [#3571](https://github.com/SemiAnalysisAI/InferenceX/pull/3571) validated it across TP4 and TP2 at concurrency 1–64, and its eval-only TP4 concurrency 64 point scored GSM8K 0.9712 strict / 0.9704 flexible. Earlier sweeps qualified superseded pins, so their points do not carry onto this image: [#3420](https://github.com/SemiAnalysisAI/InferenceX/pull/3420) re-swept `nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098`, [#3326](https://github.com/SemiAnalysisAI/InferenceX/pull/3326) qualified `nightly-rocm100-3df4ae153eb385e27b52f26c81f8edb9e20b9984`, and [run 34710937012](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34710937012) covered `nightly-eed1f3d0c6043bd494424a22443ee198dd56f657`. The merged [upstream recipe #1049](https://github.com/vllm-project/recipes/pull/1049) documents the MI355X sparse-MLA Gluon kernel, the MXFP4 sparse indexer and `--block-size 128`; the merged [#1006](https://github.com/vllm-project/recipes/pull/1006) documents the MI355X TP2 Engram offload and `--no-swa-bounded-replay`, and the merged [#968](https://github.com/vllm-project/recipes/pull/968) records the original AMD overrides and the complete InferenceX command. Follow the [AgentX procedure](./eval-agentx-procedures.md#7-run-agentx-fast-feedback-versus-canonical-evidence) for future runtime evidence; local generation and registry metadata alone are not GPU proof. diff --git a/inferencex-e2e/docs/configuration-procedures_zh.md b/inferencex-e2e/docs/configuration-procedures_zh.md index e8441382f7..694d5c633a 100644 --- a/inferencex-e2e/docs/configuration-procedures_zh.md +++ b/inferencex-e2e/docs/configuration-procedures_zh.md @@ -208,7 +208,7 @@ STP(Single Token Prediction,单 Token 预测)是每次前向传播生成 1. 确认使用原生 MTP 模块还是外部 draft。使用 draft 时,从模型/上游配方验证精确模型 ID、方法(例如 `eagle3`)和建议 speculative token 数。 2. 复制相同模型和 backend 的可工作同类项。保留其 speculative config、attention backend、token 数、模型补丁和依赖设置。 -3. 每个投机解码的定长配方变体都必须设置 `benchmark.env.USE_CHAT_TEMPLATE: "true"`;`select_recipe` 会拒绝缺少该设置的投机解码变体,[`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh) 会将其转换为传给 `run_benchmark_serving` 的 `--use-chat-template`。原始 prompt 会静默降低 acceptance。 +3. 每个投机解码的定长配方变体都必须设置 `benchmark.env.USE_CHAT_TEMPLATE: "true"`;`select_recipe` 会拒绝缺少该设置的投机解码变体,[`srt_fixed_sequence.sh`](../benchmarks/single_node/srt_fixed_sequence.sh) 运行 `python3 -m infx.bench fixed-seq srt-single`,后者将其转换为基准测试客户端的 `--use-chat-template`。原始 prompt 会静默降低 acceptance。 4. graph capture 至少按 `CONC * (1 + NUM_SPEC_TOKENS)` 确定规模,采用同类项的取整方式,并限制在框架上限内(当前 vLLM playbook 上限为 2048)。 5. 保留 backend 差异:不要把 CUDA 专用 drafter attention pin 或补丁复制到 ROCm 配方。 6. 在相应搜索空间条目设置 `spec-decoding: mtp`,并将其 `srt-recipe:` 指向 `-mtp` 配方;`select_recipe` 会据此校验配方的 speculative 配置。若使用 schema 支持的 draft-model 模式,要有意设置匹配的生成值;不要根据文件名推断。 @@ -332,10 +332,12 @@ MI355X 分支保持一致。Hopper 没有 FP4 tensor core,因此这些权重 35.9 GiB。 轨迹语料:该分支回放未截断的 `semianalysis_cc_traces_weka_062126` 语料,而不是 256k -截断的 `..._062126_256k` 变体,因为该模型服务 1M 上下文。配方本身并未指定语料 —— -`resolve_trace_source` 选中未截断的默认值,仅仅是因为其 `dsv4*` 分支同时匹配了 -`dsv41flash` 前缀。这一依赖在调用处并不可见却至关重要,且目前没有测试固定它(原 -`runners/test_dsv41flash_h200.py` 已在 #3141 中删除);收窄该分支会静默地降级本配方的轨迹。 +截断的 `..._062126_256k` 变体,因为该模型服务 1M 上下文。配方本身并未指定语料。 +`infx.bench.agentic.traces.resolve` 选中未截断的默认值,仅仅是因为其 `dsv4` 模型族前缀同时匹配了 +`dsv41flash`。这一依赖在调用处并不可见却至关重要。 +[`infx/tests/bench/test_agentic_replay.py`](../infx/tests/bench/test_agentic_replay.py) 中的 +`test_default_corpus_follows_the_model_family_context` 覆盖了 `dsv41flash`,因此收窄该前缀会使 +该测试失败,而不会静默降级本配方的轨迹。 **H100 分支单独实现。** H100 不在上游硬件表中,且瓶颈不在权重。在 1M 上下文下,稀疏 注意力 indexer 会在 `fp8_fp4_paged_mqa_logits` 中分配一个 @@ -602,7 +604,7 @@ python -m pytest infx/tests/matrix/ -v 配方 `dsv41flash-fp4-mi355x-vllm-agentic-dspark` 将 [#2958](https://github.com/SemiAnalysisAI/InferenceX/pull/2958) 扩展至 MI355X AgentX:TP4 与 TP2、并发 1–64、原生五 token DSpark。吞吐测试使用[已提交的黄金 AL](../infx/golden_al_distribution/dsv41flash_dspark.yaml):thinking 开启、五个草稿 token 对应 3.51,采用合成拒绝采样并关闭自适应验证。准确率 eval 保留真实块拒绝采样,但与 CUDA 分支不同,同样关闭自适应验证:它会在设备端裁剪验证请求,而 ROCm 的 `DeepseekV4IndexerBackend` 不支持该操作,启用后引擎拒绝启动([运行 34651830283](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34651830283))。FP4 表示 MXFP4 专家权重;检查点还包含 MXFP8 权重。 -遵循已合并的[上游配方 #968](https://github.com/vllm-project/recipes/pull/968) 中的 AMD 设置:`VLLM_ROCM_USE_AITER=1`、`VLLM_ROCM_USE_AITER_MOE=1` 和 `--moe-backend aiter`。通用 AITER 选择器允许 vLLM 选择 CK a8w4 专家内核,与 DSV4-Pro MI355X 配方一致。配方通过 `WEKA_LOADER_OVERRIDE` 固定使用完整语料 `semianalysis_cc_traces_weka_062126`。KV 驻留 GPU。在 [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) 将两处 `is_cuda()` 判断放宽为 `is_cuda_alike()` 之前,Engram 按上游 AMD 默认设置常驻 GPU。自该提交起,ROCm 会解析 `EngramConfig`,且 `cpu_offload` 经由 `VLLM_PLE_CPU_OFFLOAD` 默认开启,因此配方显式设置 `--engram-config`,而不依赖该默认值。TP=2 始终下放,因为此时表每 rank 需 94.4 GiB;TP=4 始终保持常驻,因为在并发 64 及以下 KV 池并非瓶颈。同样地,配方仅在并发高于 32 时调低 `--max-num-batched-tokens`:TP=2 c64 为 8192,因为稀疏注意力 indexer 及其配套的每 rank 缓冲区按每个批量 token 约 4.4 MiB 增长。当该分块低于 API server 默认 1024 序列所需的六倍时,`--max-num-seqs` 会被限制为 CUDA graph 捕获的规模:DSpark 每序列验证 1+5 个 token,4096 对 1024 序列会使 engram 投影在 profiling 阶段崩溃。所有情况下的原则一致:只在确实出现 KV 不足的并发点上把设备内存让给 KV,保持低并发处已验证的设置不变。早于该合并的镜像在 ROCm 上仍会拒绝该选项。MI355X launcher 使用共享 HF 缓存,并将此模型的仓库挂载至 `/ix`,同时导出 `INFMAX_CONTAINER_WORKSPACE=/ix`,确保 AgentX 依赖与输出路径位于该挂载中。 +遵循已合并的[上游配方 #968](https://github.com/vllm-project/recipes/pull/968) 中的 AMD 设置:`VLLM_ROCM_USE_AITER=1`、`VLLM_ROCM_USE_AITER_MOE=1` 和 `--moe-backend aiter`。通用 AITER 选择器允许 vLLM 选择 CK a8w4 专家内核,与 DSV4-Pro MI355X 配方一致。配方通过 `WEKA_LOADER_OVERRIDE` 固定使用完整语料 `semianalysis_cc_traces_weka_062126`。KV 驻留 GPU。在 [vllm-project/vllm#57491](https://github.com/vllm-project/vllm/pull/57491) 将两处 `is_cuda()` 判断放宽为 `is_cuda_alike()` 之前,Engram 按上游 AMD 默认设置常驻 GPU。自该提交起,ROCm 会解析 `EngramConfig`,且 `cpu_offload` 经由 `VLLM_PLE_CPU_OFFLOAD` 默认开启,因此配方显式设置 `--engram-config`,而不依赖该默认值。TP=2 始终下放,因为此时表每 rank 需 94.4 GiB;TP=4 始终保持常驻,因为在并发 64 及以下 KV 池并非瓶颈。同样地,配方仅在并发高于 32 时调低 `--max-num-batched-tokens`:TP=2 c64 为 8192,因为稀疏注意力 indexer 及其配套的每 rank 缓冲区按每个批量 token 约 4.4 MiB 增长。当该分块低于 API server 默认 1024 序列所需的六倍时,`--max-num-seqs` 会被限制为 CUDA graph 捕获的规模:DSpark 每序列验证 1+5 个 token,4096 对 1024 序列会使 engram 投影在 profiling 阶段崩溃。所有情况下的原则一致:只在确实出现 KV 不足的并发点上把设备内存让给 KV,保持低并发处已验证的设置不变。早于该合并的镜像在 ROCm 上仍会拒绝该选项。MI355X launcher 使用共享 HF 缓存,`srt_agentic.sh` 从其自身所在的检出 `/infmax-workspace` 解析 AgentX 客户端。 **GPU 验证:** 配方使用 `vllm/vllm-openai-rocm:nightly-rocm100-ac9126e58aa7bbab1856ba6593ba4d5003fea516`,该 ROCm 10.0 nightly 包含 vllm#58671(分页 MXFP4 稀疏 indexer)、vllm#58655(融合 mHC Triton seams)与 vllm#53492(Gluon sparse-MLA kernel)。[#3571](https://github.com/SemiAnalysisAI/InferenceX/pull/3571) 的[运行 36824408313](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/36824408313) 在 TP4 与 TP2、并发 1–64 下验证了该镜像,其中仅评测的 TP4 并发 64 数据点 GSM8K 为 strict 0.9712 / flexible 0.9704。此前的 sweep 验证的都是已被取代的镜像,其数据点不能沿用到本镜像:[#3420](https://github.com/SemiAnalysisAI/InferenceX/pull/3420) 重新扫描了 `nightly-rocm100-29468dde8b515031dc6d4d9d06bf0a2fa0442098`,[#3326](https://github.com/SemiAnalysisAI/InferenceX/pull/3326) 验证了 `nightly-rocm100-3df4ae153eb385e27b52f26c81f8edb9e20b9984`,[运行 34710937012](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34710937012) 覆盖的是 `nightly-eed1f3d0c6043bd494424a22443ee198dd56f657`。已合并的[上游配方 #1049](https://github.com/vllm-project/recipes/pull/1049) 记录了 MI355X 的 sparse-MLA Gluon kernel、MXFP4 稀疏 indexer 与 `--block-size 128`;已合并的 [#1006](https://github.com/vllm-project/recipes/pull/1006) 记录了 MI355X 的 TP2 Engram 卸载与 `--no-swa-bounded-replay`,已合并的 [#968](https://github.com/vllm-project/recipes/pull/968) 则记录了最初的 AMD 设置与完整的 InferenceX 命令。后续运行时证据请遵循 [AgentX 流程](./eval-agentx-procedures_zh.md);仅有本地矩阵生成和镜像元数据不能证明 GPU 验证完成。 diff --git a/inferencex-e2e/docs/eval-agentx-procedures.md b/inferencex-e2e/docs/eval-agentx-procedures.md index b3fb76a1b1..875eeb124e 100644 --- a/inferencex-e2e/docs/eval-agentx-procedures.md +++ b/inferencex-e2e/docs/eval-agentx-procedures.md @@ -33,8 +33,8 @@ There are two distinct layers: the matrix generator decides **which jobs exist** | Throughput only | `--no-evals` | No eval jobs | | Selected eval subset only | `--evals-only` | Jobs have `RUN_EVAL=true`, `EVAL_ONLY=true` | | Every eligible eval only | `--all-evals` | Equivalent to `--evals-only --all-evals` and includes all fixed-sequence 8k/1k rows plus single-node and multi-node agentic GSM8K rows | -| Throughput then eval in one recipe | `RUN_EVAL=true`, `EVAL_ONLY=false` | Server starts, throughput runs, then `run_eval` runs | -| Eval against a freshly started server | `RUN_EVAL=true`, `EVAL_ONLY=true` | Launcher expands eval context, skips throughput, and runs the eval | +| Throughput then eval in one recipe | `RUN_EVAL=true`, `EVAL_ONLY=false` | Server starts, throughput runs, then `python3 -m infx.bench eval` runs | +| Eval against a freshly started server | `RUN_EVAL=true`, `EVAL_ONLY=true` | Launcher applies the eval-only server settings, skips throughput, and runs the eval | The PR `all-evals` label instead goes through [`infx.matrix.plan`](../infx/matrix/plan.py), which expands eval selection and keeps throughput. @@ -70,7 +70,7 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p --config-files configs/nvidia-master.yaml | jq . ``` -A correct AgentX eval row contains `"scenario-type": "agentic-coding"`, `"run-eval": true`, and `"eval-only": true`. The workflow splits generated rows into throughput, fixed-sequence eval, and agentic eval jobs in [`.github/workflows/e2e-tests.yml`](../../.github/workflows/e2e-tests.yml#L351-L358). +A correct AgentX eval row contains `"scenario-type": "agentic-coding"`, `"run-eval": true`, and `"eval-only": true`. The workflow splits generated rows into throughput, fixed-sequence eval, and agentic eval jobs in [`.github/workflows/e2e-tests.yml`](../../.github/workflows/e2e-tests.yml#L328-L335). ## 2. Add a graded eval @@ -83,71 +83,75 @@ A correct AgentX eval row contains `"scenario-type": "agentic-coding"`, `"run-ev Against an already healthy OpenAI-compatible server: ```bash -source benchmarks/benchmark_lib.sh export MODEL='' export MODEL_NAME='' export MODEL_PREFIX='' export PORT='' +export EVAL_ONLY=false IS_MULTINODE=false OPENAI_API_KEY=EMPTY export EVAL_TASKS_DIR='infx/evals/.yaml' -export EVAL_CONCURRENT_REQUESTS='16' export EVAL_LIMIT='10' -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency 16 --stage-to "$EVAL_DIR" python3 -m infx.evals.validate_scores \ --model-prefix "$MODEL_PREFIX" \ - --results-glob 'results*.json' + --meta-env "$EVAL_DIR/meta_env.json" \ + --results-glob "$EVAL_DIR/results*.json" ``` For the full eval, unset the limit and repeat against a clean, correctly configured server: ```bash unset EVAL_LIMIT -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores --model-prefix "$MODEL_PREFIX" +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency 16 --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores --model-prefix "$MODEL_PREFIX" \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` -`run_lm_eval` passes concurrency through `num_concurrent` in `--model_args`. It is deliberately an environment variable, not a `run_eval` CLI option. The exact invocation is in [`run_lm_eval()`](../benchmarks/benchmark_lib.sh#L2044-L2128). +Run these with Python 3.10 or newer, normally inside the serving container, because lm-eval pip-installs its pinned harness into that `python3`. The command copies the allow-listed artifacts into `--stage-to` and writes `meta_env.json` there. It takes the concurrency from `--concurrency`, which lm-eval receives as `num_concurrent` in `--model_args`. `EVAL_CONCURRENT_REQUESTS` is no longer read. The exact invocation is in [`infx.bench.eval.lm_eval.run`](../infx/bench/eval/lm_eval.py#L117-L146). ## 3. `EVAL_ONLY` is a launcher contract -Set `EVAL_ONLY=true` **before server launch**. It is not merely a switch inside `run_eval`: +Set `EVAL_ONLY=true` **before server launch**. It is not merely a switch inside the eval command: -1. `compute_eval_context_length` chooses the requested eval context capped by the model's native maximum. -2. The launcher wires it to the server (`--context-length`, `--max-model-len`, or the framework equivalent). -3. The health check still runs. -4. Throughput returns immediately or is skipped. -5. `run_eval` and artifact staging run. +1. For a single-node fixed-sequence job, the srt binder sets the server context to the matrix `MAX_MODEL_LEN` (`isl + osl + 256`) through `context-length` for SGLang, `max_seq_len` and `max_num_tokens` for TRT-LLM, or `max-model-len` for vLLM and ATOM. AgentX points and multi-node jobs keep their recipe's own context, and multi-node jobs can select a real-verification `EVAL_CONFIG_FILE`. +2. The health check still runs. In eval-only jobs, vendor frameworks also wait for the served model on the OpenAI chat route, within `EVAL_ENDPOINT_READY_TIMEOUT_SECONDS`. +3. Throughput is skipped. +4. `python3 -m infx.bench eval` sizes each lm-eval request from `EVAL_MAX_MODEL_LEN`, else from `MAX_MODEL_LEN` capped at the model's native maximum. +5. The same command stages the artifacts and writes `meta_env.json`, whether the eval passed or failed. -Relevant implementation: [context setup](../benchmarks/benchmark_lib.sh#L2016-L2042), [eval dispatch and failure policy](../benchmarks/benchmark_lib.sh#L2893-L3073), and [workflow inputs](../../.github/workflows/benchmark-tmpl.yml#L40-L57). +Relevant implementation: [server context](../infx/srt_slurm/single_node.py#L183-L194), [request budget](../infx/bench/eval/lm_eval.py#L73-L94), [eval dispatch and failure policy](../infx/bench/eval/__init__.py#L74-L173), and [workflow inputs](../../.github/workflows/benchmark-tmpl.yml#L36-L53). Native multi-node post-eval reads the mounted checkpoint at `/model` and enables dataset downloads in the eval process, without changing worker environments. Context lookup reads numeric limits from local `config.json` before falling back to Transformers; an explicit `EVAL_MAX_MODEL_LEN` still takes precedence. -Do not toggle `EVAL_ONLY` after a throughput-sized server is already running and assume the context changed. Restart through the recipe. In eval-only mode an eval failure is returned after available artifacts are staged. In a workflow, upload happens with `always()` before score validation so failed evidence survives ([single-node upload and gate](../../.github/workflows/benchmark-tmpl.yml#L467-L494), [multi-node upload and gate](../../.github/workflows/benchmark-multinode-tmpl.yml#L487-L518)). +Do not toggle `EVAL_ONLY` after a throughput-sized server is already running and assume the context changed. Restart through the recipe. In eval-only mode an eval failure is returned after available artifacts are staged. In a workflow, upload happens with `always()` before score validation so failed evidence survives ([single-node upload and gate](../../.github/workflows/benchmark-tmpl.yml#L449-L472), [multi-node upload and gate](../../.github/workflows/benchmark-multinode-tmpl.yml#L477-L503)). ## 4. Batched eval concurrency -A space-separated `EVAL_CONCURRENT_REQUESTS` value runs several concurrency points **sequentially against one live engine**. It does not run several harnesses simultaneously. Within each point, the harness issues up to that point's concurrency. +A space-separated `--concurrency` value runs several concurrency points **sequentially against one live engine**. Multi-node jobs pass `EVAL_CONC` this way. It does not run several harnesses simultaneously. Within each point, the harness issues up to that point's concurrency. ```bash -source benchmarks/benchmark_lib.sh export MODEL='' MODEL_NAME='' MODEL_PREFIX='' export PORT='' EVAL_TASKS_DIR='infx/evals/gsm8k.yaml' -export EVAL_CONCURRENT_REQUESTS='16 32 64' -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores --expected-concs '16 32 64' +export EVAL_ONLY=false IS_MULTINODE=false OPENAI_API_KEY=EMPTY +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency '16 32 64' --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores --expected-concs '16 32 64' \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` The batch runner creates a fresh temporary output directory per point, stages files with `_conc` suffixes, and writes these arrays to `meta_env.json`: - `eval_concs`: requested points. -- `completed_eval_concs`: eval and staging both succeeded. -- `failed_eval_concs`: either eval or staging failed. +- `completed_eval_concs`: the eval succeeded and staged at least one artifact. +- `failed_eval_concs`: the eval or its staging failed, or it staged nothing. -A failed point is deferred so artifacts from every attempted point can upload. The post-upload validator then fails the job. Batched mode accepts positive integers and supports only `lm-eval`. See [`run_eval` batching](../benchmarks/benchmark_lib.sh#L2980-L3031), [artifact suffixing](../benchmarks/benchmark_lib.sh#L2130-L2188), and [manifest validation](../infx/evals/validate_scores.py#L119-L216). +A failed point is deferred so artifacts from every attempted point can upload. The post-upload validator then fails the job. Batched mode accepts positive integers and supports only `lm-eval`. See [batching](../infx/bench/eval/__init__.py#L176-L211), [artifact suffixing](../infx/bench/eval/stage.py#L20-L43), and [manifest validation](../infx/evals/validate_scores.py#L119-L216). -For multi-node `all-evals`, the workflow constructs `EVAL_CONC` by joining the topology's concurrency list ([dispatch](../../.github/workflows/e2e-tests.yml#L417-L419)). Never compare a point if its `_conc` result or completed-manifest entry is missing. +For multi-node `all-evals`, the workflow constructs `EVAL_CONC` by joining the topology's concurrency list ([dispatch](../../.github/workflows/e2e-tests.yml#L390-L392)). Never compare a point if its `_conc` result or completed-manifest entry is missing. ## 5. Validate scores, not file existence @@ -197,13 +201,13 @@ Retain `meta_env.json`, `results*.json`, and `sample*.jsonl`. The aggregate is a ## 7. Run AgentX: fast feedback versus canonical evidence -[`install_agentic_deps()`](../benchmarks/benchmark_lib.sh) declares the AgentX client dependencies directly alongside the editable `utils/aiperf` install. It installs them into the isolated `AIPERF_RUNTIME_DIR` environment with the caller-supplied `AIPERF_PYTHON_VERSION`. +`python3 -m infx.bench agentic` builds its own client runtime ([`infx/bench/agentic/venv.py`](../infx/bench/agentic/venv.py)). It declares the AgentX client dependencies directly alongside the editable `utils/aiperf` install and installs them into a fresh venv under `AIPERF_RUNTIME_DIR` (default `/inferencex-agentic-`) with the caller-supplied `AIPERF_PYTHON_VERSION`. It then re-runs itself under that venv's Python. Recipes reach it through [`benchmarks/srt_agentic.sh`](../benchmarks/srt_agentic.sh). -AgentX is AIPerf `agentx` trace replay, not a fixed-token synthetic benchmark. The checked-in default uses ten additional warmup requests per trajectory lane and the recipe's configured profile duration. `agentx-fast` forces one warmup request per lane and a 1,200-second profile. It affects single- and multi-node AgentX throughput only. Fixed-sequence throughput and evals remain canonical. Fast runs are not eligible for artifact reuse ([workflow policy](../../.github/workflows/README.md#agentx-fast-mode), [fast replay settings](../benchmarks/benchmark_lib.sh#L3255-L3259)). +AgentX is AIPerf `agentx` trace replay, not a fixed-token synthetic benchmark. The checked-in default uses ten additional warmup requests per trajectory lane and the recipe's configured profile duration. `agentx-fast` forces one warmup request per lane and a 1,200-second profile. It affects single- and multi-node AgentX throughput only. Fixed-sequence throughput and evals remain canonical. Fast runs are not eligible for artifact reuse ([workflow policy](../../.github/workflows/README.md#agentx-fast-mode), [fast replay settings](../infx/bench/agentic/replay.py#L69-L70)). -Every AgentX throughput concurrency runs against a fresh server deployment. The matrix creates a separate job per point; replay clients reject multiple concurrency values. AgentX does not flush caches or reuse a running server for another point. Warmup and profiling for the same point share the deployment. This does not change fixed-sequence sweeps or graded-eval batching. +Every AgentX throughput concurrency runs against a fresh server deployment. The matrix creates a separate job per point. `infx.launch` rejects a multi-node AgentX throughput job whose `CONC_LIST` is not exactly its one positive `CONC`, and the replay client rejects a `CONC_LIST` that differs from `CONC`. AgentX does not flush caches or reuse a running server for another point. Warmup and profiling for the same point share the deployment. This does not change fixed-sequence sweeps or graded-eval batching. -For multi-node srt-slurm jobs, the benchmark client may run on a different host from the frontend. `srt_agentic.sh` uses an explicit `AIPERF_SERVER_URL` when supplied, otherwise derives it from `SRT_FRONTEND_HOST` and `SRT_FRONTEND_PORT`, and falls back to `localhost:$PORT` only when no remote endpoint is available. +For multi-node srt-slurm jobs, the benchmark client may run on a different host from the frontend. The replay targets `http://$SRT_FRONTEND_HOST:$SRT_FRONTEND_PORT` whenever `SRT_FRONTEND_HOST` is set, otherwise an explicit `AIPERF_SERVER_URL`, and falls back to `http://localhost:$PORT` only when neither is available ([`_server_url`](../infx/bench/agentic/replay.py#L119-L126)). Keep non-index engine or router wheels reproducible and immutable: check in the source patch and builder beside the launcher, verify the upstream wheel's digest before patching, assign an explicit local version, and install the published artifact through an exact URL with a SHA256 fragment. A local backport must not use an unreleased upstream version number. @@ -225,11 +229,11 @@ gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ -f agentx-fast=true ``` -Treat fast results as bring-up evidence, never as a replacement for the canonical candidate. A duration below 900 seconds or `AIPERF_UNSAFE_OVERRIDE=true` adds AIPerf's `--unsafe-override` and flags the submission invalid. Use it only for smoke diagnosis ([source](../benchmarks/benchmark_lib.sh#L3362-L3364)). After a fast run is healthy, run the exact candidate canonically before claiming benchmark success. +Treat fast results as bring-up evidence, never as a replacement for the canonical candidate. A duration below 900 seconds or `AIPERF_UNSAFE_OVERRIDE=true` adds AIPerf's `--unsafe-override` and flags the submission invalid. Use it only for smoke diagnosis ([source](../infx/bench/agentic/replay.py#L107-L109)). After a fast run is healthy, run the exact candidate canonically before claiming benchmark success. ## 8. Preserve trace and run provenance -AgentX defaults to recorded assistant-response replay. Live server outputs are measured but discarded when constructing later turns. The selected trace corpus is model-family dependent unless `WEKA_LOADER_OVERRIDE` pins it. The resolver logs both loader and Hugging Face dataset ([trace resolution](../benchmarks/benchmark_lib.sh#L3165-L3234), [replay semantics](../benchmarks/benchmark_lib.sh#L3236-L3366)). +AgentX defaults to recorded assistant-response replay. Live server outputs are measured but discarded when constructing later turns. The selected trace corpus is model-family dependent unless `WEKA_LOADER_OVERRIDE` pins it. The resolver logs both loader and Hugging Face dataset ([trace resolution](../infx/bench/agentic/traces.py#L23-L28), [replay semantics](../infx/bench/agentic/replay.py#L154-L193)). Replays keep the model's native context. The client ignores `MAX_MODEL_LEN`, and only an explicit `AIPERF_MAX_CONTEXT_LENGTH` adds AIPerf's `--max-context-length`. Capture orchestration provenance immediately: @@ -261,7 +265,7 @@ For each concurrency retain: - server/frontend logs and every metrics endpoint represented. - run URL/ID, attempt, head SHA, recipe/config identity, image, topology, fast flag, and any override. -The runner writes the command before replay and validates raw results after aggregation ([execution path](../benchmarks/benchmark_lib.sh#L3412-L3566)). Aggregation preserves dataset provenance and hardware/model/topology fields ([aggregate construction](../infx/results/agentic/__init__.py)). Raw workflow uploads intentionally omit very large `inputs.json` and `profile_export_raw.jsonl`. If those are required for an investigation, preserve them from the live allocation before cleanup ([single-node artifact contract](../../.github/workflows/benchmark-tmpl.yml#L400-L409), [multi-node contract](../../.github/workflows/benchmark-multinode-tmpl.yml#L476-L485)). +The runner writes the command before replay and validates raw results after aggregation ([execution path](../infx/bench/agentic/run.py#L138-L222)). Aggregation preserves dataset provenance and hardware/model/topology fields ([aggregate construction](../infx/results/agentic/__init__.py)). Raw workflow uploads intentionally omit very large `inputs.json` and `profile_export_raw.jsonl`. If those are required for an investigation, preserve them from the live allocation before cleanup ([single-node artifact contract](../../.github/workflows/benchmark-tmpl.yml#L382-L391), [multi-node contract](../../.github/workflows/benchmark-multinode-tmpl.yml#L466-L475)). ## 9. Debug long AgentX runs from live evidence @@ -310,12 +314,12 @@ curl -fsS '' | \ rg -i 'request|queue|cache|token|prefill|decode|error|fail' ``` -Track trends over repeated samples: running/waiting requests, KV usage, prefix hits, input/output token rates, completed/cancelled/errored requests, frontend routing balance, and disaggregated KV transfer. AIPerf records endpoint identity for every server series ([metrics wiring](../benchmarks/benchmark_lib.sh#L3344-L3359)). +Track trends over repeated samples: running/waiting requests, KV usage, prefix hits, input/output token rates, completed/cancelled/errored requests, frontend routing balance, and disaggregated KV transfer. AIPerf records endpoint identity for every server series ([metrics wiring](../infx/bench/agentic/replay.py#L129-L151)). When `AIPERF_SERVER_METRICS_URLS` is unset and `SRTCTL_FRONTEND_TYPE` is not `dynamo`, the replay scrapes each worker's `/metrics` from `SRT_AGG_ENDPOINTS`, or from `SRT_PREFILL_ENDPOINTS` plus `SRT_DECODE_ENDPOINTS`. Use phase markers, not total Slurm age: ```bash -grep -E 'Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|replay_rc=' \ +grep -E 'Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|process_agentic_result' \ "/benchmark.out" date -u ``` diff --git a/inferencex-e2e/docs/eval-agentx-procedures_zh.md b/inferencex-e2e/docs/eval-agentx-procedures_zh.md index 376ae1d9d0..4325ecaba7 100644 --- a/inferencex-e2e/docs/eval-agentx-procedures_zh.md +++ b/inferencex-e2e/docs/eval-agentx-procedures_zh.md @@ -30,8 +30,8 @@ eval modifier,这些组合都会被拒绝。这类运行提供吞吐量证据 | 仅吞吐量 | `--no-evals` | 不生成 eval 作业 | | 仅选定的 eval 子集 | `--evals-only` | 作业带有 `RUN_EVAL=true`、`EVAL_ONLY=true` | | 仅运行所有符合条件的 eval | `--all-evals` | 等价于 `--evals-only --all-evals`;包含全部定长序列 8k/1k 行,以及单节点和多节点 agentic GSM8K 行 | -| 在一个 recipe 中先跑吞吐量再跑 eval | `RUN_EVAL=true`、`EVAL_ONLY=false` | 启动服务,运行吞吐量,然后执行 `run_eval` | -| 对新启动的服务仅运行 eval | `RUN_EVAL=true`、`EVAL_ONLY=true` | launcher 扩大 eval context,跳过吞吐量并运行 eval | +| 在一个 recipe 中先跑吞吐量再跑 eval | `RUN_EVAL=true`、`EVAL_ONLY=false` | 启动服务,运行吞吐量,然后执行 `python3 -m infx.bench eval` | +| 对新启动的服务仅运行 eval | `RUN_EVAL=true`、`EVAL_ONLY=true` | 启动器应用仅评估模式的服务设置,跳过吞吐量并运行评估 | PR 上的 `all-evals` 标签则通过 [`infx.matrix.plan`](../infx/matrix/plan.py) 生成矩阵,会扩大 eval 选择范围并保留吞吐量作业。 @@ -67,7 +67,7 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p --config-files configs/nvidia-master.yaml | jq . ``` -正确的 AgentX eval 行包含 `"scenario-type": "agentic-coding"`、`"run-eval": true` 和 `"eval-only": true`。工作流会在 [`.github/workflows/e2e-tests.yml`](../../.github/workflows/e2e-tests.yml#L351-L358) 中将生成的行拆分到吞吐量、定长序列 eval 和 agentic eval 作业。 +正确的 AgentX eval 行包含 `"scenario-type": "agentic-coding"`、`"run-eval": true` 和 `"eval-only": true`。工作流会在 [`.github/workflows/e2e-tests.yml`](../../.github/workflows/e2e-tests.yml#L328-L335) 中将生成的行拆分到吞吐量、定长序列 eval 和 agentic eval 作业。 ## 2. 添加评分 eval @@ -80,71 +80,75 @@ uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with p 对已经健康的 OpenAI-compatible 服务执行: ```bash -source benchmarks/benchmark_lib.sh export MODEL='' export MODEL_NAME='' export MODEL_PREFIX='' export PORT='' +export EVAL_ONLY=false IS_MULTINODE=false OPENAI_API_KEY=EMPTY export EVAL_TASKS_DIR='infx/evals/.yaml' -export EVAL_CONCURRENT_REQUESTS='16' export EVAL_LIMIT='10' -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency 16 --stage-to "$EVAL_DIR" python3 -m infx.evals.validate_scores \ --model-prefix "$MODEL_PREFIX" \ - --results-glob 'results*.json' + --meta-env "$EVAL_DIR/meta_env.json" \ + --results-glob "$EVAL_DIR/results*.json" ``` 完整 eval 需要取消 limit,并在干净且配置正确的服务上重复执行: ```bash unset EVAL_LIMIT -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores --model-prefix "$MODEL_PREFIX" +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency 16 --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores --model-prefix "$MODEL_PREFIX" \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` -`run_lm_eval` 通过 `--model_args` 中的 `num_concurrent` 传递并发;它刻意采用环境变量,而不是 `run_eval` CLI 选项。准确调用见 [`run_lm_eval()`](../benchmarks/benchmark_lib.sh#L2044-L2128)。 +请使用 Python 3.10 或更高版本运行这些命令,通常在服务容器内执行,因为 lm-eval 会通过 pip 把固定版本的 harness 安装到该 `python3` 中。该命令会把允许列表中的产物复制到 `--stage-to`,并在该目录写入 `meta_env.json`。并发取自 `--concurrency`,lm-eval 通过 `--model_args` 中的 `num_concurrent` 接收该值。命令不再读取 `EVAL_CONCURRENT_REQUESTS`。准确调用见 [`infx.bench.eval.lm_eval.run`](../infx/bench/eval/lm_eval.py#L117-L146)。 ## 3. `EVAL_ONLY` 是 launcher 约定 -必须在**启动服务前**设置 `EVAL_ONLY=true`。它不仅是 `run_eval` 内部的开关: +必须在**启动服务前**设置 `EVAL_ONLY=true`。它不仅是评估命令内部的开关: -1. `compute_eval_context_length` 选择请求的 eval context,并以模型原生上限为界。 -2. launcher 将其连接到服务参数(`--context-length`、`--max-model-len` 或 framework 对应参数)。 -3. 仍会运行健康检查。 -4. 吞吐量路径立即返回或被跳过。 -5. 运行 `run_eval` 和 artifact staging。 +1. 对单节点定长序列作业,srt binder 会把服务上下文设为矩阵中的 `MAX_MODEL_LEN`(`isl + osl + 256`),SGLang 使用 `context-length`,TRT-LLM 使用 `max_seq_len` 和 `max_num_tokens`,vLLM 与 ATOM 使用 `max-model-len`。AgentX 测试点和多节点作业保留配方自身的上下文,多节点作业还可选择用于真实验证的 `EVAL_CONFIG_FILE`。 +2. 仍会运行健康检查。在仅评估作业中,厂商评估框架还会在 `EVAL_ENDPOINT_READY_TIMEOUT_SECONDS` 内等待服务模型出现在 OpenAI chat 路由上。 +3. 跳过吞吐量测试。 +4. `python3 -m infx.bench eval` 按 `EVAL_MAX_MODEL_LEN` 确定每个 lm-eval 请求的预算。未设置时使用 `MAX_MODEL_LEN`,并以模型原生上限为界。 +5. 同一命令会暂存产物并写入 `meta_env.json`,无论评估成功还是失败。 -原生多节点 post-eval 从 `/model` 读取挂载的检查点,并仅在评估进程中启用数据集下载,不改变工作进程环境。上下文查询先读取本地 `config.json` 中的数值上限,再回退到 Transformers;显式设置的 `EVAL_MAX_MODEL_LEN` 仍优先。 +相关实现:[服务上下文](../infx/srt_slurm/single_node.py#L183-L194)、[请求预算](../infx/bench/eval/lm_eval.py#L73-L94)、[评估分派与失败策略](../infx/bench/eval/__init__.py#L74-L173) 和[工作流输入](../../.github/workflows/benchmark-tmpl.yml#L36-L53)。 -相关实现:[context 设置](../benchmarks/benchmark_lib.sh#L2016-L2042)、[eval 分派与失败策略](../benchmarks/benchmark_lib.sh#L2893-L3073) 和[工作流输入](../../.github/workflows/benchmark-tmpl.yml#L40-L57)。 +原生多节点 post-eval 从 `/model` 读取挂载的检查点,并仅在评估进程中启用数据集下载,不改变工作进程环境。上下文查询先读取本地 `config.json` 中的数值上限,再回退到 Transformers;显式设置的 `EVAL_MAX_MODEL_LEN` 仍优先。 -不要在吞吐量规格的服务已经运行后才切换 `EVAL_ONLY`,并假定 context 会随之变化。应通过 recipe 重启。Eval-only 模式会在暂存已有 artifact 后返回 eval 失败;在工作流中,上传步骤使用 `always()`,并位于分数校验前,因此失败证据仍会保留([单节点上传与 gate](../../.github/workflows/benchmark-tmpl.yml#L467-L494)、[多节点上传与 gate](../../.github/workflows/benchmark-multinode-tmpl.yml#L487-L518))。 +不要在吞吐量规格的服务已经运行后才切换 `EVAL_ONLY`,并假定 context 会随之变化。应通过 recipe 重启。Eval-only 模式会在暂存已有 artifact 后返回 eval 失败;在工作流中,上传步骤使用 `always()`,并位于分数校验前,因此失败证据仍会保留([单节点上传与 gate](../../.github/workflows/benchmark-tmpl.yml#L449-L472)、[多节点上传与 gate](../../.github/workflows/benchmark-multinode-tmpl.yml#L477-L503))。 ## 4. 批量 eval 并发 -空格分隔的 `EVAL_CONCURRENT_REQUESTS` 会让多个并发点在**同一个存活的 engine 上依次执行**,而不是同时运行多个 harness。每个并发点内部,harness 最多发出该并发数的请求。 +空格分隔的 `--concurrency` 值会让多个并发点在**同一个存活的 engine 上依次执行**。多节点作业以这种方式传入 `EVAL_CONC`。它不会同时运行多个 harness。每个并发点内部,harness 最多发出该并发数的请求。 ```bash -source benchmarks/benchmark_lib.sh export MODEL='' MODEL_NAME='' MODEL_PREFIX='' export PORT='' EVAL_TASKS_DIR='infx/evals/gsm8k.yaml' -export EVAL_CONCURRENT_REQUESTS='16 32 64' -run_eval --framework lm-eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores --expected-concs '16 32 64' +export EVAL_ONLY=false IS_MULTINODE=false OPENAI_API_KEY=EMPTY +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency '16 32 64' --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores --expected-concs '16 32 64' \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` 批量 runner 会为每个点创建新的临时输出目录,用 `_conc` 后缀暂存文件,并向 `meta_env.json` 写入以下数组: - `eval_concs`:请求的点; -- `completed_eval_concs`:eval 与 staging 均成功的点; -- `failed_eval_concs`:eval 或 staging 失败的点。 +- `completed_eval_concs`:评估成功且至少暂存了一个产物的点; +- `failed_eval_concs`:评估或其暂存失败,或未暂存任何产物的点。 -失败点会延迟报错,使所有已尝试点的 artifact 都能上传;随后 post-upload validator 会使作业失败。批量模式只接受正整数,且仅支持 `lm-eval`。参见 [`run_eval` batching](../benchmarks/benchmark_lib.sh#L2980-L3031)、[artifact 后缀处理](../benchmarks/benchmark_lib.sh#L2130-L2188) 和[manifest 校验](../infx/evals/validate_scores.py#L119-L216)。 +失败点会延迟报错,使所有已尝试点的 artifact 都能上传;随后 post-upload validator 会使作业失败。批量模式只接受正整数,且仅支持 `lm-eval`。参见[批量执行](../infx/bench/eval/__init__.py#L176-L211)、[产物后缀处理](../infx/bench/eval/stage.py#L20-L43) 和[manifest 校验](../infx/evals/validate_scores.py#L119-L216)。 -对于多节点 `all-evals`,工作流通过连接拓扑的并发列表构造 `EVAL_CONC`([分派](../../.github/workflows/e2e-tests.yml#L417-L419))。如果缺少某点的 `_conc` 结果或 completed manifest 条目,绝不能比较该点。 +对于多节点 `all-evals`,工作流通过连接拓扑的并发列表构造 `EVAL_CONC`([分派](../../.github/workflows/e2e-tests.yml#L390-L392))。如果缺少某点的 `_conc` 结果或 completed manifest 条目,绝不能比较该点。 ## 5. 校验分数,而不只是检查文件存在 @@ -194,13 +198,13 @@ gh run download "$RUN_ID" --repo SemiAnalysisAI/InferenceX \ ## 7. 运行 AgentX:快速反馈与 canonical 证据 -[`install_agentic_deps()`](../benchmarks/benchmark_lib.sh) 在安装可编辑模式的 `utils/aiperf` 时直接声明 AgentX client 所需的依赖,并使用调用方提供的 `AIPERF_PYTHON_VERSION` 将它们安装到隔离的 `AIPERF_RUNTIME_DIR` 环境中。 +`python3 -m infx.bench agentic` 会自行构建客户端运行时([`infx/bench/agentic/venv.py`](../infx/bench/agentic/venv.py))。它在安装可编辑模式的 `utils/aiperf` 时直接声明 AgentX client 所需的依赖,并使用调用方提供的 `AIPERF_PYTHON_VERSION` 将它们安装到 `AIPERF_RUNTIME_DIR` 下新建的 venv 中(默认 `/inferencex-agentic-`)。随后它会在该 venv 的 Python 下重新运行自身。Recipe 通过 [`benchmarks/srt_agentic.sh`](../benchmarks/srt_agentic.sh) 调用它。 -AgentX 是 AIPerf `agentx` trace replay,不是固定 token 的合成 benchmark。仓库默认设置对每条 trajectory lane 额外执行十个 warmup 请求,并使用 recipe 配置的 profile 时长。`agentx-fast` 强制每条 lane 只运行一个 warmup 请求,并将 profile 设为 1,200 秒。它只影响单节点和多节点 AgentX 吞吐量;定长序列吞吐量与 eval 保持 canonical。Fast 运行不符合 artifact reuse 条件([工作流策略](../../.github/workflows/README.md#agentx-fast-mode)、[fast replay 设置](../benchmarks/benchmark_lib.sh#L3255-L3259))。 +AgentX 是 AIPerf `agentx` trace replay,不是固定 token 的合成 benchmark。仓库默认设置对每条 trajectory lane 额外执行十个 warmup 请求,并使用 recipe 配置的 profile 时长。`agentx-fast` 强制每条 lane 只运行一个 warmup 请求,并将 profile 设为 1,200 秒。它只影响单节点和多节点 AgentX 吞吐量;定长序列吞吐量与 eval 保持 canonical。Fast 运行不符合 artifact reuse 条件([工作流策略](../../.github/workflows/README.md#agentx-fast-mode)、[fast replay 设置](../infx/bench/agentic/replay.py#L69-L70))。 -每个 AgentX 吞吐量并发点都必须使用新启动的服务。矩阵为每个点生成独立作业,replay client 会拒绝多个并发值。AgentX 不清空缓存,也不复用正在运行的服务来测试另一个并发点。同一测试点的预热和正式测量共用服务。此规则不改变定长序列 sweep 或评分 eval 的批量执行行为。 +每个 AgentX 吞吐量并发点都必须使用新启动的服务。矩阵为每个点生成独立作业。`infx.launch` 会拒绝 `CONC_LIST` 不恰好等于其唯一正整数 `CONC` 的多节点 AgentX 吞吐量作业,replay client 也会拒绝与 `CONC` 不同的 `CONC_LIST`。AgentX 不清空缓存,也不复用正在运行的服务来测试另一个并发点。同一测试点的预热和正式测量共用服务。此规则不改变定长序列 sweep 或评分 eval 的批量执行行为。 -对于多节点 srt-slurm 作业,benchmark client 与 frontend 可能运行在不同主机上。`srt_agentic.sh` 会优先使用显式提供的 `AIPERF_SERVER_URL`;否则从 `SRT_FRONTEND_HOST` 和 `SRT_FRONTEND_PORT` 推导地址;仅在没有远端 endpoint 时回退到 `localhost:$PORT`。 +对于多节点 srt-slurm 作业,benchmark client 与 frontend 可能运行在不同主机上。只要设置了 `SRT_FRONTEND_HOST`,replay 就以 `http://$SRT_FRONTEND_HOST:$SRT_FRONTEND_PORT` 为目标,否则使用显式提供的 `AIPERF_SERVER_URL`,两者都没有时才回退到 `http://localhost:$PORT`([`_server_url`](../infx/bench/agentic/replay.py#L119-L126))。 对于未发布到 package index 的 engine 或 router wheel,必须保证构建可复现且 artifact 不可变:在 launcher 旁签入源码 patch 与构建器,打 patch 前校验上游 wheel 的 digest,分配明确的 local version,并通过带 SHA256 fragment 的精确 URL 安装已发布 artifact。本地 backport 不得冒用尚未发布的上游版本号。 @@ -222,11 +226,11 @@ gh workflow run e2e-tests.yml --repo SemiAnalysisAI/InferenceX --ref "$REF" \ -f agentx-fast=true ``` -Fast 结果只能作为 bring-up 证据,绝不能替代 canonical candidate。小于 900 秒的 duration 或 `AIPERF_UNSAFE_OVERRIDE=true` 会添加 AIPerf 的 `--unsafe-override` 并将 submission 标记为无效;只能用于 smoke 诊断([源码](../benchmarks/benchmark_lib.sh#L3362-L3364))。Fast 运行健康后,必须对完全相同的 candidate 进行 canonical 运行,才能宣称 benchmark 成功。 +Fast 结果只能作为 bring-up 证据,绝不能替代 canonical candidate。小于 900 秒的 duration 或 `AIPERF_UNSAFE_OVERRIDE=true` 会添加 AIPerf 的 `--unsafe-override` 并将 submission 标记为无效;只能用于 smoke 诊断([源码](../infx/bench/agentic/replay.py#L107-L109))。Fast 运行健康后,必须对完全相同的 candidate 进行 canonical 运行,才能宣称 benchmark 成功。 ## 8. 保留 trace 与运行 provenance -AgentX 默认 replay 已记录的 assistant response。实时服务输出会被测量,但构造后续 turn 时会丢弃。除非用 `WEKA_LOADER_OVERRIDE` 固定,否则所选 trace corpus 依赖模型 family;resolver 会同时记录 loader 与 Hugging Face dataset([trace 解析](../benchmarks/benchmark_lib.sh#L3165-L3234)、[replay 语义](../benchmarks/benchmark_lib.sh#L3236-L3366))。 +AgentX 默认 replay 已记录的 assistant response。实时服务输出会被测量,但构造后续 turn 时会丢弃。除非用 `WEKA_LOADER_OVERRIDE` 固定,否则所选 trace corpus 依赖模型 family;resolver 会同时记录 loader 与 Hugging Face dataset([trace 解析](../infx/bench/agentic/traces.py#L23-L28)、[replay 语义](../infx/bench/agentic/replay.py#L154-L193))。Replay 保留模型的原生上下文。客户端忽略 `MAX_MODEL_LEN`,只有显式设置的 `AIPERF_MAX_CONTEXT_LENGTH` 才会添加 AIPerf 的 `--max-context-length`。 立即记录 orchestration provenance: @@ -258,7 +262,7 @@ gh run download "$RUN_ID" --repo SemiAnalysisAI/InferenceX \ - server/frontend 日志以及所代表的每个 metrics endpoint; - run URL/ID、attempt、head SHA、recipe/config 标识、image、topology、fast 标志和所有 override。 -Runner 会在 replay 前写入命令,并在聚合后校验原始结果([执行路径](../benchmarks/benchmark_lib.sh#L3412-L3566))。聚合会保留 dataset provenance 以及硬件/模型/拓扑字段([aggregate 构造](../infx/results/agentic/__init__.py))。工作流的 raw upload 会有意排除体积很大的 `inputs.json` 和 `profile_export_raw.jsonl`;如果调查需要这些文件,应在清理前从实时 allocation 保存([单节点 artifact 约定](../../.github/workflows/benchmark-tmpl.yml#L400-L409)、[多节点约定](../../.github/workflows/benchmark-multinode-tmpl.yml#L476-L485))。 +Runner 会在 replay 前写入命令,并在聚合后校验原始结果([执行路径](../infx/bench/agentic/run.py#L138-L222))。聚合会保留 dataset provenance 以及硬件/模型/拓扑字段([aggregate 构造](../infx/results/agentic/__init__.py))。工作流的 raw upload 会有意排除体积很大的 `inputs.json` 和 `profile_export_raw.jsonl`;如果调查需要这些文件,应在清理前从实时 allocation 保存([单节点 artifact 约定](../../.github/workflows/benchmark-tmpl.yml#L382-L391)、[多节点约定](../../.github/workflows/benchmark-multinode-tmpl.yml#L466-L475))。 ## 9. 用实时证据调试长时间 AgentX 运行 @@ -307,12 +311,12 @@ curl -fsS '' | \ rg -i 'request|queue|cache|token|prefill|decode|error|fail' ``` -通过重复 sample 跟踪趋势:running/waiting request、KV usage、prefix hit、input/output token rate、completed/cancelled/errored request、frontend routing balance,以及 disaggregated KV transfer。AIPerf 会为每条 server series 记录 endpoint identity([metrics 接线](../benchmarks/benchmark_lib.sh#L3344-L3359))。 +通过重复 sample 跟踪趋势:running/waiting request、KV usage、prefix hit、input/output token rate、completed/cancelled/errored request、frontend routing balance,以及 disaggregated KV transfer。AIPerf 会为每条 server series 记录 endpoint identity([metrics 接线](../infx/bench/agentic/replay.py#L129-L151))。未设置 `AIPERF_SERVER_METRICS_URLS` 且 `SRTCTL_FRONTEND_TYPE` 不是 `dynamo` 时,replay 会从 `SRT_AGG_ENDPOINTS`,或从 `SRT_PREFILL_ENDPOINTS` 加 `SRT_DECODE_ENDPOINTS`,抓取每个 worker 的 `/metrics`。 应使用 phase marker,而不是 Slurm 总运行时间: ```bash -grep -E 'Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|replay_rc=' \ +grep -E 'Phase warmup progress|WARMUP cache pressure|Phase warmup complete|Phase profiling started|Phase profiling complete|process_agentic_result' \ "/benchmark.out" date -u ``` diff --git a/inferencex-e2e/docs/recovery-results-procedures.md b/inferencex-e2e/docs/recovery-results-procedures.md index 0a4b1be76f..f282e5eac2 100644 --- a/inferencex-e2e/docs/recovery-results-procedures.md +++ b/inferencex-e2e/docs/recovery-results-procedures.md @@ -58,7 +58,7 @@ Sources: [single-node process/upload](https://github.com/SemiAnalysisAI/Inferenc ### Eval results -Eval jobs upload per-config artifacts named `eval_${EXP_NAME}_${RESULT_FILENAME}`. They contain the files that exist for that evaluator, including `meta_env.json`, `results*.json`, `sample*.jsonl`, and, for supported agentic evaluators, predictions, reports, or trajectories. The workflow behavior is deliberate: +Eval jobs upload per-config artifacts named `eval_${EXP_NAME}_${RESULT_FILENAME}`. They contain the files the eval command staged for that evaluator under the allow-list in [`infx/bench/eval/stage.py`](../infx/bench/eval/stage.py), namely `meta_env.json`, `results*.json`, `sample*.jsonl`, and native vendor-eval reports (`*_report.json`), detailed results (`*_results.jsonl`), and archives (`*_artifacts.tar.gz`). The workflow behavior is deliberate: - an eval-only job errors when no eval files are found. - eval files upload under `always()`, preserving partial evidence from a failed job. @@ -281,16 +281,16 @@ Canonical source: [complete failed-ingest recovery command](../../.claude/comman ### Prevent recurrence -Containers can run as root while the GitHub workspace is bind-mounted. The shared benchmark library prevents root-owned Python cache directories by setting: +Containers can run as root while the GitHub workspace is bind-mounted. The benchmark workflows keep Python bytecode caches out of the workspace by setting these variables for every job: -```bash -export PYTHONDONTWRITEBYTECODE=1 -export PYTHONPYCACHEPREFIX="${PYTHONPYCACHEPREFIX:-/tmp/inferencex-pycache}" +```yaml +PYTHONDONTWRITEBYTECODE: '1' +PYTHONPYCACHEPREFIX: /tmp/inferencex-pycache ``` Do not override these paths back into the workspace. Use the recovery scan below after an `EACCES` cleanup failure, including failures caused by logs left by retired launchers. -Source: [Python-cache prevention](https://github.com/SemiAnalysisAI/InferenceX/blob/0c28706b33d4a796b82f6f9c3594c19c46365575/benchmarks/benchmark_lib.sh#L5-L10). +Source: [Python-cache prevention](../../.github/workflows/benchmark-tmpl.yml#L145-L146). ### Recover an MI355X TW runner workspace @@ -437,20 +437,6 @@ Remaining durable fix: This evidence is the completion gate. “Workflow green” without artifact identity, source/merge identity, and ingest counts is not a verified result recovery. -### AMD multi-node SGLang teardown - -On exit, including a failed startup/readiness check, the AMD SGLang launcher sends -TERM only to its recorded `setsid` process groups. Normal completion stages results -before this cleanup. It allows 30 seconds for graceful -exit, then sends KILL to surviving groups and checks for exit for another five -seconds. This handles orphaned or TERM-resistant workers that otherwise hold log -pipes open. These cleanup deadlines do not change profiling, evaluation, or server -readiness deadlines. A failed client retains its exit status; unresolved cleanup -fails an otherwise successful node. Kernel-blocked processes may still require -separately authorized node repair. Do not change or discard completed metrics to -work around teardown failures. A single EXIT handler owns group cleanup and the -existing UMBP standalone PID cleanup; the latter still runs if group cleanup fails. - ### AMD multi-node GPU preflight coordination The Slurm launcher completes Docker pre-clean and the existing GPU VRAM drain diff --git a/inferencex-e2e/docs/recovery-results-procedures_zh.md b/inferencex-e2e/docs/recovery-results-procedures_zh.md index 88670e1dc8..822e3cce1b 100644 --- a/inferencex-e2e/docs/recovery-results-procedures_zh.md +++ b/inferencex-e2e/docs/recovery-results-procedures_zh.md @@ -56,7 +56,7 @@ DECODE_GPUS="$decode_gpus" \ ### 评测结果 -评测任务上传以 `eval_${EXP_NAME}_${RESULT_FILENAME}` 命名的逐配置制品。制品包含该评测器实际生成的文件,例如 `meta_env.json`、`results*.json`、`sample*.jsonl`;对于受支持的 agentic 评测器,还可能包含 predictions、reports 或 trajectories。工作流的以下行为都是有意设计的: +评测任务上传以 `eval_${EXP_NAME}_${RESULT_FILENAME}` 命名的逐配置制品。制品包含评估命令按 [`infx/bench/eval/stage.py`](../infx/bench/eval/stage.py) 中的允许列表为该评测器暂存的文件,即 `meta_env.json`、`results*.json`、`sample*.jsonl`,以及厂商评估的原生报告(`*_report.json`)、详细结果(`*_results.jsonl`)和归档(`*_artifacts.tar.gz`)。工作流的以下行为都是有意设计的: - eval-only 任务没有任何评测文件时会报错; - 评测文件在 `always()` 条件下上传,以保留失败任务的部分证据; @@ -279,16 +279,16 @@ git diff --check origin/main...HEAD ### 防止复发 -容器可能以 root 身份运行,同时 GitHub 工作区被 bind mount。共享基准库通过以下设置防止工作区中出现 root-owned Python 缓存目录: +容器可能以 root 身份运行,同时 GitHub 工作区被 bind mount。基准测试工作流为每个作业设置以下变量,使 Python 字节码缓存不写入工作区: -```bash -export PYTHONDONTWRITEBYTECODE=1 -export PYTHONPYCACHEPREFIX="${PYTHONPYCACHEPREFIX:-/tmp/inferencex-pycache}" +```yaml +PYTHONDONTWRITEBYTECODE: '1' +PYTHONPYCACHEPREFIX: /tmp/inferencex-pycache ``` 不要把这些路径重新覆盖到工作区。出现 `EACCES` 清理错误后,应执行下述恢复扫描,包括由已退役启动器遗留的日志导致的错误。 -来源:[Python 缓存预防](https://github.com/SemiAnalysisAI/InferenceX/blob/0c28706b33d4a796b82f6f9c3594c19c46365575/benchmarks/benchmark_lib.sh#L5-L10)。 +来源:[Python 缓存预防](../../.github/workflows/benchmark-tmpl.yml#L145-L146)。 ### 恢复 MI355X TW runner 工作区 @@ -435,17 +435,6 @@ Remaining durable fix: 这些证据就是完成关卡。如果没有制品身份、source/merge 身份和摄取数量,仅仅“工作流绿色”并不代表结果恢复已经验证。 -### AMD 多节点 SGLang 清理 - -退出时(包括启动或就绪检查失败),AMD SGLang 启动器仅向其记录的 `setsid` -进程组发送 TERM,等待最多 30 秒。正常完成时,先暂存结果再进行清理。随后向仍存活的进程组发送 KILL,再等待最多 -5 秒并检查退出状态。这可以清理已成为孤儿进程或忽略 TERM 的工作进程,避免其 -持续占用日志管道。这些清理期限不会改变性能采集、评估或服务器就绪检查的期限。 -客户端失败时保留原退出码;若客户端成功但清理仍未完成,则节点任务失败。 -内核阻塞的进程仍可能需要另行授权的节点修复。不要为绕过清理失败而修改或丢弃 -已完成的指标。单一 EXIT 处理器统一负责进程组清理和现有 UMBP 独立进程 PID -清理;即使进程组清理失败,后者仍会执行。 - ### AMD 多节点 GPU 预检协调 Slurm 启动器先在独立步骤中完成所有选定节点的 Docker 预清理和现有 GPU VRAM diff --git a/inferencex-e2e/docs/testing.md b/inferencex-e2e/docs/testing.md index 8e25915a2a..1724aa7183 100644 --- a/inferencex-e2e/docs/testing.md +++ b/inferencex-e2e/docs/testing.md @@ -146,7 +146,8 @@ Inspect the emitted values, not only the exit code or row count: config key, mod | Changelog content or PR gating | `python -m pytest infx/tests/matrix/test_process_changelog.py infx/tests/workflows/test_validate_perf_changelog.py infx/tests/workflows/test_prepare_perf_changelog_merge.py -v` | | Result processing and topology | `python -m pytest infx/tests/results/power/test_process_result.py infx/tests/results/agentic/test_process_agentic_result.py infx/tests/results/power/test_aggregate_power.py infx/tests/workflows/test_calc_success_rate.py -v` | | AgentX aggregation and artifact loading | `python -m pytest infx/tests/results/agentic/ -v` | -| Eval dispatch, batching, or patches | `python -m pytest infx/tests/evals/ -v` | +| Eval dispatch, batching, staging, or patches | `python -m pytest infx/tests/bench/test_eval_command.py infx/tests/bench/test_eval_meta.py infx/tests/bench/test_vendor_eval.py infx/tests/evals/ -v` | +| Container-side `python3 -m infx.bench` commands | `python -m pytest infx/tests/bench/ -v` | | Eval collection | `python -m pytest infx/tests/results/test_collect_eval_results.py -v` | | Sweep reuse or reusable artifacts | `python -m pytest infx/tests/test_github.py infx/tests/workflows/test_find_reusable_sweep_run.py infx/tests/workflows/test_acknowledge_sweep_reuse.py infx/tests/workflows/test_validate_reusable_sweep_artifacts.py -v` | diff --git a/inferencex-e2e/docs/testing_zh.md b/inferencex-e2e/docs/testing_zh.md index b534525b10..bacd3ca5fa 100644 --- a/inferencex-e2e/docs/testing_zh.md +++ b/inferencex-e2e/docs/testing_zh.md @@ -146,7 +146,8 @@ uv run --locked \ | Changelog 内容或 PR 门禁 | `python -m pytest infx/tests/matrix/test_process_changelog.py infx/tests/workflows/test_validate_perf_changelog.py infx/tests/workflows/test_prepare_perf_changelog_merge.py -v` | | 结果处理与拓扑 | `python -m pytest infx/tests/results/power/test_process_result.py infx/tests/results/agentic/test_process_agentic_result.py infx/tests/results/power/test_aggregate_power.py infx/tests/workflows/test_calc_success_rate.py -v` | | AgentX 聚合与工件加载 | `python -m pytest infx/tests/results/agentic/ -v` | -| 评测分发、批处理或补丁 | `python -m pytest infx/tests/evals/ -v` | +| 评估分发、批处理、暂存或补丁 | `python -m pytest infx/tests/bench/test_eval_command.py infx/tests/bench/test_eval_meta.py infx/tests/bench/test_vendor_eval.py infx/tests/evals/ -v` | +| 容器内的 `python3 -m infx.bench` 命令 | `python -m pytest infx/tests/bench/ -v` | | 评测收集 | `python -m pytest infx/tests/results/test_collect_eval_results.py -v` | | 扫描复用或可复用制品 | `python -m pytest infx/tests/test_github.py infx/tests/workflows/test_find_reusable_sweep_run.py infx/tests/workflows/test_acknowledge_sweep_reuse.py infx/tests/workflows/test_validate_reusable_sweep_artifacts.py -v` | diff --git a/inferencex-e2e/docs/troubleshooting.md b/inferencex-e2e/docs/troubleshooting.md index 5b960b8013..f37a0bbbff 100644 --- a/inferencex-e2e/docs/troubleshooting.md +++ b/inferencex-e2e/docs/troubleshooting.md @@ -24,7 +24,7 @@ Classify a failure by the first layer that did not establish its contract. Prese ## Sources of truth - [`KLAUD_DEBUG.md`](KLAUD_DEBUG.md) records recurring Klaud-Cold/image-bump incidents and their observed signatures. It is incident knowledge, not a substitute for current workflow or review policy. -- [`run-sweep.yml`](../../.github/workflows/run-sweep.yml), [`benchmark-tmpl.yml`](../../.github/workflows/benchmark-tmpl.yml), and [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh) define orchestration, artifact upload, server readiness, benchmark, and eval behavior. +- [`run-sweep.yml`](../../.github/workflows/run-sweep.yml), [`benchmark-tmpl.yml`](../../.github/workflows/benchmark-tmpl.yml), and the container-side `python3 -m infx.bench` commands in [`infx/bench/`](../infx/bench) define orchestration, artifact upload, readiness waits, benchmark, and eval behavior. - [`validate_perf_changelog.py`](../infx/workflows/validate_perf_changelog.py), [`generate.py`](../infx/matrix/generate.py), and [`validation.py`](../infx/matrix/validation.py) own changelog, matrix, and schema failures. - [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md) and [`runners/`](../runners) own provisioning and launcher routing. [`CONTRIBUTING.md`](../../CONTRIBUTING.md#amd-cluster-never-leave-root-owned-files-in-runner-workspaces) owns AMD workspace safety. - [`infx/evals/EVALS.md`](../infx/evals/EVALS.md), [`validate_scores.py`](../infx/evals/validate_scores.py), and [`collect_eval_results.py`](../infx/results/collect_eval_results.py) own eval execution, validation, and collection. @@ -93,9 +93,7 @@ For recovery, follow [`.claude/commands/clean-amd-mi355-runner-root-files.md`](. ## Server -[`wait_for_server_ready`](../benchmarks/benchmark_lib.sh) distinguishes “server died before log,” “server died before healthy,” and a live process whose `/health` endpoint has not passed. Preserve the server log and PID status. The workflow's final timeout alone is not a diagnosis. - -After readiness, the shared helper snapshots the server and recognized persistent engine workers. Benchmark, AgentX and eval clients use `infx.bench_serving.server_watch`: a missing process, zombie or reused PID stops only the owned client process group. Healthy slow work has no new time cutoff. Recipes using a different readiness path need explicit server monitoring; a live wrapper alone is insufficient proof that its workers are alive. +On srt-slurm recipes, srt-slurm starts the servers and waits for them to become healthy, so read its sweep log and worker logs first. The SPEED-Bench collectors start their own server and wait with `python3 -m infx.bench wait --url --pid --log ` ([`wait_ready`](../infx/bench/server.py#L61-L79)). It streams the server log while it polls, fails with `process died before became ready` once the server exits, and otherwise waits until the URL answers below 400. The scheduler owns the time budget. Preserve the server log and PID status. The workflow's final timeout alone is not a diagnosis. Client dependency setup uses uv's bounded HTTP retries and a 120-second read timeout, retaining its download cache. A network/download failure is infrastructure evidence, not a reason to change engine flags. H100 srt-slurm resolves the requested image to its own squash path and checks staged model/image assets; B300 checks node-local staged model configuration on the allocated compute node before launching its container. Missing assets are readiness blockers, never grounds to substitute an old image or different weights. @@ -105,7 +103,7 @@ Use the earliest specific signature: - **Weight/KV/CUDA-graph OOM:** capture free memory, configured utilization, per-rank concurrency, graph limits, and where startup failed. Apply only the setting supported by the matching known case. Confirm startup and workload afterward. - **Kernel/architecture assertion or illegal address:** preserve the complete stack and GPU architecture. Prefer a fixed/pinned upstream image or supported backend over an unreviewed local engine patch. - **Address in use:** identify the owning process and cluster owner before terminating it. Do not kill an unverified PID or unrelated service. -- **Healthy server dies during benchmark:** [`run_benchmark_serving`](../benchmarks/benchmark_lib.sh) monitors the server PID. Preserve both client and server logs and classify the server's first error, not the client's downstream connection failure. +- **Healthy server dies during benchmark:** benchmark and eval clients do not watch the server PID. Preserve both client and server logs and classify the server's first error, not the client's downstream connection failure. Stop if the proposed workaround changes model semantics, reduces model FLOPs, patches the serving stack, or lacks an exact-source guard. The current [PR checklist](PR_REVIEW_CHECKLIST.md) prohibits inference-engine patches unless the documented waiver path is satisfied. diff --git a/inferencex-e2e/docs/troubleshooting_zh.md b/inferencex-e2e/docs/troubleshooting_zh.md index 2788ecee8f..3bb9d9112b 100644 --- a/inferencex-e2e/docs/troubleshooting_zh.md +++ b/inferencex-e2e/docs/troubleshooting_zh.md @@ -24,7 +24,7 @@ ## 事实来源 - [`KLAUD_DEBUG.md`](KLAUD_DEBUG.md) 记录反复出现的 Klaud-Cold/镜像升级事故及其已观测特征。它是事故知识,不替代当前工作流或评审政策。 -- [`run-sweep.yml`](../../.github/workflows/run-sweep.yml)、[`benchmark-tmpl.yml`](../../.github/workflows/benchmark-tmpl.yml) 和 [`benchmarks/benchmark_lib.sh`](../benchmarks/benchmark_lib.sh) 定义编排、制品上传、服务器就绪、基准测试和评测行为。 +- [`run-sweep.yml`](../../.github/workflows/run-sweep.yml)、[`benchmark-tmpl.yml`](../../.github/workflows/benchmark-tmpl.yml) 以及 [`infx/bench/`](../infx/bench) 中在容器内运行的 `python3 -m infx.bench` 命令,共同定义编排、产物上传、就绪等待、基准测试和评估行为。 - [`validate_perf_changelog.py`](../infx/workflows/validate_perf_changelog.py)、[`generate.py`](../infx/matrix/generate.py) 和 [`validation.py`](../infx/matrix/validation.py) 分别负责 changelog、矩阵和模式失败。 - [`utils/runner_setup/RUNNER_SETUP.md`](../utils/runner_setup/RUNNER_SETUP.md) 与 [`runners/`](../runners) 负责预置和启动器路由。[`CONTRIBUTING.md`](../../CONTRIBUTING.md#amd-cluster-never-leave-root-owned-files-in-runner-workspaces) 负责 AMD 工作区安全规则。 - [`infx/evals/EVALS.md`](../infx/evals/EVALS.md)、[`validate_scores.py`](../infx/evals/validate_scores.py) 和 [`collect_eval_results.py`](../infx/results/collect_eval_results.py) 分别负责评测执行、验证和收集。 @@ -93,9 +93,7 @@ Setup 阶段的删除错误通常意味着陈旧分支或改变空白的合并 ## 服务器 -[`wait_for_server_ready`](../benchmarks/benchmark_lib.sh) 会区分“服务器在日志出现前死亡”“服务器在健康前死亡”和进程存活但 `/health` 尚未通过。保留服务器日志和 PID 状态;仅有工作流最终超时不能构成诊断。 - -就绪后,共享 helper 会记录服务器及已识别的持久 engine worker。Benchmark、AgentX 和 eval 客户端通过 `infx.bench_serving.server_watch` 监控:进程消失、成为 zombie 或 PID 被复用时,仅停止所属客户端进程组。健康但缓慢的工作没有新增时限。使用其他就绪路径的 recipe 需要显式监控服务器;wrapper 存活不能单独证明 worker 健康。 +在 srt-slurm 配方中,服务器由 srt-slurm 启动并等待其健康,因此应先阅读其 sweep 日志和 worker 日志。SPEED-Bench 采集脚本自行启动服务器,并使用 `python3 -m infx.bench wait --url --pid --log ` 等待([`wait_ready`](../infx/bench/server.py#L61-L79))。它在轮询时持续输出服务器日志,服务器退出时以 `process died before became ready` 失败,否则一直等到该 URL 返回低于 400 的状态码。时间预算由调度器负责。保留服务器日志和 PID 状态;仅有工作流最终超时不能构成诊断。 客户端依赖安装使用 uv 有限次 HTTP 重试和 120 秒读取超时,并保留下载缓存。网络或下载失败属于基础设施证据,不应据此更改 engine 参数。H100 srt-slurm 将请求镜像解析到其独立 squash 路径并检查已暂存的模型/镜像资源;B300 在分配到的计算节点上检查节点本地模型配置,再启动容器。资源缺失属于就绪性阻塞,不能替换为旧镜像或其他权重。 @@ -105,7 +103,7 @@ Setup 阶段的删除错误通常意味着陈旧分支或改变空白的合并 - **权重/KV/CUDA graph OOM:**记录空闲显存、配置利用率、每 rank 并发、graph 限制和启动失败位置。只应用与匹配案例一致的设置;之后同时确认服务器启动和工作负载完成。 - **内核/架构断言或非法地址:**保留完整堆栈和 GPU 架构。优先使用已修复/固定的上游镜像或受支持后端,而不是未经评审的本地引擎补丁。 - **地址被占用:**终止进程前确认占用者及其集群所有者。不要杀死未经验证的 PID 或无关服务。 -- **健康服务器在基准测试中死亡:**[`run_benchmark_serving`](../benchmarks/benchmark_lib.sh) 会监控服务器 PID。同时保留客户端与服务器日志,并按服务器的第一个错误分类,而不是按客户端后续连接错误分类。 +- **健康服务器在基准测试中死亡:**基准测试和评估客户端不监控服务器 PID。同时保留客户端与服务器日志,并按服务器的第一个错误分类,而不是按客户端后续连接错误分类。 如果拟议 workaround 会改变模型语义、减少模型 FLOPs、修补服务栈,或没有精确源代码 guard,请停止。当前 [PR 清单](PR_REVIEW_CHECKLIST.md) 禁止推理引擎补丁,除非满足规定的豁免流程。 diff --git a/inferencex-e2e/infx/bench/__init__.py b/inferencex-e2e/infx/bench/__init__.py new file mode 100644 index 0000000000..808bbd51b2 --- /dev/null +++ b/inferencex-e2e/infx/bench/__init__.py @@ -0,0 +1 @@ +"""Benchmark, eval, and AgentX clients run inside serving containers (stdlib only, Python 3.10).""" diff --git a/inferencex-e2e/infx/bench/__main__.py b/inferencex-e2e/infx/bench/__main__.py new file mode 100644 index 0000000000..7f2a440314 --- /dev/null +++ b/inferencex-e2e/infx/bench/__main__.py @@ -0,0 +1,35 @@ +"""``python3 -m infx.bench ``: container-side clients, each loaded on first use.""" + +from __future__ import annotations + +import importlib +import os +import sys + +from infx.bench.env import BenchError + +COMMANDS = { + "wait": "infx.bench.server", + "fixed-seq": "infx.bench.fixed_seq", + "agentic": "infx.bench.agentic.run", + "eval": "infx.bench.eval", +} + + +def main(argv: list[str]) -> int: + """Dispatch ``argv[0]``; a ``BenchError`` is one ``ERROR:`` line, not a traceback.""" + if not argv or argv[0] not in COMMANDS: + print(f"usage: python3 -m infx.bench {{{','.join(COMMANDS)}}} [args]", file=sys.stderr) + return 2 + # Only this interpreter needs it; Python-script tools such as amd-smi break under it. + os.environ.pop("PYTHONSAFEPATH", None) + command = importlib.import_module(COMMANDS[argv[0]]) + try: + return command.main(argv[1:]) + except BenchError as error: + print(f"ERROR: {error}", file=sys.stderr) + return 1 + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:])) diff --git a/inferencex-e2e/infx/bench/agentic/__init__.py b/inferencex-e2e/infx/bench/agentic/__init__.py new file mode 100644 index 0000000000..0ed85fab00 --- /dev/null +++ b/inferencex-e2e/infx/bench/agentic/__init__.py @@ -0,0 +1 @@ +"""AgentX trace replay: the isolated AIPerf runtime, the replay argv, and ``agentic``.""" diff --git a/inferencex-e2e/infx/bench/agentic/replay.py b/inferencex-e2e/infx/bench/agentic/replay.py new file mode 100644 index 0000000000..dbc692c915 --- /dev/null +++ b/inferencex-e2e/infx/bench/agentic/replay.py @@ -0,0 +1,193 @@ +"""The AIPerf ``profile`` argv for one AgentX point.""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + +from infx.bench import env as inputs +from infx.bench.agentic import traces + +REQUIRED = ( + "MODEL", + "MODEL_PREFIX", + "FRAMEWORK", + "CONC", + "DURATION", + "AIPERF_LIVE_FAILED_REQUEST_THRESHOLD", + "AIPERF_TRACE_IDLE_GAP_CAP_SECONDS", + "AGENTIC_WARMUP_GRACE_PERIOD", + "AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS", + "AIPERF_EXPERIMENTAL_FAST", + "AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID", + "AIPERF_UNSAFE_OVERRIDE", + "AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING", + "AIPERF_WARMUP_REQUESTS_PER_LANE", +) + +# The scenario rejects shorter profiles unless --unsafe-override marks the run +# submission_valid=false. +MIN_SCENARIO_DURATION_S = 900 +FAST_DURATION_S = 1200 +FAST_WARMUP_REQUESTS_PER_LANE = "1" + + +@dataclass(frozen=True) +class ReplayConfig: + """Everything ``replay_argv`` needs, validated.""" + + url: str + model: str + """Served name sent as ``--model``.""" + tokenizer: str + """Hugging Face id; a wire name is not necessarily a valid repo id.""" + concurrency: int + duration: int + warmup_requests_per_lane: str + live_failed_request_threshold: str + trace_idle_gap_cap_seconds: str + warmup_grace_period: str + extra_inputs: tuple[str, ...] + dynamo_session_timeout: str | None + """Set only when Dynamo conversation-aware routing applies.""" + max_context_length: int | None + server_metrics_urls: tuple[str, ...] + artifact_dir: Path + unsafe_override: bool + loader: str + dataset: str + apply_chat_template: bool + benchmark_grace_period: str | None + + @classmethod + def from_env(cls, env: Mapping[str, str], result_dir: Path) -> ReplayConfig: + """Validate the replay inputs in ``env``; artifacts go below ``result_dir``.""" + values = inputs.require(*REQUIRED, env=env) + duration = inputs.parse_positive_int("DURATION", values["DURATION"]) + warmup = values["AIPERF_WARMUP_REQUESTS_PER_LANE"] + if values["AIPERF_EXPERIMENTAL_FAST"] == "1": + duration, warmup = FAST_DURATION_S, FAST_WARMUP_REQUESTS_PER_LANE + # Dynamo routes later turns to their prefix's prefill worker via nvext.session_control; + # builds after #9920 reject that field, so their recipes route by header or opt out. + conv_aware = ( + values["FRAMEWORK"].startswith("dynamo-") + and values["AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING"] != "0" + and values["AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID"] != "true" + ) + # Replays keep the model's native context: an inherited MAX_MODEL_LEN is a workflow + # value, so only this opt-in sets a smaller service limit. + max_context = inputs.optional("AIPERF_MAX_CONTEXT_LENGTH", env) + loader, dataset = traces.resolve( + values["MODEL_PREFIX"], inputs.optional("WEKA_LOADER_OVERRIDE", env) + ) + extra_inputs = inputs.optional("AIPERF_EXTRA_INPUTS", env) or "" + return cls( + url=_server_url(env), + model=inputs.optional("SERVED_MODEL_NAME", env) or values["MODEL"], + tokenizer=values["MODEL"], + concurrency=inputs.parse_positive_int("CONC", values["CONC"]), + duration=duration, + warmup_requests_per_lane=warmup, + live_failed_request_threshold=values["AIPERF_LIVE_FAILED_REQUEST_THRESHOLD"], + trace_idle_gap_cap_seconds=values["AIPERF_TRACE_IDLE_GAP_CAP_SECONDS"], + warmup_grace_period=values["AGENTIC_WARMUP_GRACE_PERIOD"], + # One ``key:value`` pair per word, as AIPerf's multi-value flag takes them. + extra_inputs=tuple(extra_inputs.split()), + dynamo_session_timeout=( + values["AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS"] if conv_aware else None + ), + max_context_length=( + inputs.parse_positive_int("AIPERF_MAX_CONTEXT_LENGTH", max_context) + if max_context + else None + ), + server_metrics_urls=_server_metrics_urls(env), + artifact_dir=result_dir / "aiperf_artifacts", + unsafe_override=( + duration < MIN_SCENARIO_DURATION_S or values["AIPERF_UNSAFE_OVERRIDE"] == "true" + ), + loader=loader, + dataset=dataset, + # Recipes whose legacy launch rendered prompts client-side keep doing so. + apply_chat_template=env.get("AIPERF_APPLY_CHAT_TEMPLATE") == "true", + # Lets this point's admitted responses finish before the deployment is torn down. + benchmark_grace_period=inputs.optional("AIPERF_BENCHMARK_GRACE_PERIOD", env), + ) + + +def _server_url(env: Mapping[str, str]) -> str: + """srt-slurm's frontend, else ``AIPERF_SERVER_URL``, else ``http://localhost:$PORT``.""" + host = inputs.optional("SRT_FRONTEND_HOST", env) + if host: + return f"http://{host}:{inputs.positive_int('SRT_FRONTEND_PORT', env)}" + return inputs.optional("AIPERF_SERVER_URL", env) or ( + f"http://localhost:{inputs.positive_int('PORT', env)}" + ) + + +def _server_metrics_urls(env: Mapping[str, str]) -> tuple[str, ...]: + """Every Prometheus endpoint AIPerf scrapes; it keeps ``endpoint_url`` per series.""" + urls = inputs.optional("AIPERF_SERVER_METRICS_URLS", env) + if not urls and env.get("SRTCTL_FRONTEND_TYPE") != "dynamo": + # A router frontend does not re-export engine metrics; read each worker's. + prefill = env.get("SRT_PREFILL_ENDPOINTS") + endpoints = env.get("SRT_AGG_ENDPOINTS") or ( + (f"{prefill}," if prefill else "") + env.get("SRT_DECODE_ENDPOINTS", "") + ) + endpoints = endpoints.removesuffix(",") + if endpoints: + urls = ",".join(f"http://{e}/metrics" if e else "" for e in endpoints.split(",")) + if not urls: + return () + parts = urls.split(",") + if parts[-1] == "": + parts.pop() + if not parts or any(not url or any(c.isspace() for c in url) for url in parts): + raise inputs.InputError( + "AIPERF_SERVER_METRICS_URLS must be a comma-separated list of non-empty URLs, " + f"got {urls!r}" + ) + return tuple(parts) + + +def replay_argv(cfg: ReplayConfig, cli: str | Path) -> list[str]: + """The ``aiperf profile`` argv for ``cfg``; ``cli`` is the ``aiperf`` executable.""" + # The agentx scenario owns endpoint type, streaming, seeds, start ratios, server token + # counts, telemetry, dataset size, and server-metric slices. + argv = [str(cli), "profile", "--scenario", "agentx"] + argv += ["--url", cfg.url, "--endpoint", "/v1/chat/completions"] + argv += ["--model", cfg.model, "--tokenizer", cfg.tokenizer] + argv += ["--concurrency", str(cfg.concurrency)] + argv += ["--benchmark-duration", str(cfg.duration)] + # Live abort threshold; AIPERF_FAILED_REQUEST_THRESHOLD stays the post-run gate. + argv += ["--failed-request-threshold", cfg.live_failed_request_threshold] + # One-token advances per lane after the t* snapshot primers. The default spread keeps + # each lane's recorded phase-start offset, so no --burst-phase-starts. + argv += ["--warmup-requests-per-lane", cfg.warmup_requests_per_lane] + # Caps end-to-start idle time per trajectory tree without reordering requests. + argv += ["--trace-idle-gap-cap-seconds", cfg.trace_idle_gap_cap_seconds] + # The maximum wait for warmup to drain, not a fixed sleep. + argv += ["--warmup-grace-period", cfg.warmup_grace_period] + if cfg.extra_inputs: + argv += ["--extra-inputs", *cfg.extra_inputs] + if cfg.dynamo_session_timeout is not None: + # The router's inactivity lease; the upstream 300 s is shorter than an overloaded request. + argv += ["--use-dynamo-conv-aware-routing"] + argv += ["--dynamo-session-timeout-seconds", cfg.dynamo_session_timeout] + # The dataset manager loads the tokenizer anyway, and Kimi ships custom tokenizer code. + argv.append("--tokenizer-trust-remote-code") + if cfg.max_context_length is not None: + # Longer traces would be deterministic 4xxs that still pressure the engine while queued. + argv += ["--max-context-length", str(cfg.max_context_length)] + if cfg.server_metrics_urls: + argv += ["--server-metrics", *cfg.server_metrics_urls] + argv += ["--output-artifact-dir", str(cfg.artifact_dir)] + if cfg.unsafe_override: + argv.append("--unsafe-override") + argv += ["--public-dataset", cfg.loader] + if cfg.apply_chat_template: + argv.append("--apply-chat-template") + if cfg.benchmark_grace_period is not None: + argv += ["--benchmark-grace-period", cfg.benchmark_grace_period] + return argv diff --git a/inferencex-e2e/infx/bench/agentic/run.py b/inferencex-e2e/infx/bench/agentic/run.py new file mode 100644 index 0000000000..12d50c5f68 --- /dev/null +++ b/inferencex-e2e/infx/bench/agentic/run.py @@ -0,0 +1,296 @@ +"""``agentic``: replay AgentX traces at one concurrency against a ready server, then score them.""" + +from __future__ import annotations + +import argparse +import datetime +import math +import os +import shlex +import sys +from collections.abc import Mapping +from contextlib import nullcontext +from dataclasses import dataclass +from pathlib import Path +from typing import Literal + +from infx.bench import env as inputs, proc, server +from infx.bench.agentic.replay import ( + REQUIRED as REPLAY_REQUIRED, + ReplayConfig, + replay_argv, +) +from infx.bench.agentic.venv import Runtime, bootstrap +from infx.bench.gpu_monitor import GpuMonitor + +REQUIRED = ( + "RESULT_DIR", + "RESULT_FILENAME", + "EVAL_ONLY", + "IS_MULTINODE", + "PRECISION", + "ENABLE_AGENTX_POWER", + "REQUIRE_POWER", + "KV_OFFLOADING", + "AIPERF_PYTHON_VERSION", + "AIPERF_FAILED_REQUEST_THRESHOLD", +) +# Spellings the power switches have always accepted as enabled. +TRUE_VALUES = frozenset({"1", "true", "TRUE", "yes", "YES"}) +POWER_SAMPLE_INTERVAL_S = 1 + +# monitor: sample local GPUs; window: mark srt-slurm's multi-node measurement window; +# missing: a multi-node job without that window, recorded as invalid power. +PowerMode = Literal["off", "monitor", "window", "missing"] + + +@dataclass(frozen=True) +class Plan: + """One validated AgentX point.""" + + replay: ReplayConfig + result_dir: Path + output_dir: Path + result_filename: str + python_version: str + chat_budget: tuple[int, int] | None + """``(timeout, stabilization)`` seconds when an eval-only job waits for the chat route.""" + power: PowerMode + require_power: bool + expected_num_gpus: int | None + failed_request_threshold: str + required_metric_prefix: str | None + + @classmethod + def from_env(cls, env: Mapping[str, str]) -> Plan: + """Validate every input before any setup, so a bad point fails in seconds.""" + values = inputs.require(*REQUIRED, *REPLAY_REQUIRED, env=env) + multinode = inputs.flag("IS_MULTINODE", env) + result_dir = Path(values["RESULT_DIR"]).absolute() + result_filename = values["RESULT_FILENAME"] + if multinode or env.get("CONC_LIST"): + # Multi-node collection globs one ${RESULT_FILENAME}_conc*.json per point. Some + # multi-node recipes pin IS_MULTINODE=false, so the workflow's CONC_LIST counts too. + result_dir /= f"conc_{values['CONC']}" + result_filename += f"_conc{values['CONC']}" + replay = ReplayConfig.from_env(env, result_dir) + _require_single_point(env, values["CONC"]) + _validate_kv_offload(env) + eval_only = inputs.flag("EVAL_ONLY", env) + power = _power_mode(values["ENABLE_AGENTX_POWER"], multinode, env) + return cls( + replay=replay, + result_dir=result_dir, + output_dir=Path(env.get("AGENTIC_OUTPUT_DIR") or proc.REPO_ROOT).absolute(), + result_filename=result_filename, + python_version=values["AIPERF_PYTHON_VERSION"], + chat_budget=server.chat_route_budget(env) if eval_only else None, + power=power, + require_power=values["REQUIRE_POWER"] in TRUE_VALUES, + expected_num_gpus=_expected_num_gpus(env) if power == "monitor" else None, + failed_request_threshold=values["AIPERF_FAILED_REQUEST_THRESHOLD"], + required_metric_prefix=inputs.optional("AIPERF_REQUIRED_SERVER_METRIC_PREFIX", env), + ) + + +def _require_single_point(env: Mapping[str, str], conc: str) -> None: + """AgentX measures one concurrency per fresh server deployment.""" + points = env.get("CONC_LIST") + if points is not None and points != conc: + raise inputs.InputError( + "AgentX requires exactly one positive concurrency per server deployment; " + f"CONC_LIST={points!r} must equal CONC={conc!r}. Launch a fresh server for each " + "concurrency." + ) + + +def _validate_kv_offload(env: Mapping[str, str]) -> None: + """The served KV-offload configuration, as the matrix ``kv-offloading`` field allows.""" + mode = env["KV_OFFLOADING"] + backend = env.get("KV_OFFLOAD_BACKEND") + if mode == "none": + if backend: + raise inputs.InputError("KV_OFFLOAD_BACKEND must be empty when KV_OFFLOADING=none") + elif mode == "dram": + if not backend or backend == "none": + raise inputs.InputError("KV_OFFLOAD_BACKEND is required when KV_OFFLOADING=dram") + inputs.parse_positive_int("TOTAL_CPU_DRAM_GB", env.get("TOTAL_CPU_DRAM_GB", "")) + else: + raise inputs.InputError( + f"unsupported KV_OFFLOADING value {mode!r} (expected one of: none, dram)" + ) + + +def _power_mode(enabled: str, multinode: bool, env: Mapping[str, str]) -> PowerMode: + if enabled not in TRUE_VALUES: + return "off" + if not multinode: + return "monitor" + # srt-slurm's telemetry measures multi-node points and exports the window directory. + return "window" if env.get("SRT_MEASUREMENT_WINDOW_DIR") else "missing" + + +def _expected_num_gpus(env: Mapping[str, str]) -> int: + values = inputs.require("TP", "PP_SIZE", "PCP_SIZE", env=env) + return math.prod(inputs.parse_positive_int(name, value) for name, value in values.items()) + + +def execute(plan: Plan, runtime: Runtime, environ: Mapping[str, str]) -> int: + """Run ``plan`` with ``runtime``'s tools; ``environ`` seeds every child's environment.""" + cfg = plan.replay + python = str(runtime.python) + env = { + **environ, + "RESULT_DIR": str(plan.result_dir), + "RESULT_FILENAME": plan.result_filename, + "AGENTIC_OUTPUT_DIR": str(plan.output_dir), + "PYTHONPATH": proc.pythonpath(environ), + } + rc = _download_traces(cfg, runtime, env) + if rc: + return rc + print(f"Using server endpoint: {cfg.url}", flush=True) + if plan.chat_budget is not None: + server.wait_chat_route(cfg.url.rstrip("/"), cfg.model, plan.chat_budget) + + plan.result_dir.mkdir(parents=True, exist_ok=True) + print(f"Running agentic concurrency {cfg.concurrency} on this server deployment", flush=True) + rc = _open_power_window(plan, python, env) + if rc: + return rc + argv = replay_argv(cfg, runtime.aiperf) + (plan.result_dir / "benchmark_command.txt").write_text(f"{shlex.join(argv)}\n") + monitor = ( + GpuMonitor(plan.result_dir / "gpu_metrics.csv", POWER_SAMPLE_INTERVAL_S) + if plan.power == "monitor" + else nullcontext() + ) + log = plan.result_dir / "benchmark.log" + with proc.DeferSignals() as signals, monitor: + replay_rc = 0 if signals.received else proc.tee(argv, log, env) + if signals.received: + return 128 + signals.received + return _score(plan, python, env, replay_rc) + + +def _download_traces(cfg: ReplayConfig, runtime: Runtime, env: Mapping[str, str]) -> int: + print(f"Loading traces via aiperf public-dataset: {cfg.loader} ({cfg.dataset})", flush=True) + # Into the shared HF_HUB_CACHE: later jobs hit it, and the aggregate reads traces there. + rc = proc.call([str(runtime.hf), "download", "--repo-type", "dataset", cfg.dataset], env) + if rc: + print(f"ERROR: downloading {cfg.dataset} failed with code {rc}", file=sys.stderr) + return rc + + +def _open_power_window(plan: Plan, python: str, env: Mapping[str, str]) -> int: + """Record the replay clock's UTC offset; mark srt-slurm's multi-node window running.""" + if plan.power in {"monitor", "window"}: + # AIPerf exports naive local datetimes and SMI the same wall clock; the power + # adapter needs the offset to normalize the profiling window. + now = datetime.datetime.now(datetime.timezone.utc).astimezone() + (plan.result_dir / "agentic_power_timezone_offset.txt").write_text(f"{now:%z}\n") + if plan.power != "window": + return 0 + rc = _power_adapter(plan, python, env, *_window(plan, "running")) + if rc: + print("ERROR: failed to publish the AgentX formal running power window", file=sys.stderr) + return rc + + +def _score(plan: Plan, python: str, env: Mapping[str, str], replay_rc: int) -> int: + """Aggregate, plot, audit power, and validate; return the first failure by precedence.""" + result_dir, artifacts = str(plan.result_dir), str(plan.replay.artifact_dir) + aggregate_rc = _results(python, env, "agentic.process_agentic_result") + # Best effort: the aggregate JSON is the success gate. + _results(python, env, "generate_aiperf_plots", result_dir) + audit = _power_audit(plan, replay_rc) + power_rc = _power_adapter(plan, python, env, *audit) if audit else 0 + _results(python, env, "agentic.analyze_benchmark_distributions", artifacts, "-o", result_dir) + threshold = ["--failed-request-threshold", plan.failed_request_threshold] + validation_rc = _results(python, env, "agentic.validate_agentic_result", artifacts, *threshold) + for failed, message in ( + (replay_rc, f"agentic trace replay exited with code {replay_rc} after writing results"), + (aggregate_rc, f"AgentX aggregation exited with code {aggregate_rc}"), + (validation_rc, "agentic trace replay produced invalid results"), + (power_rc, "AgentX power validation failed after writing audit artifacts"), + ): + if failed: + print(f"ERROR: {message}", file=sys.stderr) + return failed + _check_server_metrics(plan.replay.artifact_dir, plan.required_metric_prefix) + return 0 + + +def _power_audit(plan: Plan, replay_rc: int) -> list[str] | None: + """The power adapter's post-replay arguments; a failed replay leaves the window running.""" + aggregate = ["--agg-result", str(plan.output_dir / f"{plan.result_filename}.json")] + return { + "monitor": [*aggregate, "--expected-num-gpus", str(plan.expected_num_gpus)], + "window": _window(plan, "completed") if replay_rc == 0 else None, + "missing": [*aggregate, "--multinode-contract-missing"], + }.get(plan.power) + + +def _window(plan: Plan, state: str) -> list[str]: + return ["--concurrency", str(plan.replay.concurrency), "--write-multinode-window", state] + + +def _power_adapter(plan: Plan, python: str, env: Mapping[str, str], *args: str) -> int: + strict = ["--require-power"] if plan.require_power else [] + adapter = ["agentic.power_adapter", "--result-dir", str(plan.result_dir), *args, *strict] + return _results(python, env, *adapter) + + +def _results(python: str, env: Mapping[str, str], module: str, *args: str) -> int: + return proc.call([python, "-m", f"infx.results.{module}", *args], env) + + +def _check_server_metrics(artifact_dir: Path, prefix: str | None) -> None: + """Opt-in: fail rather than publish trace charts without the engine's metrics.""" + if not prefix: + return + exported = artifact_dir / "server_metrics_export.json" + if not (_nonempty(exported) and _nonempty(artifact_dir / "server_metrics_export.csv")): + raise inputs.BenchError( + f"required AIPerf server metrics artifacts are missing or empty in {artifact_dir}" + ) + # The export can be several GiB. Metric names are object keys, so a quoted prefix + # anywhere proves the engine's metrics were captured. + if not _contains(exported, f'"{prefix}'.encode()): + raise inputs.BenchError(f"{exported} has no metric with required prefix {prefix!r}") + print(f"Validated required AIPerf server metrics prefix {prefix!r}", flush=True) + + +def _contains(path: Path, needle: bytes) -> bool: + """Scan ``path`` in 1 MiB chunks, keeping enough overlap for a match across chunks.""" + overlap = b"" + with path.open("rb") as stream: + while chunk := stream.read(1 << 20): + window = overlap + chunk + if needle in window: + return True + overlap = window[max(0, len(window) - len(needle) + 1) :] + return False + + +def _nonempty(path: Path) -> bool: + return path.is_file() and path.stat().st_size > 0 + + +def main(argv: list[str]) -> int: + """Run the ``agentic`` command.""" + argparse.ArgumentParser( + prog="python3 -m infx.bench agentic", + description="Replay AgentX traces at one concurrency against a ready server.", + ).parse_args(argv) + plan = Plan.from_env(os.environ) + runtime = Runtime.for_job(os.environ) + if not runtime.active(): + print(f"Preparing the AIPerf runtime in {runtime.venv}", flush=True) + rc = bootstrap(runtime, plan.python_version, proc.REPO_ROOT / "utils" / "aiperf") + if rc: + return rc + # The venv interpreter needs the same safe sys.path as this one had. + environ = {**os.environ, "PYTHONPATH": proc.pythonpath(), "PYTHONSAFEPATH": "1"} + runtime.exec_python(["-m", "infx.bench", "agentic", *argv], environ) + return execute(plan, runtime, os.environ) diff --git a/inferencex-e2e/infx/bench/agentic/traces.py b/inferencex-e2e/infx/bench/agentic/traces.py new file mode 100644 index 0000000000..0232c946d2 --- /dev/null +++ b/inferencex-e2e/infx/bench/agentic/traces.py @@ -0,0 +1,28 @@ +"""Which AIPerf public-dataset loader an AgentX point replays, and its Hugging Face corpus.""" + +from __future__ import annotations + +from infx.bench.env import InputError + +# Each loader's corpus is the ``hf_dataset_name`` AIPerf registers for it in +# utils/aiperf/src/aiperf/plugin/plugins.yaml; the run pre-downloads exactly that dataset. +LOADERS = { + "semianalysis_cc_traces_weka_062126": "semianalysisai/cc-traces-weka-062126", + "semianalysis_cc_traces_weka_062126_256k": "semianalysisai/cc-traces-weka-062126-256k", + # Rolling aliases; AIPerf currently points them at the 062126 corpora. + "semianalysis_cc_traces_weka_with_subagents": "semianalysisai/cc-traces-weka-062126", + "semianalysis_cc_traces_weka_with_subagents_256k": "semianalysisai/cc-traces-weka-062126-256k", +} +UNCAPPED = "semianalysis_cc_traces_weka_062126" +CAPPED = "semianalysis_cc_traces_weka_062126_256k" +# Families that serve 1M context replay the unfiltered corpus; others take the 256k cap. +# Matching is by prefix, so ``dsv4`` also selects ``dsv41flash``. +UNCAPPED_FAMILIES = ("dsv4", "glm5.2", "glm5.3", "minimaxm3", "kimik3") + + +def resolve(model_prefix: str, override: str | None) -> tuple[str, str]: + """``(loader, hf_dataset)``; ``override`` (``WEKA_LOADER_OVERRIDE``) pins another corpus.""" + loader = override or (UNCAPPED if model_prefix.startswith(UNCAPPED_FAMILIES) else CAPPED) + if loader not in LOADERS: + raise InputError(f"unknown WEKA_LOADER_OVERRIDE={loader!r}; allowed: {', '.join(LOADERS)}") + return loader, LOADERS[loader] diff --git a/inferencex-e2e/infx/bench/agentic/venv.py b/inferencex-e2e/infx/bench/agentic/venv.py new file mode 100644 index 0000000000..cc7126260c --- /dev/null +++ b/inferencex-e2e/infx/bench/agentic/venv.py @@ -0,0 +1,124 @@ +"""The isolated AIPerf runtime: a per-job uv venv with ``utils/aiperf`` installed editable.""" + +from __future__ import annotations + +import os +import shutil +import sys +import tempfile +from collections.abc import Mapping, Sequence +from dataclasses import dataclass +from pathlib import Path +from typing import NoReturn + +from infx.bench import proc + +UV_INSTALLER = "https://astral.sh/uv/install.sh" +# AIPerf plus what the post-run AgentX processing and plots import. +DEPENDENCIES = ( + "numpy>=1.24", + "pandas>=2.0.0", + "aiohttp>=3.10", + "transformers>=4.46", + "xlsxwriter>=3.2.1", + "tqdm>=4.66", + "datasets>=4.7.0", + "tiktoken", + "matplotlib", + "huggingface_hub[cli]>=0.25.0", + "urllib3", + "requests", +) + + +@dataclass(frozen=True) +class Runtime: + """Paths of one job's AIPerf runtime directory.""" + + root: Path + + @classmethod + def for_job(cls, environ: Mapping[str, str]) -> Runtime: + """``AIPERF_RUNTIME_DIR``, else ``inferencex-agentic-`` in the temp directory.""" + explicit = environ.get("AIPERF_RUNTIME_DIR") + if explicit: + return cls(Path(explicit).absolute()) + # The process id survives the re-exec into the venv, so both phases agree. + job = environ.get("SLURM_JOB_ID") or str(os.getpid()) + return cls(Path(tempfile.gettempdir()) / f"inferencex-agentic-{job}") + + @property + def venv(self) -> Path: + return self.root / "venv" + + @property + def python(self) -> Path: + return self.venv / "bin" / "python" + + @property + def aiperf(self) -> Path: + return self.venv / "bin" / "aiperf" + + @property + def hf(self) -> Path: + return self.venv / "bin" / "hf" + + def active(self) -> bool: + """Whether this process already runs the venv's python.""" + return Path(sys.executable) == self.python + + def exec_python(self, args: Sequence[str], environ: Mapping[str, str]) -> NoReturn: + """Replace this process with the venv's python running ``args``.""" + sys.stdout.flush() + sys.stderr.flush() + os.execve(self.python, [str(self.python), *args], environ) # noqa: S606 + + +def bootstrap(runtime: Runtime, python_version: str, aiperf_source: Path) -> int: + """Create ``runtime``'s venv from scratch; return a shell-style status.""" + uv = _uv(runtime.root / "uv" / "bin") + if uv is None: + return 1 + shutil.rmtree(runtime.venv, ignore_errors=True) + cache = runtime.root / "uv-cache" + cache.mkdir(parents=True, exist_ok=True) + env = {**os.environ, "UV_CACHE_DIR": str(cache)} + # AIPerf dropped Python 3.10, which some ROCm images still ship; uv fetches a pinned build. + rc = proc.call([uv, "venv", "--python", python_version, str(runtime.venv)], env) + if rc: + return rc + install = [uv, "pip", "install", "--python", str(runtime.python), "-e", str(aiperf_source)] + retries = {"UV_HTTP_TIMEOUT": "120", "UV_HTTP_RETRIES": "3"} + if proc.call([*install, *DEPENDENCIES], {**env, **retries}): + print( + "ERROR: benchmark client dependency bootstrap failed; inspect network/package " + "resolution before recipe repairs", + file=sys.stderr, + ) + return 1 + if not (_executable(runtime.aiperf) and _executable(runtime.hf)): + print( + f"ERROR: isolated AIPerf environment is incomplete at {runtime.venv}", file=sys.stderr + ) + return 1 + return 0 + + +def _uv(install_dir: Path) -> str | None: + """``uv`` from ``PATH``, else Astral's installer (rootless enroot cannot mutate dpkg).""" + found = shutil.which("uv") + if found: + return found + uv = install_dir / "uv" + if not _executable(uv): + install_dir.mkdir(parents=True, exist_ok=True) + env = {**os.environ, "UV_INSTALL_DIR": str(install_dir), "UV_NO_MODIFY_PATH": "1"} + proc.call(["sh", "-c", f"curl -LsSf {UV_INSTALLER} | sh"], env) + if not _executable(uv): + print(f"ERROR: uv installation did not create {uv}", file=sys.stderr) + return None + return str(uv) + + +def _executable(path: Path) -> bool: + return path.is_file() and os.access(path, os.X_OK) diff --git a/inferencex-e2e/infx/bench/env.py b/inferencex-e2e/infx/bench/env.py new file mode 100644 index 0000000000..2a1b92895f --- /dev/null +++ b/inferencex-e2e/infx/bench/env.py @@ -0,0 +1,57 @@ +"""Caller-supplied inputs: a missing or malformed one fails, naming every offender.""" + +from __future__ import annotations + +import os +import re +from collections.abc import Mapping + + +class BenchError(Exception): + """Reported by ``python3 -m infx.bench`` as one ``ERROR:`` line, exit code 1.""" + + +class InputError(BenchError): + """A required caller input is missing or malformed.""" + + +def require(*names: str, env: Mapping[str, str] = os.environ) -> dict[str, str]: + """Return the named values; unset and empty both count as missing.""" + missing = [name for name in names if not env.get(name)] + if missing: + listed = "\n".join(f" - {name}" for name in missing) + raise InputError(f"The following required environment variables are not set:\n{listed}") + return {name: env[name] for name in names} + + +def optional(name: str, env: Mapping[str, str] = os.environ) -> str | None: + """Return a deliberately optional input, or ``None`` when unset or empty.""" + return env.get(name) or None + + +def flag(name: str, env: Mapping[str, str] = os.environ) -> bool: + """Parse a required ``true``/``false`` input.""" + value = require(name, env=env)[name] + if value not in {"true", "false"}: + raise InputError(f"{name} must be true or false, got {value!r}") + return value == "true" + + +def positive_int(name: str, env: Mapping[str, str] = os.environ) -> int: + """Parse a required positive decimal integer input.""" + return parse_positive_int(name, require(name, env=env)[name]) + + +def parse_positive_int(name: str, value: str) -> int: + """Parse ``value`` as a positive decimal integer named ``name`` in errors.""" + if not re.fullmatch(r"[1-9][0-9]*", value): + raise InputError(f"{name} must be a positive integer, got {value!r}") + return int(value) + + +def non_negative_int(name: str, env: Mapping[str, str] = os.environ) -> int: + """Parse a required non-negative decimal integer input.""" + value = require(name, env=env)[name] + if not re.fullmatch(r"[0-9]+", value): + raise InputError(f"{name} must be a non-negative integer, got {value!r}") + return int(value) diff --git a/inferencex-e2e/infx/bench/eval/__init__.py b/inferencex-e2e/infx/bench/eval/__init__.py new file mode 100644 index 0000000000..38c1f4ef5f --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/__init__.py @@ -0,0 +1,211 @@ +"""Evaluate a ready server with one eval framework, then stage and record the eval.""" + +from __future__ import annotations + +import argparse +import functools +import os +import re +import sys +import tempfile +import urllib.parse +from collections.abc import Callable, Mapping, Sequence +from pathlib import Path + +from infx.bench import env, server +from infx.bench.eval import lm_eval, meta, stage, vendor +from infx.bench.eval.context import EvalContext, EvalOutcome + +Framework = Callable[[EvalContext], EvalOutcome] +FRAMEWORKS: dict[str, Framework] = { + "lm-eval": lm_eval.run, + **{name: functools.partial(vendor.run, p) for name, p in vendor.PROVIDERS.items()}, +} +_SUITE = re.compile(r"[A-Za-z0-9_.-]+") + + +def main(argv: list[str]) -> int: + """Run the ``eval`` command.""" + parser = argparse.ArgumentParser( + prog="python3 -m infx.bench eval", + description=__doc__, + epilog="Exits with the framework's code, else 1 when staging or meta_env.json failed.", + ) + parser.add_argument("--endpoint", required=True, help="server root, e.g. http://localhost:8000") + parser.add_argument( + "--concurrency", + required=True, + help='concurrent requests; a list ("1 4 8") runs lm-eval once per value, suffixes the ' + "artifacts _conc, and exits 0, leaving failed values to score validation", + ) + parser.add_argument( + "--stage-to", + required=True, + type=Path, + help="directory receiving the allow-listed artifacts and meta_env.json, even on failure", + ) + parser.add_argument( + "--framework", + help=f"one of {', '.join(FRAMEWORKS)}; EVAL_FRAMEWORK overrides it (default: lm-eval)", + ) + args = parser.parse_args(argv) + return evaluate(args.endpoint, args.concurrency, args.stage_to, args.framework) + + +def _server_root(endpoint: str) -> str: + root = endpoint.rstrip("/") + try: + parts = urllib.parse.urlsplit(root) + valid = parts.scheme in {"http", "https"} and bool(parts.hostname) and parts.port != 0 + except ValueError: + valid = False + if not valid: + raise env.InputError(f"--endpoint must be an http(s) server root, got {endpoint!r}") + return root + + +def _concurrencies(values: str) -> list[int]: + concurrencies = [env.parse_positive_int("--concurrency", value) for value in values.split()] + if not concurrencies: + raise env.InputError("--concurrency must name at least one concurrency") + return concurrencies + + +def evaluate( + endpoint: str, + concurrency: str, + destination: Path, + framework: str | None = None, + environ: Mapping[str, str] = os.environ, +) -> int: + """Run, stage, and record one eval; return its exit code.""" + environ = dict(environ) + name = environ.get("EVAL_FRAMEWORK") or framework or "lm-eval" + if name not in FRAMEWORKS: + raise env.InputError( + f"unknown eval framework {name!r}; expected one of {', '.join(sorted(FRAMEWORKS))}" + ) + suite = env.optional("EVAL_SUITE", env=environ) + if suite is not None and not _SUITE.fullmatch(suite): + raise env.InputError("EVAL_SUITE may contain only letters, digits, '.', '_', and '-'") + if suite is not None and name not in vendor.PROVIDERS: + raise env.InputError( + f"EVAL_SUITE is only supported with {', '.join(sorted(vendor.PROVIDERS))}" + ) + concurrencies = _concurrencies(concurrency) + if len(concurrencies) > 1 and name != "lm-eval": + raise env.InputError("batched eval concurrency is only supported for lm-eval") + eval_only = env.flag("EVAL_ONLY", env=environ) + checkpoint = env.require("MODEL", env=environ)["MODEL"] + model = environ.get("MODEL_NAME") or checkpoint + base_url = _server_root(endpoint) + # Reject malformed metadata inputs before the eval spends hours. + meta.build(environ, conc=concurrencies[0], suite="") + if eval_only and name in vendor.PROVIDERS: + server.wait_chat_route(base_url, model, server.chat_route_budget(environ)) + context = functools.partial( + EvalContext, + base_url=base_url, + model=model, + context_length=lm_eval.prepare(environ) if name == "lm-eval" else 0, + suite=suite, + env=environ, + ) + if len(concurrencies) > 1: + return _batched(FRAMEWORKS[name], context, concurrencies, destination, environ) + return _single(FRAMEWORKS[name], context, concurrencies[0], destination, environ) + + +def _run( + framework: Framework, + context: Callable[..., EvalContext], + conc: int, + destination: Path, + suffix: str, +) -> tuple[EvalOutcome, list[Path] | None]: + """Run ``framework`` in a fresh results directory; stage what it wrote (``None``: failed).""" + with tempfile.TemporaryDirectory( + prefix=f"eval_out-conc{conc}-", ignore_cleanup_errors=True + ) as results: + outcome = framework(context(concurrency=conc, results_dir=Path(results))) + try: + staged = stage.copy(Path(results), destination, suffix=suffix) + except OSError as error: + print( + f"ERROR: failed to stage eval artifacts in {destination}: {error}", file=sys.stderr + ) + staged = None + return outcome, staged + + +def _record( + destination: Path, + environ: Mapping[str, str], + conc: int, + suite: str, + batch: Mapping[str, Sequence[int]] | None = None, +) -> bool: + path = destination / "meta_env.json" + try: + destination.mkdir(parents=True, exist_ok=True) + meta.write(path, meta.build(environ, conc=conc, suite=suite, batch=batch)) + except OSError as error: + print(f"ERROR: failed to write {path}: {error}", file=sys.stderr) + return False + return True + + +def _single( + framework: Framework, + context: Callable[..., EvalContext], + conc: int, + destination: Path, + environ: Mapping[str, str], +) -> int: + outcome, staged = _run(framework, context, conc, destination, "") + recorded = _record(destination, environ, conc, outcome.suite) + if outcome.returncode: + print(f"ERROR: eval failed with exit code {outcome.returncode}", file=sys.stderr) + return outcome.returncode + if staged is None or not recorded: + return 1 + print(f"Staged eval artifacts in: {destination}") + return 0 + + +def _batched( + framework: Framework, + context: Callable[..., EvalContext], + concurrencies: list[int], + destination: Path, + environ: Mapping[str, str], +) -> int: + completed: list[int] = [] + failed: list[int] = [] + suite = "" + for conc in concurrencies: + print(f"Running lm-eval at concurrency {conc}") + outcome, staged = _run(framework, context, conc, destination, f"_conc{conc}") + suite = outcome.suite + if staged == []: + print(f"WARN: no eval artifacts were produced for concurrency {conc}", file=sys.stderr) + if outcome.returncode == 0 and staged: + completed.append(conc) + else: + print(f"ERROR: lm-eval failed at concurrency {conc}", file=sys.stderr) + failed.append(conc) + if failed: + print( + f"ERROR: batched eval failed for concurrency: {' '.join(map(str, failed))}; " + "score validation fails the job after upload", + file=sys.stderr, + ) + batch = { + "eval_concs": concurrencies, + "completed_eval_concs": completed, + "failed_eval_concs": failed, + } + if not _record(destination, environ, concurrencies[0], suite, batch): + return 1 + print(f"Prepared batched eval artifacts in: {destination}") + return 0 diff --git a/inferencex-e2e/infx/bench/eval/context.py b/inferencex-e2e/infx/bench/eval/context.py new file mode 100644 index 0000000000..893fd9403b --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/context.py @@ -0,0 +1,35 @@ +"""The contract between the eval dispatcher and each eval framework.""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from pathlib import Path + + +@dataclass(frozen=True) +class EvalContext: + """One eval invocation against a ready OpenAI-compatible server.""" + + base_url: str + """Server root without a trailing slash.""" + model: str + """Served model name sent in requests.""" + concurrency: int + context_length: int + """Prompt-plus-generation token budget; 0 for vendor suites, which fix their own.""" + results_dir: Path + """Empty directory for the framework's raw artifacts.""" + suite: str | None + """``EVAL_SUITE``, if set.""" + env: Mapping[str, str] + + +@dataclass(frozen=True) +class EvalOutcome: + """What a framework reports back to the dispatcher.""" + + returncode: int + """Shell-style exit status: 128 + N after signal N.""" + suite: str + """Suite actually run; recorded as ``eval_suite`` in ``meta_env.json``.""" diff --git a/inferencex-e2e/infx/bench/eval/lm_eval.py b/inferencex-e2e/infx/bench/eval/lm_eval.py new file mode 100644 index 0000000000..4b76c8b153 --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/lm_eval.py @@ -0,0 +1,146 @@ +"""The lm-evaluation-harness framework: pinned install, per-request token budget, and one run.""" + +from __future__ import annotations + +import json +import os +import shutil +import sys +import tempfile +from collections.abc import Mapping +from pathlib import Path + +from infx.bench import env, proc +from infx.bench.eval.context import EvalContext, EvalOutcome + +REPOSITORY = "https://github.com/EleutherAI/lm-evaluation-harness" +REF = "b315ef3b05176acc9732bb7fdec116abe1ecc476" # installed over the lm-eval[api] release +DEFAULT_TASKS = "infx/evals/gsm8k.yaml" +PATCH = "infx/evals/patches/lm_eval_sitecustomize.py" +FALLBACK_CONTEXT = 16384 +PROMPT_RESERVE = 4096 +MAX_OUTPUT_TOKENS = 16384 +CONTEXT_FIELDS = ("max_position_embeddings", "max_sequence_length", "seq_length", "n_positions") + + +def install(environ: Mapping[str, str]) -> None: + """Install lm-eval at ``REF`` into this interpreter; failures only warn, as images may ship it.""" + pip = [sys.executable, "-m", "pip"] + flags = ["-q", "--no-cache-dir", "--break-system-packages"] + # torchvision causes circular imports in ATOM; TRT-LLM/SGLang need it at module level. + if "atom" in environ.get("IMAGE", ""): + _pip([*pip, "uninstall", "-y", "torchvision"], environ) + _pip([*pip, "install", *flags, "lm-eval[api]"], environ) + pinned = [*pip, "install", *flags, "--no-deps", "--force-reinstall"] + git = shutil.which("git", path=environ.get("PATH")) + if git and _pip([*pinned, f"git+{REPOSITORY}.git@{REF}"], environ): + return + _pip([*pinned, f"{REPOSITORY}/archive/{REF}.tar.gz"], environ) + + +def _pip(argv: list[str], environ: Mapping[str, str]) -> bool: + rc = proc.call(argv, environ) + if rc: + print(f"WARN: pip {' '.join(argv[3:])} failed with exit code {rc}", file=sys.stderr) + return rc == 0 + + +def native_context_length(model: str) -> int: + """The model config's maximum sequence length, or 0 when it cannot be read.""" + # A local config.json works even when transformers does not know the model type yet. + try: + config = json.loads((Path(model) / "config.json").read_text()) + except (OSError, ValueError): + config = None + if isinstance(config, dict): + for name in CONTEXT_FIELDS: + value = config.get(name) + if type(value) is int and value > 0: + return value + try: + from transformers import AutoConfig # the serving image's copy, loaded only here + + config = AutoConfig.from_pretrained(model, trust_remote_code=True) + except Exception: # noqa: BLE001 - any failure leaves the native maximum unknown + return 0 + for name in CONTEXT_FIELDS: + if hasattr(config, name): + value = getattr(config, name) + return value if isinstance(value, int) and value > 0 else 0 + return 0 + + +def context_length(environ: Mapping[str, str]) -> int: + """``EVAL_MAX_MODEL_LEN``, else ``MAX_MODEL_LEN`` capped at the model's native maximum.""" + explicit = env.optional("EVAL_MAX_MODEL_LEN", env=environ) + if explicit is not None: + return env.parse_positive_int("EVAL_MAX_MODEL_LEN", explicit) + benchmark = env.optional("MAX_MODEL_LEN", env=environ) or "0" # 0: the native maximum + benchmark_length = 0 if benchmark == "0" else env.parse_positive_int("MAX_MODEL_LEN", benchmark) + # MODEL can be a served alias that is neither a repo id nor a path (deepseek-r1-fp4 on B300). + local = env.optional("MODEL_PATH", env=environ) + model = local if local and Path(local).is_dir() else env.require("MODEL", env=environ)["MODEL"] + native = native_context_length(model) + length = min(benchmark_length or native, native) if native else benchmark_length + if length: + return length + print(f"WARN: no context length known for {model}; using {FALLBACK_CONTEXT}", file=sys.stderr) + return FALLBACK_CONTEXT + + +def max_output_tokens(context: int) -> int: + """Leave room for the prompt, and bound the KV cache TRT-LLM reserves per request.""" + budget = context - PROMPT_RESERVE if context > PROMPT_RESERVE else context // 2 + return min(budget, MAX_OUTPUT_TOKENS) + + +def tasks(environ: Mapping[str, str]) -> str: + return env.optional("EVAL_TASKS_DIR", env=environ) or DEFAULT_TASKS + + +def suite(environ: Mapping[str, str]) -> str: + """The task YAML's stem, or the task name.""" + name = Path(tasks(environ)).name + for extension in (".yaml", ".yml"): + name = name.removesuffix(extension) + return name + + +def prepare(environ: Mapping[str, str]) -> int: + """Validate the inputs, install lm-eval, and return every request's context length.""" + env.require("OPENAI_API_KEY", env=environ) + length = context_length(environ) + install(environ) + return length + + +def run(ctx: EvalContext) -> EvalOutcome: + """Evaluate ``ctx.model`` through the server's chat completions route.""" + max_tokens = max_output_tokens(ctx.context_length) + print(f"Eval budget: eval_context_len={ctx.context_length}, max_output_tokens={max_tokens}") + model_args = [ + f"model={ctx.model}", + f"base_url={ctx.base_url}/v1/chat/completions", + f"api_key={ctx.env['OPENAI_API_KEY']}", + "eos_string=", + "max_retries=5", + f"num_concurrent={ctx.concurrency}", + "timeout=1800", + "tokenized_requests=False", + f"max_length={ctx.context_length}", + ] + argv = [ + sys.executable, "-m", "lm_eval", "--model", "local-chat-completions", + "--apply_chat_template", "--tasks", tasks(ctx.env), + "--output_path", str(ctx.results_dir), "--log_samples", + "--model_args", ",".join(model_args), + "--gen_kwargs", f"max_tokens={max_tokens},temperature=0,top_p=1", + ] # fmt: skip + if limit := env.optional("EVAL_LIMIT", env=ctx.env): + argv += ["--limit", limit] + with tempfile.TemporaryDirectory(prefix="lm-eval-patch-") as patch: + shutil.copyfile(proc.REPO_ROOT / PATCH, Path(patch, "sitecustomize.py")) + pythonpath = os.pathsep.join([patch, proc.pythonpath(ctx.env)]) + # Relative task paths resolve against the checkout, whatever the image WORKDIR. + rc = proc.call(argv, {**ctx.env, "PYTHONPATH": pythonpath}, cwd=proc.REPO_ROOT) + return EvalOutcome(rc, suite(ctx.env)) diff --git a/inferencex-e2e/infx/bench/eval/meta.py b/inferencex-e2e/infx/bench/eval/meta.py new file mode 100644 index 0000000000..644d138d23 --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/meta.py @@ -0,0 +1,127 @@ +"""``meta_env.json``: the identity, topology, and concurrency of one eval. + +Score validation, result collection, and ingestion read its keys with these value types. +""" + +from __future__ import annotations + +import json +import re +from collections.abc import Mapping +from pathlib import Path + +from infx.bench import env + +BATCH_KEYS = ("eval_concs", "completed_eval_concs", "failed_eval_concs") +"""The manifest of a batched eval, next to ``conc`` (its first concurrency).""" +_TRUE = {"1", "true", "yes", "on"} +_RESULT_FILENAME = re.compile(r".*_([^_]+)_([^_]+)_tp") + + +def _first(values: Mapping[str, str], *names: str, default: str) -> str: + return next((values[name] for name in names if values.get(name)), default) + + +def _is_true(value: str) -> bool: + return value.lower() in _TRUE + + +def _disaggregated(values: Mapping[str, str]) -> dict[str, str]: + """Metadata names of a disaggregated job from the workflow's ``PREFILL_*``/``DECODE_*``.""" + tp = _first(values, "PREFILL_TP", "TP", default="1") + prefill_ep = _first(values, "PREFILL_EP", "EP_SIZE", "EP", default="1") + prefill_dp = _first(values, "PREFILL_DP_ATTN", default="false") + return { + "TP": tp, + "PREFILL_TP": tp, + "PREFILL_EP": prefill_ep, + "EP_SIZE": prefill_ep, + "PREFILL_NUM_WORKERS": _first(values, "PREFILL_NUM_WORKERS", default="1"), + "DECODE_TP": _first(values, "DECODE_TP", default=tp), + "DECODE_EP": _first(values, "DECODE_EP", default=prefill_ep), + "DECODE_NUM_WORKERS": _first(values, "DECODE_NUM_WORKERS", default="1"), + "DP_ATTENTION": prefill_dp, + "PREFILL_DP_ATTENTION": prefill_dp, + "DECODE_DP_ATTENTION": _first(values, "DECODE_DP_ATTN", default="false"), + } + + +def build( + environ: Mapping[str, str], + *, + conc: object, + suite: str, + batch: Mapping[str, object] | None = None, +) -> dict[str, object]: + """The document of an eval of ``suite`` at ``conc``; ``batch`` supplies its ``BATCH_KEYS``.""" + multinode = env.flag("IS_MULTINODE", env=environ) + values = dict(environ) + if multinode: + # Only here: the bridge resets unset per-phase DP flags, which would overwrite a + # single-node job's real DP_ATTENTION. + values.update(_disaggregated(values)) + + def size(field: str, *names: str) -> int: + name = next((name for name in names if values.get(name)), None) + if name is None: + return 1 + if not re.fullmatch(r"[0-9]+", values[name]): + raise env.InputError( + f"{name} must be an integer for meta_env.json {field}, got {values[name]!r}" + ) + return int(values[name]) + + def dp(*names: str) -> bool: + return _is_true(_first(values, *names, default="false")) + + framework, precision = values.get("FRAMEWORK", ""), values.get("PRECISION", "") + parsed = _RESULT_FILENAME.match(values.get("RESULT_FILENAME", "")) + if parsed is not None: + precision = precision or parsed.group(1) + framework = framework or parsed.group(2) + return { + "is_multinode": multinode, + "framework": framework or "unknown", + "precision": precision or "unknown", + "spec_decoding": values.get("SPEC_DECODING", ""), + "eval_suite": suite, + "recipe_fingerprint": values.get("RECIPE_FINGERPRINT", ""), + "tp": size("tp", "TP"), + "pp": size("pp", "PP_SIZE"), + "dcp_size": size("dcp_size", "DCP_SIZE"), + "pcp_size": size("pcp_size", "PCP_SIZE"), + "conc": conc, + **({key: batch[key] for key in BATCH_KEYS} if batch is not None else {}), + "ep": size("ep", "EP_SIZE"), + "dp_attention": dp("DP_ATTENTION"), + "prefill_tp": size("prefill_tp", "PREFILL_TP", "TP"), + "prefill_pp": size("prefill_pp", "PREFILL_PP_SIZE", "PP_SIZE"), + "prefill_dcp_size": size("prefill_dcp_size", "PREFILL_DCP_SIZE", "DCP_SIZE"), + "prefill_pcp_size": size("prefill_pcp_size", "PREFILL_PCP_SIZE", "PCP_SIZE"), + "prefill_ep": size("prefill_ep", "PREFILL_EP", "EP_SIZE"), + "prefill_dp_attention": dp("PREFILL_DP_ATTENTION", "DP_ATTENTION"), + "prefill_num_workers": size("prefill_num_workers", "PREFILL_NUM_WORKERS"), + "decode_tp": size("decode_tp", "DECODE_TP", "TP"), + "decode_pp": size("decode_pp", "DECODE_PP_SIZE", "PP_SIZE"), + "decode_dcp_size": size("decode_dcp_size", "DECODE_DCP_SIZE", "DCP_SIZE"), + "decode_pcp_size": size("decode_pcp_size", "DECODE_PCP_SIZE", "PCP_SIZE"), + "decode_ep": size("decode_ep", "DECODE_EP", "EP_SIZE"), + "decode_dp_attention": dp("DECODE_DP_ATTENTION", "DP_ATTENTION"), + "decode_num_workers": size("decode_num_workers", "DECODE_NUM_WORKERS"), + "model": values.get("MODEL_NAME") or values.get("MODEL", ""), + "infmax_model_prefix": values.get("MODEL_PREFIX") or "unknown", + "hw": values.get("RUNNER_TYPE") or "unknown", + "isl": values.get("ISL") or "0", + "osl": values.get("OSL") or "0", + } + + +def write(path: Path, document: Mapping[str, object]) -> None: + path.write_text(json.dumps(document, indent=2) + "\n") + + +def refresh(path: Path, environ: Mapping[str, str]) -> None: + """Rebuild a staged ``meta_env.json`` from ``environ``, keeping its suite, conc, and batch.""" + staged = json.loads(path.read_text()) + batch = staged if "eval_concs" in staged else None + write(path, build(environ, conc=staged["conc"], suite=staged["eval_suite"], batch=batch)) diff --git a/inferencex-e2e/infx/bench/eval/stage.py b/inferencex-e2e/infx/bench/eval/stage.py new file mode 100644 index 0000000000..9e3e140afb --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/stage.py @@ -0,0 +1,43 @@ +"""Copying an eval's allow-listed artifacts out of its results directory.""" + +from __future__ import annotations + +import fnmatch +import shutil +from pathlib import Path + +# The workflows upload exactly these names from the workspace root; collectors read them there. +ARTIFACTS = ( + "results*.json", + "*_report.json", + "*_results.jsonl", + "*_artifacts.tar.gz", + "sample*.jsonl", +) +_EXTENSIONS = (".tar.gz", ".jsonl", ".json") + + +def copy(results_dir: Path, destination: Path, *, suffix: str = "") -> list[Path]: + """Copy the artifacts anywhere under ``results_dir`` into ``destination``; return them.""" + destination.mkdir(parents=True, exist_ok=True) + staged = [] + for source in sorted(results_dir.rglob("*")): + if source.is_file() and any(fnmatch.fnmatchcase(source.name, glob) for glob in ARTIFACTS): + target = _target(destination, source.name, suffix) + shutil.copyfile(source, target) + staged.append(target) + return staged + + +def _target(destination: Path, name: str, suffix: str) -> Path: + # Batched runs share one destination, so a suffixed name never replaces another. + if not suffix: + return destination / name + extension = next(extension for extension in _EXTENSIONS if name.endswith(extension)) + stem = name[: -len(extension)] + target = destination / f"{stem}{suffix}{extension}" + count = 2 + while target.exists(): + target = destination / f"{stem}{suffix}_{count}{extension}" + count += 1 + return target diff --git a/inferencex-e2e/infx/bench/eval/vendor.py b/inferencex-e2e/infx/bench/eval/vendor.py new file mode 100644 index 0000000000..d8d6d68803 --- /dev/null +++ b/inferencex-e2e/infx/bench/eval/vendor.py @@ -0,0 +1,360 @@ +"""Vendor verifier evals: one runner drives each provider's pinned adapter in ``infx/evals/``.""" + +from __future__ import annotations + +import gzip +import os +import sys +import tarfile +import tempfile +from collections.abc import Callable, Mapping +from dataclasses import dataclass, replace +from pathlib import Path + +from infx.bench import env, proc +from infx.bench.eval.context import EvalContext, EvalOutcome + +EVALS = proc.REPO_ROOT / "infx" / "evals" +UV_REQUIREMENT = "uv==0.11.33" +KIMI_VERIFIER_REPO = "https://github.com/MoonshotAI/Kimi-Vendor-Verifier.git" +KIMI_VERIFIER_REF = "3dad65a760a8867cda72f6dd8848d876a4e851b4" +KIMI_VERIFIER_ARCHIVE_SHA256 = "ede9ea300c72ccfde9d8975ea4b1b54e423c7625690f6631ab1e65a715821e01" +KIMI_REQUIREMENTS = ( + "httpx[http2]==0.28.1", + "openai==2.14.0", + "jsonschema==4.25.1", + "pytest==8.4.2", +) +MINIMAX_REQUIREMENTS = ( + "jsonschema==4.25.1", + "loguru==0.7.3", + "megfile==4.2.5", + "numpy==2.3.4", + "openai==2.7.1", + "tqdm==4.67.1", +) +BFCL_INSTALL_TIMEOUT_S = 600 +BFCL_ARCHIVE = "bfcl_upstream_artifacts.tar.gz" + +_PROVISION = "Python runtime preparation" +_VERSION_CHECK = "import sys; raise SystemExit(sys.version_info < (3, int(sys.argv[1])))" + + +class StepError(Exception): + """A setup step failed; ``returncode`` becomes the eval's exit code.""" + + def __init__(self, message: str, returncode: int) -> None: + super().__init__(message) + self.returncode = returncode + + +@dataclass(frozen=True) +class Suite: + """One selectable suite of a provider.""" + + label: str + """Names the verifier in logs and integration-error messages.""" + adapter: str + """Script in ``infx/evals/``.""" + results: tuple[str, ...] + """Globs of the adapter's own report; all matching means it recorded the outcome itself.""" + requirements: tuple[str, ...] = () + timeout_s: int | None = None + """Adapter deadline; expiry kills it with exit code 124.""" + finalize: Callable[[Job], int] | None = None + """Runs after the adapter, even a failed one; nonzero fails a passing run.""" + + +@dataclass(frozen=True) +class Provider: + """How to provision, prepare, and drive one provider's adapter.""" + + default_suite: str + suites: Mapping[str, Suite] + python_minor: int + """The verifier interpreter must be Python ``3.`` or newer.""" + prepare: Callable[[Job], tuple[list[str], dict[str, str]]] + """Installs the pinned runtime; returns extra adapter arguments and environment.""" + system_site_packages: bool = False + """Run the adapter in a venv that also sees the image's site-packages.""" + suite_flag: str | None = None + """Adapter flag naming the suite; ``None`` when each suite has its own adapter.""" + run_command: tuple[str, ...] = () + failure_command: tuple[str, ...] = () + """Leading adapter arguments that write an integration-error result.""" + message_flag: str = "--integration-error" + + +@dataclass +class Job: + """One suite run, shared by the runner and the provider hooks.""" + + provider: Provider + ctx: EvalContext + suite: str + spec: Suite + scratch: Path + """Private directory for runtimes and checkouts; deleted after the run.""" + python: str = "python3" + """The verifier interpreter; the image python3 (3.10 floor) until provisioning replaces it.""" + + def step( + self, + stage: str, + argv: list[str], + *, + environ: Mapping[str, str] | None = None, + timeout_s: int | None = None, + ) -> None: + """Run one setup command; failure ends the run with an integration error.""" + rc = proc.call(argv, self.ctx.env if environ is None else environ, timeout=timeout_s) + if rc: + raise StepError(f"{self.spec.label} {stage} failed with exit code {rc}", rc) + + def adapter_argv(self, command: tuple[str, ...], *args: str) -> list[str]: + suite = [self.provider.suite_flag, self.suite] if self.provider.suite_flag else [] + return [ + self.python, str(EVALS / self.spec.adapter), *command, *args, + "--model", self.ctx.model, "--output-dir", str(self.ctx.results_dir), *suite, + ] # fmt: skip + + def published(self) -> bool: + results = self.ctx.results_dir + return all(any(p.is_file() for p in results.glob(glob)) for glob in self.spec.results) + + +def run(provider: Provider, ctx: EvalContext) -> EvalOutcome: + """Run ``ctx.suite``, else the provider's default suite.""" + suite = ctx.suite or provider.default_suite + spec = provider.suites.get(suite) + if spec is None: + title = provider.suites[provider.default_suite].label + raise env.InputError(f"unsupported {title} suite {suite!r}") + with tempfile.TemporaryDirectory( + prefix="infx-vendor-eval-", ignore_cleanup_errors=True + ) as scratch: + job = Job(provider, ctx, suite, spec, Path(scratch)) + try: + job.python = _provision(job) + args, adapter_env = provider.prepare(job) + except StepError as failure: + _publish_failure(job, str(failure)) + return EvalOutcome(failure.returncode, suite) + rc = proc.call( + job.adapter_argv(provider.run_command, *args, "--base-url", f"{ctx.base_url}/v1"), + {**ctx.env, **adapter_env}, + timeout=spec.timeout_s, + ) + finalized = spec.finalize(job) if spec.finalize else 0 + if rc and not job.published(): + _publish_failure(job, f"{spec.label} evaluation failed with exit code {rc}") + return EvalOutcome(rc or finalized, suite) + + +def _publish_failure(job: Job, message: str) -> None: + """Have the adapter record ``message`` as a zero-score integration-error result.""" + print(f"ERROR: {message}", file=sys.stderr) + provider = job.provider + argv = [*job.adapter_argv(provider.failure_command), provider.message_flag, message] + rc = proc.call(argv, job.ctx.env) + if not job.published(): + print( + f"ERROR: failed to write the {job.spec.label} failure artifact (exit code {rc})", + file=sys.stderr, + ) + + +def _provision(job: Job) -> str: + """Return the verifier interpreter, building a venv when the image python3 will not do.""" + minor, site_packages = job.provider.python_minor, job.provider.system_site_packages + new_enough = proc.call(["python3", "-c", _VERSION_CHECK, str(minor)], job.ctx.env) == 0 + if new_enough and not site_packages: + return "python3" + root = job.scratch / "python" + venv = root / "venv" + site = ["--system-site-packages"] if site_packages else [] + if new_enough: + job.step(_PROVISION, ["python3", "-m", "venv", *site, str(venv)]) + else: + prefix = root / "uv" + job.step(_PROVISION, [ + "python3", "-m", "pip", "install", "-q", "--no-cache-dir", "--break-system-packages", + "--prefix", str(prefix), UV_REQUIREMENT, + ]) # fmt: skip + uv = _executable(job, prefix / "bin" / "uv") + uv_env = { + **job.ctx.env, + "UV_CACHE_DIR": str(root / "uv-cache"), + "UV_PYTHON_INSTALL_DIR": str(root / "python"), + } + job.step( + _PROVISION, + [uv, "venv", "--python", f"3.{minor}", "--seed", *site, str(venv)], + environ=uv_env, + ) + return _executable(job, venv / "bin" / "python") + + +def _executable(job: Job, path: Path) -> str: + if not os.access(path, os.X_OK): + print(f"ERROR: {_PROVISION} did not create {path}", file=sys.stderr) + raise StepError(f"{job.spec.label} {_PROVISION} failed with exit code 1", 1) + return str(path) + + +def _pip_target(job: Job, target: Path) -> list[str]: + """Install the suite's pinned requirements into ``target``, outside the interpreter.""" + return [ + job.python, "-m", "pip", "install", "-q", "--no-cache-dir", "--target", str(target), + *job.spec.requirements, + ] # fmt: skip + + +def _prepare_kimi(job: Job) -> tuple[list[str], dict[str, str]]: + runtime = job.scratch / "kimi-runtime" + job.step("dependency installation", _pip_target(job, runtime)) + checkout = job.scratch / "kimi-verifier" + checkout.mkdir() + job.step("checkout", [ + job.python, str(EVALS / "_kimi_verifier_archive.py"), + KIMI_VERIFIER_REPO, KIMI_VERIFIER_REF, KIMI_VERIFIER_ARCHIVE_SHA256, str(checkout), + ]) # fmt: skip + args = ["--verifier-dir", str(checkout)] + if model_prefix := env.optional("MODEL_PREFIX", job.ctx.env): + args += ["--model-prefix", model_prefix] + return args, {"PYTHONPATH": os.pathsep.join([str(runtime), proc.pythonpath(job.ctx.env)])} + + +def _prepare_minimax(job: Job) -> tuple[list[str], dict[str, str]]: + source = job.scratch / "minimax-source" + dependencies = job.scratch / "minimax-deps" + stage = "pinned runtime preparation" + job.step(stage, [ + job.python, str(EVALS / "minimax_m3_full_eval.py"), + "prepare-source", "--source-dir", str(source), + ]) # fmt: skip + job.step(stage, _pip_target(job, dependencies)) + return [ + "--python", job.python, "--source-dir", str(source), "--dependency-dir", str(dependencies), + ], {} # fmt: skip + + +def _bfcl_project(job: Job) -> Path: + return job.scratch / "bfcl-project" + + +def _prepare_bfcl(job: Job) -> tuple[list[str], dict[str, str]]: + job.step("dependency installation", [ + job.python, str(EVALS / job.spec.adapter), + "--install-runtime", str(job.scratch / "bfcl-wheel"), + ], timeout_s=BFCL_INSTALL_TIMEOUT_S) # fmt: skip + project = _bfcl_project(job) + project.mkdir() + return ["--bfcl-project-root", str(project)], {} + + +def _archive_bfcl_project(job: Job) -> int: + """Keep BFCL's raw generations and scores next to the projected results.""" + try: + archive_tree(_bfcl_project(job), job.ctx.results_dir / BFCL_ARCHIVE) + except (OSError, ValueError) as error: + print(f"ERROR: failed to archive BFCL upstream artifacts: {error}", file=sys.stderr) + return 1 + return 0 + + +def archive_tree(root: Path, archive: Path) -> None: + """Write ``root`` as a byte-reproducible ``.tar.gz``; refuse symlinks and special files.""" + temporary = archive.with_name(f".{archive.name}.tmp") + temporary.unlink(missing_ok=True) + try: + with ( + temporary.open("xb") as raw, + gzip.GzipFile(filename="", mode="wb", fileobj=raw, mtime=0) as compressed, + tarfile.open(fileobj=compressed, mode="w", format=tarfile.PAX_FORMAT) as tar, + ): + for path in sorted( + root.rglob("*"), key=lambda entry: entry.relative_to(root).as_posix() + ): + name = path.relative_to(root).as_posix() + if path.is_symlink(): + raise ValueError(f"refusing to archive symbolic link: {name}") + info = tar.gettarinfo(str(path), arcname=name) + info.uid = info.gid = 0 + info.uname = info.gname = "" + info.mtime = 0 + if info.isdir(): + tar.addfile(info) + elif info.isfile(): + with path.open("rb") as source: + tar.addfile(info, source) + else: + raise ValueError(f"refusing to archive special file: {name}") + temporary.replace(archive) + except BaseException: + temporary.unlink(missing_ok=True) + raise + + +_KIMI = Suite( + label="Kimi Vendor Verifier", + adapter="kimi_vendor_eval.py", + results=("results_kimi_vendor_*.json",), + requirements=KIMI_REQUIREMENTS, +) +_BFCL = Suite( + label="BFCL", + adapter="bfcl_adapter.py", + results=("bfcl_report.json", "results_bfcl.json"), +) + +PROVIDERS: dict[str, Provider] = { + "kimi-vendor": Provider( + default_suite="kimi_tool_call_schema", + suites={ + "kimi_tool_call_schema": _KIMI, + "kimi_tool_call_schema_full": replace( + _KIMI, requirements=(*KIMI_REQUIREMENTS, "pytest-xdist==3.8.0") + ), + }, + python_minor=12, + prepare=_prepare_kimi, + suite_flag="--task-name", + ), + "minimax-vendor": Provider( + default_suite="minimax_m3_smoke", + suites={ + "minimax_m3_smoke": Suite( + label="MiniMax Provider Verifier", + adapter="minimax_provider_eval.py", + results=("results_minimax_vendor_*.json",), + requirements=MINIMAX_REQUIREMENTS, + ), + "minimax_m3_full": Suite( + label="MiniMax M3 full", + adapter="minimax_m3_full_eval.py", + results=("results_minimax_vendor_full_*.json",), + requirements=MINIMAX_REQUIREMENTS, + ), + }, + python_minor=12, + prepare=_prepare_minimax, + run_command=("run",), + failure_command=("failure",), + message_flag="--message", + ), + # BFCL supports Python 3.10 and reuses the image's installed stack. + "bfcl": Provider( + default_suite="bfcl_smoke", + suites={ + "bfcl_smoke": replace(_BFCL, timeout_s=900), + "bfcl_vllm_minimax_m3": replace(_BFCL, timeout_s=7200, finalize=_archive_bfcl_project), + "bfcl_vllm_kimi": replace(_BFCL, timeout_s=14400, finalize=_archive_bfcl_project), + }, + python_minor=10, + system_site_packages=True, + prepare=_prepare_bfcl, + suite_flag="--suite", + ), +} +"""Vendor providers by ``EVAL_FRAMEWORK`` name.""" diff --git a/inferencex-e2e/infx/bench/fixed_seq.py b/inferencex-e2e/infx/bench/fixed_seq.py new file mode 100644 index 0000000000..b943025c8d --- /dev/null +++ b/inferencex-e2e/infx/bench/fixed_seq.py @@ -0,0 +1,186 @@ +"""Fixed-sequence throughput lanes; ``client_argv`` owns the client policy they share.""" + +from __future__ import annotations + +import argparse +from dataclasses import dataclass +from pathlib import Path + +from infx.bench import env, gpu_monitor, proc, server + +# The client needs numpy and transformers, so it runs in a child, never in this process. +PYTHON = "python3" +# Single-node srt frameworks and the client backend that speaks their completions API. +SINGLE_NODE_BACKENDS = {"sglang": "vllm", "atom": "vllm", "trt": "openai"} +SINGLE_NODE_DEPENDENCIES = ( + "pip3", "install", "--break-system-packages", "sentencepiece", "datasets", "pandas", +) # fmt: skip +# Multi-node sweeps render prompts serially and keep progress bars out of the job log. +SWEEP_FLAGS = ("--random-num-workers", "1", "--disable-tqdm") + + +@dataclass(frozen=True) +class Point: + """One client run: ``num_prompts`` random ISL/OSL requests at concurrency ``conc``.""" + + base_url: str + model: str + backend: str + isl: int + osl: int + random_range_ratio: str + conc: int + num_prompts: int + result: Path + endpoint: str | None = None + tokenizer: str | None = None + use_chat_template: bool = False + trust_remote_code: bool = False + extra: tuple[str, ...] = () + + +def client_argv(point: Point) -> list[str]: + """An unthrottled burst that ignores EOS, after 2x-concurrency warmups.""" + argv = [ + PYTHON, "-m", "infx.bench_serving.benchmark_serving", + "--model", point.model, + "--backend", point.backend, + "--base-url", point.base_url, + "--dataset-name", "random", + "--random-input-len", str(point.isl), + "--random-output-len", str(point.osl), + "--random-range-ratio", point.random_range_ratio, + "--num-prompts", str(point.num_prompts), + "--max-concurrency", str(point.conc), + "--request-rate", "inf", + "--ignore-eos", + "--save-result", + "--num-warmups", str(2 * point.conc), + "--percentile-metrics", "ttft,tpot,itl,e2el", + "--result-dir", str(point.result.parent), + "--result-filename", point.result.name, + ] # fmt: skip + if point.endpoint is not None: + argv += ["--endpoint", point.endpoint] + if point.tokenizer is not None: + argv += ["--tokenizer", point.tokenizer] + if point.use_chat_template: + argv.append("--use-chat-template") + if point.trust_remote_code: + argv.append("--trust-remote-code") + return [*argv, *point.extra] + + +def served_model(base_url: str) -> str: + """The first model id the frontend lists.""" + listing = server.http_json(f"{base_url}/v1/models") + data = listing.get("data") if isinstance(listing, dict) else None + first = data[0] if isinstance(data, list) and data else None + model = first.get("id") if isinstance(first, dict) else None + if not isinstance(model, str) or not model: + raise env.BenchError(f"{base_url}/v1/models lists no served model") + return model + + +def srt_single(args: argparse.Namespace) -> int: + """One srt-slurm single-node point under the GPU monitor; srt-slurm owns the server.""" + values = env.require( + "MODEL", "CONC", "ISL", "OSL", "RANDOM_RANGE_RATIO", "RESULT_FILENAME", "RESULT_DIR", + "SRT_FRONTEND_HOST", "SRT_FRONTEND_PORT", "RUN_EVAL", "EVAL_ONLY", + "GPU_MONITOR_INTERVAL", "USE_CHAT_TEMPLATE", "FRAMEWORK", + ) # fmt: skip + env.flag("RUN_EVAL") + eval_only = env.flag("EVAL_ONLY") + backend = SINGLE_NODE_BACKENDS.get(values["FRAMEWORK"]) + if backend is None: + raise env.InputError(f"unsupported fixed-sequence FRAMEWORK: {values['FRAMEWORK']}") + conc = env.positive_int("CONC") + result_dir = Path(values["RESULT_DIR"]) + point = Point( + base_url=f"http://{values['SRT_FRONTEND_HOST']}:{env.positive_int('SRT_FRONTEND_PORT')}", + model=values["MODEL"], + backend=backend, + isl=env.positive_int("ISL"), + osl=env.positive_int("OSL"), + random_range_ratio=values["RANDOM_RANGE_RATIO"], + conc=conc, + num_prompts=10 * conc, + result=result_dir / f"{values['RESULT_FILENAME']}.json", + use_chat_template=env.flag("USE_CHAT_TEMPLATE"), + trust_remote_code=args.trust_remote_code, + ) + interval = env.positive_int("GPU_MONITOR_INTERVAL") + if not result_dir.is_dir(): + raise env.InputError("RESULT_DIR must be an existing runtime-provided directory") + if eval_only: + print("EVAL_ONLY mode: skipping throughput benchmark", flush=True) + return 0 + rc = proc.call(SINGLE_NODE_DEPENDENCIES) + if rc: + return rc + return gpu_monitor.run(result_dir / "gpu_metrics.csv", interval, client_argv(point)) + + +def srt_sweep(args: argparse.Namespace) -> int: + """Every ``CONC_LIST`` point of an srt-slurm multi-node job; srt-slurm owns the servers.""" + values = env.require( + "ISL", "OSL", "RANDOM_RANGE_RATIO", "SRT_FRONTEND_HOST", "SRT_FRONTEND_PORT", + "CONC_LIST", "PREFILL_NUM_WORKERS", "PREFILL_TP", "DECODE_NUM_WORKERS", "DECODE_TP", + ) # fmt: skip + isl = env.positive_int("ISL") + osl = env.positive_int("OSL") + base_url = f"http://{values['SRT_FRONTEND_HOST']}:{env.positive_int('SRT_FRONTEND_PORT')}" + # Aggregated rows export DECODE_NUM_WORKERS=0; the result name still records it. + ctx = env.non_negative_int("PREFILL_NUM_WORKERS") * env.positive_int("PREFILL_TP") + gen = env.non_negative_int("DECODE_NUM_WORKERS") * env.positive_int("DECODE_TP") + concs = [env.parse_positive_int("CONC_LIST", word) for word in values["CONC_LIST"].split()] + windows = env.optional("SRT_MEASUREMENT_WINDOW_DIR") + # The name the workers registered; the workflow's MODEL is the HF id, which can differ. + model = served_model(base_url) + tokenizer = env.optional("TOKENIZER") or model + result_dir = args.logs_dir / f"sa-bench_isl_{isl}_osl_{osl}" + result_dir.mkdir(parents=True, exist_ok=True) + with proc.RelaySignals() as relay: + for conc in concs: + name = f"results_concurrency_{conc}_gpus_{ctx + gen}_ctx_{ctx}_gen_{gen}.json" + point = Point( + base_url=base_url, + model=model, + backend="openai", + endpoint="/v1/completions", + tokenizer=tokenizer, + isl=isl, + osl=osl, + random_range_ratio=values["RANDOM_RANGE_RATIO"], + conc=conc, + num_prompts=10 * conc, + result=result_dir / name, + use_chat_template=True, + trust_remote_code=True, + extra=SWEEP_FLAGS, + ) + rc = relay.run(client_argv(point)) + # Power lanes: tell srt-slurm which interval this concurrency's result measured. + if rc == 0 and windows: + window = [PYTHON, "-m", "infx.results.power.window", str(point.result), str(conc)] + rc = relay.run(window) + if rc: + return rc + return 0 + + +def main(argv: list[str]) -> int: + """Run the ``fixed-seq`` command.""" + parser = argparse.ArgumentParser(prog="python3 -m infx.bench fixed-seq") + modes = parser.add_subparsers(dest="mode", required=True) + + single = modes.add_parser("srt-single", help="one srt-slurm single-node point (env)") + single.add_argument("--trust-remote-code", action="store_true") + single.set_defaults(run=srt_single) + + sweep = modes.add_parser("srt-sweep", help="every CONC_LIST point of a multi-node job (env)") + sweep.add_argument("--logs-dir", type=Path, required=True, help="srt-slurm's log mount") + sweep.set_defaults(run=srt_sweep) + + args = parser.parse_args(argv) + return args.run(args) diff --git a/inferencex-e2e/infx/bench/gpu_monitor.py b/inferencex-e2e/infx/bench/gpu_monitor.py new file mode 100644 index 0000000000..8a58c260e7 --- /dev/null +++ b/inferencex-e2e/infx/bench/gpu_monitor.py @@ -0,0 +1,189 @@ +"""GPU telemetry around a workload, in the files ``infx.results.power`` reads.""" + +from __future__ import annotations + +import os +import shutil +import subprocess +import sys +import threading +import time +from collections.abc import Sequence +from pathlib import Path +from typing import IO, final + +from infx.bench import proc + +NVIDIA_QUERY = ( + "timestamp,index,power.draw,temperature.gpu,clocks.current.sm," + "clocks.current.memory,utilization.gpu,utilization.memory" +) +NVIDIA_IDENTITY = ( + "nvidia-smi", "--query-gpu=index,uuid,pci.bus_id,name,driver_version", "--format=csv", +) # fmt: skip +NVIDIA_SAMPLE = ("nvidia-smi", f"--query-gpu={NVIDIA_QUERY}", "--format=csv,noheader") +AMD_ENERGY = ("amd-smi", "metric", "-E", "--csv") +AMD_IDENTITY = ("amd-smi", "static", "--json") +STOP_TIMEOUT_S = 10 + + +def _say(message: str, *, error: bool = False) -> None: + print(f"[GPU Monitor] {message}", file=sys.stderr if error else sys.stdout, flush=True) + + +def _write(path: Path, command: Sequence[str], mode: str = "wb") -> bool: + """Send ``command``'s stdout to ``path``; return whether it succeeded.""" + try: + with path.open(mode) as out: + done = subprocess.run(command, stdout=out, stderr=subprocess.DEVNULL, check=False) + except OSError: + return False + return done.returncode == 0 + + +def _copy_amd_rows(source: IO[bytes], sink: IO[bytes]) -> None: + """Forward whole ``amd-smi`` rows under one header, dropping its banner and repeated headers.""" + with source, sink: + header_seen = False + for line in source: + if not line.endswith(b"\n"): + break + if line.startswith(b"timestamp,"): + if header_seen: + continue + header_seen = True + if header_seen: + sink.write(line) + sink.flush() + + +def _repair_tail(path: Path) -> bool: + """Drop a partial final row; return whether ``path`` now ends on a row boundary.""" + try: + with path.open("r+b") as stream: + size = keep = stream.seek(0, os.SEEK_END) + while keep: + start = max(0, keep - 4096) + stream.seek(start) + newline = stream.read(keep - start).rfind(b"\n") + if newline != -1: + keep = start + newline + 1 + break + keep = start + if keep == size: + return True + stream.truncate(keep) + except FileNotFoundError: + return True + except OSError: + _say("Warning: could not repair truncated trailing sample", error=True) + return False + _say("Dropped truncated trailing sample") + return True + + +def _count_lines(path: Path) -> int: + with path.open("rb") as stream: + return sum(chunk.count(b"\n") for chunk in iter(lambda: stream.read(1 << 16), b"")) + + +@final +class GpuMonitor: + """Sample ``nvidia-smi``, else ``amd-smi``, into output; ``vendor`` is None without either.""" + + def __init__(self, output: Path, interval: int) -> None: + self.output = output + self.interval = interval + self.vendor: str | None = None + self._stem = str(output).removesuffix(".csv") + self._sampler: subprocess.Popen[bytes] | None = None + self._copier: threading.Thread | None = None + + def __enter__(self) -> GpuMonitor: + interval = str(self.interval) + if shutil.which("nvidia-smi"): + self.vendor = "nvidia" + self._sidecar("_identity.csv", NVIDIA_IDENTITY, "NVIDIA identity") + with self.output.open("wb") as out: + self._sampler = subprocess.Popen( + ["nvidia-smi", f"--query-gpu={NVIDIA_QUERY}", "--format=csv", "-l", interval], + stdout=out, + stderr=subprocess.DEVNULL, + ) + elif shutil.which("amd-smi"): + self.vendor = "amd" + sink = self.output.open("wb") + # amd-smi is Python and block-buffers a pipe; unbuffered, no tick waits for exit. + self._sampler = subprocess.Popen( + ["amd-smi", "metric", "-p", "-c", "-t", "-u", "-w", interval, "--csv"], + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + env={**os.environ, "PYTHONUNBUFFERED": "1"}, + ) + self._copier = threading.Thread( + target=_copy_amd_rows, args=(self._sampler.stdout, sink), daemon=True + ) + self._copier.start() + # Accumulator snapshots bracket the stream so auditors can cross-check its energy. + self._sidecar("_energy_start.csv", AMD_ENERGY, "amd-smi metric") + self._sidecar("_identity.json", AMD_IDENTITY, "amd-smi static") + else: + _say("No GPU monitoring tool found (nvidia-smi or amd-smi), skipping") + return self + _say( + f"Started {self.vendor.upper()} (PID={self._sampler.pid}, " + f"interval={self.interval}s, output={self.output})" + ) + return self + + def __exit__(self, *_: object) -> None: + sampler, self._sampler = self._sampler, None + if sampler is None: + return + if sampler.poll() is not None: + self._join_copier() + rc = proc.status(sampler.returncode) + _say(f"Warning: sampler exited early (rc={rc})", error=True) + return + if self.vendor == "amd": + # The stream must bracket the workload's end, and amd-smi stamps whole seconds. + time.sleep(self.interval + 2) + sampler.terminate() + try: + sampler.wait(timeout=STOP_TIMEOUT_S) + except subprocess.TimeoutExpired: + sampler.kill() + sampler.wait() + self._join_copier() + whole = _repair_tail(self.output) + if self.vendor == "nvidia": + # The stream can stop just before the workload ends; one more sample brackets it. + if whole and not _write(self.output, NVIDIA_SAMPLE, "ab"): + _say("Warning: final NVIDIA sample failed", error=True) + else: + self._sidecar("_energy_end.csv", AMD_ENERGY, "amd-smi metric") + _say(f"Stopped (PID={sampler.pid})") + if self.output.is_file(): + _say(f"Collected {_count_lines(self.output)} rows -> {self.output}") + + def _sidecar(self, suffix: str, command: Sequence[str], what: str) -> None: + """Snapshot ``command`` beside the stream; a failure costs only this file.""" + path = Path(f"{self._stem}{suffix}") + if not _write(path, command): + path.unlink(missing_ok=True) + _say(f"Warning: {what} sidecar failed", error=True) + + def _join_copier(self) -> None: + copier, self._copier = self._copier, None + if copier is not None: + copier.join(timeout=STOP_TIMEOUT_S) + if copier.is_alive(): + _say("Warning: amd-smi output did not close after the sampler stopped", error=True) + + +def run(output: Path, interval: int, command: Sequence[str]) -> int: + """Run ``command`` under the monitor, relaying signals; return its status.""" + with proc.RelaySignals() as relay, GpuMonitor(output, interval): + rc = relay.run(command) + # A signal while the sampler stops still fails a clean run. + return 128 + relay.received if rc == 0 and relay.received else rc diff --git a/inferencex-e2e/infx/bench/proc.py b/inferencex-e2e/infx/bench/proc.py new file mode 100644 index 0000000000..439925b618 --- /dev/null +++ b/inferencex-e2e/infx/bench/proc.py @@ -0,0 +1,133 @@ +"""Child processes: shell-style exit statuses, signal handling, and this checkout's PYTHONPATH.""" + +from __future__ import annotations + +import io +import os +import shlex +import signal +import subprocess +import sys +from collections.abc import Mapping, Sequence +from pathlib import Path +from types import FrameType +from typing import TYPE_CHECKING, Any, cast + +if TYPE_CHECKING: + from typing_extensions import Self + +REPO_ROOT = Path(__file__).resolve().parents[2] + + +def status(returncode: int) -> int: + """Shell-style status: death by signal N is 128 + N.""" + return 128 - returncode if returncode < 0 else returncode + + +def echo(argv: Sequence[str]) -> None: + """Print ``argv`` like a ``set -x`` shell line, after any pending output.""" + sys.stdout.flush() + print(f"+ {shlex.join(argv)}", file=sys.stderr, flush=True) + + +def call( + argv: Sequence[str], + env: Mapping[str, str] | None = None, + *, + cwd: Path | None = None, + timeout: float | None = None, +) -> int: + """Run ``argv`` echoed like ``set -x``; 127 if it cannot start, 124 past ``timeout``.""" + echo(argv) + try: + return status( + subprocess.run(argv, env=env, cwd=cwd, timeout=timeout, check=False).returncode + ) + except subprocess.TimeoutExpired: + print(f"ERROR: {argv[0]} exceeded its {timeout:g}s deadline", file=sys.stderr) + return 124 + except OSError as error: + print(f"ERROR: cannot run {argv[0]}: {error}", file=sys.stderr) + return 127 + + +def tee(argv: Sequence[str], log: Path, env: Mapping[str, str] | None = None) -> int: + """``argv 2>&1 | tee log``; return argv's status.""" + echo(argv) + with ( + log.open("wb") as sink, + subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, env=env) as child, + ): + output = cast("io.BufferedReader", child.stdout) + while chunk := output.read1(1 << 16): + for stream in (sys.stdout.buffer, sink): + stream.write(chunk) + stream.flush() + return status(child.returncode) + + +def pythonpath(environ: Mapping[str, str] = os.environ) -> str: + """``PYTHONPATH`` with this checkout first; containers never install ``infx``.""" + root = str(REPO_ROOT) + rest = [ + path for path in environ.get("PYTHONPATH", "").split(os.pathsep) if path not in {"", root} + ] + return os.pathsep.join([root, *rest]) + + +SIGNALS = (signal.SIGINT, signal.SIGTERM, signal.SIGHUP) + + +class _Signals: + """Handle ``SIGNALS`` for the ``with`` block and remember the first one received.""" + + def __init__(self) -> None: + self.received: int | None = None + self._previous: dict[int, Any] = {} + + def __enter__(self) -> Self: + for signum in SIGNALS: + self._previous[signum] = signal.signal(signum, self._handle) + return self + + def __exit__(self, *_: object) -> None: + for signum, handler in self._previous.items(): + signal.signal(signum, handler) + + def _handle(self, signum: int, _frame: FrameType | None) -> None: + self.received = self.received or signum + + +class DeferSignals(_Signals): + """Hold signals until the block ends; the foreground child in our group gets them itself.""" + + +class RelaySignals(_Signals): + """Run commands one at a time, relaying signals; after one, nothing new starts.""" + + def __init__(self) -> None: + super().__init__() + self._child: subprocess.Popen[bytes] | None = None + + def _handle(self, signum: int, frame: FrameType | None) -> None: + super()._handle(signum, frame) + if self._child is not None: + self._child.send_signal(signum) + + def run(self, command: Sequence[str]) -> int: + """Run ``command`` to completion; return its status.""" + if self.received: + return 128 + self.received + echo(command) + try: + self._child = child = subprocess.Popen(command) + except OSError as error: + print(f"ERROR: cannot run {command[0]}: {error}", file=sys.stderr, flush=True) + return 127 + if self.received: + child.send_signal(self.received) + try: + rc = status(child.wait()) + finally: + self._child = None + return 128 + self.received if rc == 0 and self.received else rc diff --git a/inferencex-e2e/infx/bench/server.py b/inferencex-e2e/infx/bench/server.py new file mode 100644 index 0000000000..51d80e10ce --- /dev/null +++ b/inferencex-e2e/infx/bench/server.py @@ -0,0 +1,134 @@ +"""Server readiness; ``wait --url URL [--pid PID] [--log FILE]`` blocks until it is ready.""" + +from __future__ import annotations + +import argparse +import http.client +import json +import os +import sys +import time +import urllib.error +import urllib.request +from collections.abc import Mapping +from pathlib import Path + +from infx.bench import env + +REQUEST_TIMEOUT_S = 10 +POLL_S = 5.0 + + +class NotReadyError(env.BenchError): + """The server died or the readiness budget ran out.""" + + +def http_status(url: str) -> int | None: + """The response status, or ``None`` when no HTTP response arrived.""" + try: + with urllib.request.urlopen(url, timeout=REQUEST_TIMEOUT_S) as response: # noqa: S310 + return response.status + except urllib.error.HTTPError as error: + return error.code + except (OSError, ValueError, http.client.HTTPException): + return None + + +def http_json(url: str) -> object | None: + """A successful JSON response body, or ``None``.""" + try: + with urllib.request.urlopen(url, timeout=REQUEST_TIMEOUT_S) as response: # noqa: S310 + return json.load(response) + except (OSError, ValueError, http.client.HTTPException): + return None + + +def process_alive(pid: int) -> bool: + """A zombie waiting to be reaped is not a live server.""" + try: + os.kill(pid, 0) + except ProcessLookupError: + return False + except PermissionError: + return True + try: + state = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split()[0] + except (FileNotFoundError, IndexError): + return True + return state not in {"Z", "X"} + + +def wait_ready( + url: str, *, pid: int | None = None, log: Path | None = None, poll_s: float = POLL_S +) -> None: + """Poll ``url`` until it answers below 400, streaming ``log``; fail once ``pid`` exits.""" + offset = 0 + while True: + if log is not None and log.exists(): + with log.open("rb") as stream: + stream.seek(offset) + chunk = stream.read() + offset += len(chunk) + sys.stdout.write(chunk.decode(errors="replace")) + sys.stdout.flush() + status = http_status(url) + if status is not None and status < 400: + return + if pid is not None and not process_alive(pid): + raise NotReadyError(f"process {pid} died before {url} became ready") + time.sleep(poll_s) + + +def chat_route_budget(environ: Mapping[str, str] = os.environ) -> tuple[int, int]: + """The workflow's chat-route readiness ``(timeout, stabilization)`` in seconds.""" + names = ("EVAL_ENDPOINT_READY_TIMEOUT_SECONDS", "EVAL_MODEL_STABILIZATION_SECONDS") + env.require(*names, env=environ) + return env.positive_int(names[0], environ), env.non_negative_int(names[1], environ) + + +def wait_chat_route( + base_url: str, model: str, budget: tuple[int, int], *, poll_s: float = POLL_S +) -> None: + """Wait until ``model`` is served and the chat route is mounted.""" + timeout_s, stabilization_s = budget + chat_url = f"{base_url}/v1/chat/completions" + start = time.monotonic() + healthy_since: float | None = None + next_report = 0.0 + while True: + now = time.monotonic() + models = http_json(f"{base_url}/v1/models") + entries = models.get("data", []) if isinstance(models, dict) else [] + served = any(isinstance(entry, dict) and entry.get("id") == model for entry in entries) + # A bare GET answering 401/403/405 proves the route is mounted. + if served and http_status(chat_url) in {401, 403, 405}: + break + status = http_status(f"{base_url}/health") + if status is None or status >= 400: + healthy_since = None + elif healthy_since is None: + healthy_since = now + # Some frontends list no model before the first request; sustained health counts. + if healthy_since is not None and now - healthy_since >= stabilization_s: + break + elapsed = now - start + if elapsed >= timeout_s: + raise NotReadyError( + f"chat endpoint for model {model!r} not ready within {timeout_s}s: {chat_url}" + ) + if elapsed >= next_report: + print(f"Waiting for {chat_url} ({model!r}): {int(elapsed)}/{timeout_s}s", flush=True) + next_report += 60 + time.sleep(poll_s) + print(f"OpenAI chat endpoint ready for model {model!r}: {chat_url}", flush=True) + + +def main(argv: list[str]) -> int: + """Run the ``wait`` command.""" + parser = argparse.ArgumentParser(prog="python3 -m infx.bench wait") + parser.add_argument("--url", required=True, help="health URL that answers below 400 when ready") + parser.add_argument("--pid", type=int, help="server process; fail if it exits") + parser.add_argument("--log", type=Path, help="server log to stream while waiting") + args = parser.parse_args(argv) + wait_ready(args.url, pid=args.pid, log=args.log) + return 0 diff --git a/inferencex-e2e/infx/bench_serving/server_watch.py b/inferencex-e2e/infx/bench_serving/server_watch.py deleted file mode 100644 index efb0f67df8..0000000000 --- a/inferencex-e2e/infx/bench_serving/server_watch.py +++ /dev/null @@ -1,155 +0,0 @@ -"""Stop an owned benchmark/eval client when its ready server or worker exits.""" - -from __future__ import annotations - -import argparse -import json -import os -import re -import signal -import subprocess -import sys -from pathlib import Path -from types import FrameType - - -def process_state(pid: int) -> tuple[str, str] | None: - """A zombie is dead, and a reused PID is a different process.""" - try: - stat = Path(f"/proc/{pid}/stat").read_text().rsplit(")", 1)[1].split() - return stat[0], stat[19] - except FileNotFoundError: - if Path("/proc").exists(): - return None - # Local macOS checks; production runners use Linux /proc. - result = subprocess.run( - ["ps", "-p", str(pid), "-o", "stat=", "-o", "lstart="], - capture_output=True, - text=True, - check=False, - ) - fields = result.stdout.strip().split(maxsplit=1) - return (fields[0], fields[1]) if len(fields) == 2 else None - - -def snapshot(pid: int) -> dict[str, str]: - if pid <= 0: - raise ValueError("Expected a positive server PID") - listing = subprocess.check_output(["ps", "-eo", "pid=,ppid=,args="], text=True) - processes = [line.strip().split(maxsplit=2) for line in listing.splitlines()] - descendants = {pid} - while True: - expanded = descendants | { - int(p[0]) for p in processes if len(p) == 3 and int(p[1]) in descendants - } - if expanded == descendants: - break - descendants = expanded - # These are persistent required engine processes, not transient tokenizer or - # HTTP request subprocesses. The root alone would miss a dead scheduler. - worker = re.compile(r"sglang::scheduler|EngineCore|VllmWorker|TPWorker|trtllm-worker") - required = {pid} | { - int(p[0]) - for p in processes - if len(p) == 3 and int(p[0]) in descendants and worker.search(p[2]) - } - result = {} - for required_pid in required: - state = process_state(required_pid) - if not state or state[0].startswith(("Z", "X")): - raise ValueError("Server exited before the client started") - result[str(required_pid)] = state[1] - return result - - -def healthy(required: dict[str, str]) -> bool: - return bool(required) and all( - (state := process_state(int(pid))) - and state[1] == start - and not state[0].startswith(("Z", "X")) - for pid, start in required.items() - ) - - -def _signal_group(pgid: int, sig: signal.Signals) -> bool: - try: - os.killpg(pgid, sig) - return True - except ProcessLookupError: - return False - except PermissionError: - if sys.platform != "darwin": - raise - listing = subprocess.check_output(["ps", "-eo", "pgid=,stat="], text=True) - if any( - group == str(pgid) and not state.startswith(("Z", "X")) - for group, state in (line.split() for line in listing.splitlines()) - ): - raise - return False - - -def stop(client: subprocess.Popen) -> None: - # The leader may already have exited while leaving its own workers behind. - if not _signal_group(client.pid, signal.SIGTERM): - return - try: - client.wait(timeout=5) - except subprocess.TimeoutExpired: - pass - finally: - _signal_group(client.pid, signal.SIGKILL) - client.wait() - - -def run(required: dict[str, str], command: list[str], interval: float = 2) -> int: - if not healthy(required): - print( - "ERROR: ready server or required worker exited before client dispatch", - flush=True, - ) - return 1 - - def interrupted(_signum: int, _frame: FrameType | None) -> None: - raise KeyboardInterrupt - - previous = signal.signal(signal.SIGTERM, interrupted) - try: - with subprocess.Popen(command, start_new_session=True) as client: - try: - while True: - try: - return client.wait(timeout=interval) - except subprocess.TimeoutExpired: - if not healthy(required): - print( - "ERROR: ready server or required worker exited; stopping its benchmark/eval client", - flush=True, - ) - return 1 - finally: - stop(client) - finally: - signal.signal(signal.SIGTERM, previous) - - -def main() -> int: - parser = argparse.ArgumentParser(description=__doc__) - commands = parser.add_subparsers(dest="operation", required=True) - capture = commands.add_parser("capture") - capture.add_argument("--pid", type=int, required=True) - execute = commands.add_parser("run") - execute.add_argument("--state", type=Path, required=True) - execute.add_argument("command", nargs=argparse.REMAINDER) - args = parser.parse_args() - if args.operation == "capture": - print(json.dumps(snapshot(args.pid))) - return 0 - command = args.command[1:] if args.command[:1] == ["--"] else args.command - if not command: - parser.error("Missing client command") - return run(json.loads(args.state.read_text()), command) - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/inferencex-e2e/infx/evals/EVALS.md b/inferencex-e2e/infx/evals/EVALS.md index 69e3387279..dbe0ee7cab 100644 --- a/inferencex-e2e/infx/evals/EVALS.md +++ b/inferencex-e2e/infx/evals/EVALS.md @@ -128,29 +128,52 @@ malformed metadata, duplicates, and raw/aggregate mismatches are not. See ## How? -`run_eval` in `benchmarks/benchmark_lib.sh` dispatches to the selected eval -runner. `e2e-tests.yml` defaults `eval-framework` to `auto`, then reads the -concrete framework and suite from each eval matrix row. Fixed-sequence and -opted-in generic agentic evals use +`python3 -m infx.bench eval` ([`infx/bench/eval/`](../bench/eval/__init__.py)) runs +the selected eval framework against a ready server. `e2e-tests.yml` defaults +`eval-framework` to `auto`, then reads the concrete framework and suite from each +eval matrix row. Fixed-sequence and opted-in generic agentic evals use [lm-evaluation-harness](https://github.com/EleutherAI/lm-evaluation-harness) (`lm-eval`) with GSM8K. Workflow inputs can explicitly override the matrix-selected framework or suite for manual diagnostics. +To run it by hand, use Python 3.10 or newer from `inferencex-e2e/`, normally inside +the serving container: + +```text +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint URL --concurrency "N [N ...]" --stage-to DIR [--framework NAME] +``` + +The command requires `MODEL` and the `true`/`false` flags `EVAL_ONLY` and +`IS_MULTINODE`. `MODEL_NAME` is the served name sent in requests and defaults to +`MODEL`. The framework is `EVAL_FRAMEWORK`, else `--framework`, else `lm-eval`. +lm-eval also requires `OPENAI_API_KEY` (the workflows set `EMPTY`) and pip-installs +its pinned harness into the running `python3`. Each run writes into a fresh temporary +directory, then copies the allow-listed artifacts into `--stage-to` and writes +`meta_env.json` there, also when the eval fails. `EVAL_CONCURRENT_REQUESTS` and +`EVAL_RESULT_DIR` are no longer read. + The Kimi full suite runs automatically for every generated `kimik3` agentic point. The matrix selects `eval-framework: kimi-vendor` and `eval-suite: kimi_tool_call_schema_full`. To invoke the full suite manually from the `inferencex-e2e/` directory after a server is ready: ```bash -source benchmarks/benchmark_lib.sh +export MODEL='' MODEL_NAME='' +export MODEL_PREFIX='' PORT='' CONC='' +export EVAL_ONLY=false IS_MULTINODE=false export EVAL_FRAMEWORK=kimi-vendor export EVAL_SUITE=kimi_tool_call_schema_full -export EVAL_RESULT_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" -run_eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency "$CONC" --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` +Vendor suites do not use the concurrency for their requests. For them, +`--concurrency` only records the point's `conc` in `meta_env.json`. + For a short endpoint check, explicitly set `EVAL_SUITE=kimi_tool_call_schema` instead (or the same `eval-suite` workflow input). Historical smoke artifacts retain their original task name and sample counts; they are not full-suite @@ -170,10 +193,14 @@ is diagnostic, while missing outcomes and integration failures fail the job. The framework selects a suite-specific subprocess adapter, while the suite selects a case set understood by that adapter. Each adapter owns its endpoint -format, dependencies, native report, metrics, and integration-failure policy. -Kimi, MiniMax, and BFCL use separate explicit `run_eval` cases rather than a -shared request or report abstraction. Automatic selection chooses only the Kimi -and MiniMax vendor cases; BFCL remains an explicit workflow override. +format, native report, metrics, and integration-failure policy. Kimi, MiniMax, +and BFCL are data entries in `PROVIDERS` in +[`infx/bench/eval/vendor.py`](../bench/eval/vendor.py), run by one generic runner +that provisions the verifier interpreter, installs the suite's pinned runtime, +runs the adapter under the suite deadline, and has the adapter write an +integration-error result when a step fails. There is no shared request or report +abstraction. Automatic selection chooses only the Kimi and MiniMax vendor cases; +BFCL remains an explicit workflow override. Agentic eval jobs forward the matrix `spec-decoding` value, so MTP entries launch their existing `*_mtp.sh` server instead of silently falling back to STP. @@ -252,14 +279,16 @@ agentic points select the full 102-case suite. To invoke the smoke manually from the `inferencex-e2e/` directory against an already-ready server: ```bash -source benchmarks/benchmark_lib.sh +export MODEL='' MODEL_NAME="" +export PORT='' CONC='' +export EVAL_ONLY=false IS_MULTINODE=false export EVAL_FRAMEWORK=minimax-vendor -export MODEL_NAME="" export EVAL_SUITE=minimax_m3_smoke -export EVAL_RESULT_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" -run_eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency "$CONC" --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` `infx/evals/minimax_m3_smoke.json` is derived from @@ -312,13 +341,16 @@ It covers all 102 rows in the pinned MiniMax Provider Verifier dataset; complete quality scores remain diagnostic. It can also be selected explicitly: ```bash -source benchmarks/benchmark_lib.sh +export MODEL='' MODEL_NAME='' +export PORT='' CONC='' +export EVAL_ONLY=false IS_MULTINODE=false export EVAL_FRAMEWORK=minimax-vendor export EVAL_SUITE=minimax_m3_full -export MODEL_NAME='' -run_eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency "$CONC" --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` The runner downloads only the eight source and validator files allowlisted in @@ -344,14 +376,16 @@ chat-completions endpoint. Select `eval-framework: bfcl` and against an already-ready server: ```bash -source benchmarks/benchmark_lib.sh +export MODEL='' MODEL_NAME="" +export PORT='' CONC='' +export EVAL_ONLY=false IS_MULTINODE=false export EVAL_FRAMEWORK=bfcl -export MODEL_NAME="" export EVAL_SUITE=bfcl_smoke -export EVAL_RESULT_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" -run_eval --port "$PORT" -append_lm_eval_summary -python3 -m infx.evals.validate_scores +EVAL_DIR="$(mktemp -d /tmp/eval_out-XXXXXX)" +PYTHONSAFEPATH=1 PYTHONPATH="$PWD${PYTHONPATH:+:$PYTHONPATH}" python3 -m infx.bench eval \ + --endpoint "http://localhost:$PORT" --concurrency "$CONC" --stage-to "$EVAL_DIR" +python3 -m infx.evals.validate_scores \ + --meta-env "$EVAL_DIR/meta_env.json" --results-glob "$EVAL_DIR/results*.json" ``` The validator reads BFCL's declared `acc` metric from the compatibility result, @@ -422,8 +456,8 @@ installed engine version are authoritative; BFCL does not replace a missing or mismatched parser/chat template. `bfcl_report.json` is the native report. `results_bfcl.json` is the -`inferencex-eval-v1` compatibility result consumed by the existing artifact -upload, `append_lm_eval_summary`, collector, and score validator. It projects +`inferencex-eval-v1` compatibility result consumed by the eval command's artifact +staging, the workflow upload, the collector, and the score validator. It projects the four-case aggregate as task `bfcl_smoke` and the four one-case diagnostic tasks shown above. Every row uses lm-eval-compatible `acc,none` (plus `acc_stderr,none`); BFCL workflows therefore validate with metric prefix @@ -483,68 +517,52 @@ case IDs, failure records, and sampling settings. The compatibility `results_bfcl.json` remains the only input to the normal InferenceX eval collector and dashboard path. -### Benchmark script flow +### Benchmark and eval flow -All benchmark scripts in `benchmarks/` follow one of two flows: +Every lane runs its eval with `python3 -m infx.bench eval` against the job's own +server. In combined mode (`RUN_EVAL=true`, `EVAL_ONLY=false`) the server starts, +throughput runs, and then the eval runs against the same server. In eval-only mode +(`EVAL_ONLY=true`) the server starts with its eval-only settings, throughput is +skipped, and the eval runs. -```bash -# Combined mode (benchmark + eval): -# 1. Start server (with context-length expansion if EVAL_ONLY=true) -# 2. wait_for_server_ready -# 3. run_benchmark_serving (skipped automatically when EVAL_ONLY=true) -# 4. Run evals: -if [ "${RUN_EVAL}" = "true" ]; then - run_eval --framework lm-eval --port "$PORT" - append_lm_eval_summary # Writes meta_env.json and stages artifacts -fi - -# Eval-only mode (EVAL_ONLY=true): -# 1. Compute eval context via compute_eval_context_length -# 2. Start server with that context (--context-length or --max-model-len) -# 3. wait_for_server_ready -# 4. run_benchmark_serving returns immediately (skipped) -# 5. run_eval + append_lm_eval_summary -``` +| Lane | Eval entrypoint | `--concurrency` | `--stage-to` | +|------|-----------------|-----------------|--------------| +| srt-slurm single-node | `post_eval.command` runs `benchmarks/single_node/srt_eval.sh /logs/infx-eval-exit-code` | `CONC` | The checkout root | +| srt-slurm multi-node | `post_eval.command` runs `benchmarks/multi_node/srt_eval.sh /infmax-workspace` | `EVAL_CONC` | `/logs/eval_results` | + +Key eval modules in `infx/bench/eval/`: + +| Module | Description | +|--------|-------------| +| `__init__.py` (`evaluate`) | The `eval` command. Selects the framework, checks `EVAL_SUITE`, waits for the chat route before vendor evals in eval-only jobs, batches lm-eval concurrencies, stages artifacts, writes `meta_env.json`, and sets the exit code | +| `lm_eval.py` | lm-eval framework. `install` installs the pinned harness commit, `context_length` and `native_context_length` size each request, and `run` drives `local-chat-completions` with the sitecustomize patch | +| `vendor.py` | One generic `run` for the Kimi, MiniMax, and BFCL entries in `PROVIDERS`. `_provision` chooses the verifier interpreter (the image `python3` when new enough, otherwise a venv built with pinned `uv`, and always a system-site-packages venv for BFCL). The `_prepare_kimi`, `_prepare_minimax`, and `_prepare_bfcl` hooks install the pinned runtimes, and failed steps get integration-error results | +| `meta.py` | `build` and `write` produce `meta_env.json`. `_disaggregated` maps the multi-node `PREFILL_*`/`DECODE_*` topology, and `refresh` serves the host-side srt collector | +| `stage.py` | `copy` applies the artifact allow-list and the `_conc` suffixes | +| `context.py` | `EvalContext` and `EvalOutcome`, the contract each framework's `run(ctx)` implements | -Key eval functions in `benchmarks/benchmark_lib.sh`: - -| Function | Description | -|----------|-------------| -| `run_eval` | Unified entrypoint - dispatches to framework-specific runner | -| `run_lm_eval` | Runs lm-eval harness against the OpenAI-compatible endpoint | -| `run_kimi_vendor_eval` | Selects and runs a pinned Kimi Vendor Verifier suite | -| `run_minimax_vendor_eval` | Selects the pinned MiniMax smoke or full diagnostic | -| `run_bfcl_eval` | Selects a pinned BFCL V4 smoke or model-quality suite | -| `append_lm_eval_summary` | Writes `meta_env.json` and stages eval artifacts in the workspace | -| `_install_lm_eval_deps` | Installs lm-eval dependencies | -| `_prepare_vendor_verifier_python` | Uses system Python 3.12+ or provisions an isolated pinned Python 3.12 runtime for provider verifiers | -| `_prepare_kimi_vendor_runtime` | Installs the pinned verifier dependencies in an isolated temp path | -| `_prepare_minimax_m3_full_runtime` | Downloads hash-verified stock MiniMax sources and installs their pinned dependencies for smoke and full suites | -| `_prepare_bfcl_runtime` | Installs the verified BFCL wheel in a temporary virtual environment | -| `_install_bfcl_eval_deps` | Downloads, verifies, and installs the pinned BFCL wheel | -| `_prepare_kimi_vendor_verifier` | Downloads, hash-verifies, and safely extracts a fresh subset of the pinned source archive | -| `_patch_lm_eval` | Patches lm-eval for reasoning tokens and TRT compatibility | -| `compute_eval_context_length` | Computes eval context length (requested benchmark context, capped at model native max) | -| `get_native_max_context_length` | Extracts model's native max context length from HF config | - -`EVAL_FRAMEWORK` is the orchestration-level selection and takes precedence over -legacy `--framework lm-eval` arguments embedded in fixed-sequence recipes. -Without that environment variable, an explicit `--framework` argument takes -precedence over the scenario default. +The exit code is the framework's (a child killed by signal N reports 128+N), else 1 +when metadata or staging failed. A batched run exits 0 and records failed +concurrencies in `meta_env.json`. An invalid environment input or flag value exits 1 +with an `ERROR:` line before anything is staged, and a malformed command line exits 2. + +`EVAL_FRAMEWORK` is the orchestration-level selection and takes precedence over a +`--framework` argument. Without that environment variable, `--framework` selects the +framework, and the default is `lm-eval`. ### Single-node -For default lm-eval jobs in eval-only mode (`EVAL_ONLY=true`), the benchmark script computes `EVAL_MAX_MODEL_LEN` via `compute_eval_context_length`, starts the server with that context length, skips throughput, and runs lm-eval. Each framework wires that context differently (`--context-length` for SGLang, `--max_seq_len` for TRT-LLM). +Single-node jobs run through srt-slurm recipes. For a fixed-sequence eval-only job, `runtime_arguments` in `infx/srt_slurm/single_node.py` starts the server with the matrix `MAX_MODEL_LEN` (`isl + osl + 256`) as its context (`context-length` for SGLang, `max_seq_len` and `max_num_tokens` for TRT-LLM, `max-model-len` for vLLM and ATOM), and srt-slurm skips the benchmark stage. AgentX evals keep the recipe context and receive `MAX_MODEL_LEN=0`. lm-eval sizes each request from `EVAL_MAX_MODEL_LEN` when set, otherwise from `MAX_MODEL_LEN` capped at the model's native maximum (`0` means the native maximum), and falls back to 16384 when neither is known. The shim writes the eval's exit code to `/logs/infx-eval-exit-code`, and the srt collector fails the job unless it is `0`. ### Multi-node -Multi-node evals on AMD and NVIDIA Slurm clusters run through [srt-slurm](https://github.com/NVIDIA/srt-slurm) at the shared Git submodule revision at `utils/srt-slurm`. Native `post_eval.command` and `post_eval.passthrough_env` select the InferenceX eval dispatcher without modifying the upstream checkout. +Multi-node evals on AMD and NVIDIA Slurm clusters run through [srt-slurm](https://github.com/NVIDIA/srt-slurm) at the shared Git submodule revision at `utils/srt-slurm`. Native `post_eval.command` and `post_eval.passthrough_env` select the InferenceX eval entrypoint without modifying the upstream checkout. - `do_sweep.py` skips the benchmark stage when `EVAL_ONLY=true`, runs `_run_post_eval()` directly - In eval-only mode, uses the full `wait_for_model()` health check (same as benchmark stage) since the benchmark health check was skipped -- Native `post_eval.command` invokes `benchmarks/multi_node/srt_eval.sh` from the mounted InferenceX workspace (`/infmax-workspace`). It sources `benchmark_lib.sh` and calls `run_eval`, which selects the eval implementation from `EVAL_FRAMEWORK` without patching upstream runners. +- InferenceX always sets `post_eval.command` to `benchmarks/multi_node/srt_eval.sh` (single-node jobs use `benchmarks/single_node/srt_eval.sh`), which runs `python3 -m infx.bench eval` from the mounted workspace (`/infmax-workspace`). srt-slurm falls back to its own registered `lm-eval` runner only when `post_eval.command` is unset, which InferenceX launches never do, and that runner expects a shell helper this repository no longer ships. `EVAL_FRAMEWORK` and `EVAL_SUITE` reach the eval through `post_eval.passthrough_env`, so vendor frameworks need no hook changes - Eval artifacts written to `/logs/eval_results/` inside the container, collected by `infx/launch/drivers/srt/collect.py` when `RUN_EVAL=true` or `EVAL_ONLY=true` - The srt driver always collects server logs for debugging but skips benchmark result collection when `EVAL_ONLY=true` - Env vars threaded: `RUN_EVAL`, `EVAL_ONLY`, `EVAL_FRAMEWORK`, `EVAL_SUITE`, `IS_MULTINODE`, `FRAMEWORK`, `PRECISION`, `MODEL_PREFIX`, `RUNNER_TYPE`, `RESULT_FILENAME`, `SPEC_DECODING`, `ISL`, `OSL`, `PREFILL_TP/EP/NUM_WORKERS/DP_ATTN`, `DECODE_TP/EP/NUM_WORKERS/DP_ATTN`, `MODEL_NAME`, `EVAL_CONC` -For multi-node `all-evals`, `EVAL_CONC` is a space-separated list. When it contains multiple values, `run_eval` runs those concurrency points sequentially against the same live engine, stages each result with a `_concN` filename suffix, and records expected/completed/failed points in `meta_env.json`. +For multi-node `all-evals`, `EVAL_CONC` is a space-separated list. When it contains multiple values, `python3 -m infx.bench eval` runs those concurrency points sequentially against the same live engine, stages each result with a `_concN` filename suffix, and records expected/completed/failed points in `meta_env.json`. ### Workflow structure - `e2e-tests.yml`: `test-sweep-evals` (single-node fixed-seq-len), `test-sweep-multi-node-evals` @@ -630,21 +648,27 @@ attempt cannot replace a newer failed retry. ### Adding a provider verifier 1. Add a provider-specific adapter under `infx/evals/`. -2. Add an explicit framework case in `run_eval`; keep suite-specific policy in - that adapter's shell runner. -3. Install dependencies in a provider-specific isolated runtime. -4. Emit `result_format: inferencex-eval-v1`, preserve the native report in an - explicitly uploaded suite-specific path, set `EVAL_SUITE`, and add a threshold. +2. Add a `Provider` entry, with its `Suite` specs and `prepare` hook, to `PROVIDERS` + in `infx/bench/eval/vendor.py`. `infx.bench.eval.FRAMEWORKS` registers it with + the eval command. Keep suite-specific request and report policy in the adapter. +3. Install dependencies in a provider-specific isolated runtime from the `prepare` + hook. +4. Emit `result_format: inferencex-eval-v1`, name the native report so it matches the + staging allow-list in `infx/bench/eval/stage.py` and the workflow upload paths, + set `EVAL_SUITE`, and add a threshold. +5. Keep the adapter's integration-error path stdlib-only and Python 3.10 + compatible. When provisioning fails, the runner invokes it under the image's + `python3`. ### Runtime patches (`infx/evals/patches/`) -The benchmark helpers invoke these standalone scripts against pinned dependencies. -Source rewrites are anchor-checked, idempotent, and atomic. +`infx.bench.eval.lm_eval` applies this standalone patch to the pinned lm-eval. -- `lm_eval_sitecustomize.py` (`_patch_lm_eval`): reasoning-token handling +- `lm_eval_sitecustomize.py`: reasoning-token handling (extracts `reasoning_content` when `message.content` is empty) and TRT compatibility (no `{"type": "text"}` injection for non-HF tokenizers). - Copied into a temp dir as `sitecustomize.py` on `PYTHONPATH`. + Each lm-eval run copies it into a temp dir as `sitecustomize.py` on `PYTHONPATH`. + ## Task files The following files are task definitions from lm-eval. More information on changes lives within the files: - `infx/evals/gsm8k.yaml` diff --git a/inferencex-e2e/infx/evals/_kimi_verifier_archive.py b/inferencex-e2e/infx/evals/_kimi_verifier_archive.py index 052aafa66f..29973d952e 100644 --- a/inferencex-e2e/infx/evals/_kimi_verifier_archive.py +++ b/inferencex-e2e/infx/evals/_kimi_verifier_archive.py @@ -1,4 +1,4 @@ -"""Internal pinned Kimi verifier archive preparation for benchmark_lib.sh.""" +"""Fetch the pinned Kimi Vendor Verifier archive subset (run by ``infx.bench.eval.vendor``).""" import re import sys diff --git a/inferencex-e2e/infx/evals/bfcl_adapter.py b/inferencex-e2e/infx/evals/bfcl_adapter.py index 21333575ca..c281eae1fb 100644 --- a/inferencex-e2e/infx/evals/bfcl_adapter.py +++ b/inferencex-e2e/infx/evals/bfcl_adapter.py @@ -3,12 +3,15 @@ from __future__ import annotations import argparse +import hashlib import inspect import json import math import os +import subprocess import sys import urllib.parse +import urllib.request from collections.abc import Callable, Mapping, Sequence from dataclasses import dataclass from importlib.metadata import distribution @@ -31,6 +34,13 @@ BFCL_PACKAGE = "bfcl-eval" BFCL_PACKAGE_VERSION = "2026.3.23" BFCL_WHEEL_SHA256 = "3bb6dfa5f0c68ad403c9ec50b00db2bb3b4cc9b38ab1ff33f48fe30d853d3a0a" +BFCL_WHEEL_URL = ( + "https://files.pythonhosted.org/packages/ba/41/" + "ed458527c770c50225b60bae3b0c3444b26804ee455fa2d8f187018d2cb2/" + "bfcl_eval-2026.3.23-py3-none-any.whl" +) +BFCL_WHEEL_MAX_BYTES = 512 * 1024 * 1024 +RUNTIME_REQUIREMENTS = ("soundfile==0.13.1",) UPSTREAM_REPOSITORY = "https://github.com/ShishirPatil/gorilla" UPSTREAM_SOURCE = "https://pypi.org/project/bfcl-eval/2026.3.23/" UPSTREAM_REF = f"{BFCL_PACKAGE}=={BFCL_PACKAGE_VERSION}" @@ -907,12 +917,64 @@ def run_evaluation( return True +def install_runtime(download_dir: Path) -> None: + """Install the pinned BFCL wheel into this interpreter, refusing any other bytes.""" + download_dir.mkdir(parents=True, exist_ok=True) + wheel_path = download_dir / BFCL_WHEEL_URL.rsplit("/", 1)[1] + request = urllib.request.Request( + BFCL_WHEEL_URL, + headers={"User-Agent": "InferenceX-BFCL-Smoke"}, + ) + digest = hashlib.sha256() + downloaded = 0 + try: + with ( + urllib.request.urlopen(request, timeout=180) as response, # noqa: S310 + wheel_path.open("xb") as output, + ): + while chunk := response.read(1024 * 1024): + downloaded += len(chunk) + if downloaded > BFCL_WHEEL_MAX_BYTES: + raise ValueError("BFCL wheel exceeds the 512 MiB safety limit") + digest.update(chunk) + output.write(chunk) + if downloaded == 0: + raise ValueError("downloaded BFCL wheel is empty") + if digest.hexdigest() != BFCL_WHEEL_SHA256: + raise ValueError( + f"BFCL wheel SHA256 mismatch: expected {BFCL_WHEEL_SHA256}, " + f"got {digest.hexdigest()}" + ) + except BaseException: + wheel_path.unlink(missing_ok=True) + raise + subprocess.run( + [ + sys.executable, + "-m", + "pip", + "install", + "-q", + "--no-cache-dir", + str(wheel_path), + *RUNTIME_REQUIREMENTS, + ], + check=True, + ) + + def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: parser = argparse.ArgumentParser(description="Run a pinned BFCL V4 OpenAI completions suite.") + parser.add_argument( + "--install-runtime", + type=Path, + metavar="DOWNLOAD_DIR", + help="Install the verified pinned BFCL wheel into this interpreter, then exit.", + ) parser.add_argument("--base-url", type=_absolute_http_url) parser.add_argument("--api-key", type=_nonempty_string, default="EMPTY") - parser.add_argument("--model", type=_nonempty_string, required=True) - parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--model", type=_nonempty_string) + parser.add_argument("--output-dir", type=Path) parser.add_argument("--bfcl-project-root", type=Path) parser.add_argument( "--suite", @@ -922,21 +984,30 @@ def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: parser.add_argument("--num-threads", type=_positive_int) parser.add_argument("--integration-error") args = parser.parse_args(argv) + if args.install_runtime is None: + required = {"--model": args.model, "--output-dir": args.output_dir} + if args.integration_error is None: + required |= {"--base-url": args.base_url, "--bfcl-project-root": args.bfcl_project_root} + missing = [option for option, value in required.items() if value is None] + if missing: + parser.error(f"the following arguments are required: {', '.join(missing)}") args.num_threads = ( SUITE_SPECS[args.suite].default_num_threads if args.num_threads is None else args.num_threads ) - if args.integration_error is None: - if args.base_url is None: - parser.error("--base-url required unless --integration-error is provided") - if args.bfcl_project_root is None: - parser.error("--bfcl-project-root required unless --integration-error is provided") return args def main(argv: Sequence[str] | None = None) -> int: args = parse_args(argv) + if args.install_runtime is not None: + try: + install_runtime(args.install_runtime) + except (OSError, ValueError, subprocess.CalledProcessError) as error: + print(f"ERROR: failed to install the pinned BFCL runtime: {error}", file=sys.stderr) + return 1 + return 0 suite = SUITE_SPECS[args.suite] if args.integration_error is not None: publish_integration_error( diff --git a/inferencex-e2e/infx/evals/kimi_vendor_eval.py b/inferencex-e2e/infx/evals/kimi_vendor_eval.py index ca7a1a3f92..7085d40070 100755 --- a/inferencex-e2e/infx/evals/kimi_vendor_eval.py +++ b/inferencex-e2e/infx/evals/kimi_vendor_eval.py @@ -9,10 +9,13 @@ import subprocess import sys from collections.abc import Mapping, Sequence -from datetime import UTC, datetime +from datetime import datetime, timezone from pathlib import Path from typing import Any +# The integration-error path runs under the serving image's python3, which may be 3.10. +UTC = timezone.utc # noqa: UP017 + TASK_NAME = "kimi_tool_call_schema" FULL_TASK_NAME = "kimi_tool_call_schema_full" SUPPORTED_TASK_NAMES = (TASK_NAME, FULL_TASK_NAME) diff --git a/inferencex-e2e/infx/evals/minimax_m3_full_eval.py b/inferencex-e2e/infx/evals/minimax_m3_full_eval.py index 274425be55..9cca301dcb 100755 --- a/inferencex-e2e/infx/evals/minimax_m3_full_eval.py +++ b/inferencex-e2e/infx/evals/minimax_m3_full_eval.py @@ -14,10 +14,13 @@ import urllib.parse import urllib.request from collections.abc import Callable, Mapping, Sequence -from datetime import UTC, datetime +from datetime import datetime, timezone from pathlib import Path, PurePosixPath from typing import Any +# The failure path runs under the serving image's python3, which may be 3.10. +UTC = timezone.utc # noqa: UP017 + TASK_NAME = "minimax_m3_full" RESULT_FORMAT = "inferencex-eval-v1" ADAPTER_NAME = "minimax-provider-verifier" diff --git a/inferencex-e2e/infx/evals/minimax_provider_eval.py b/inferencex-e2e/infx/evals/minimax_provider_eval.py index e983746e88..2b62a31b5f 100755 --- a/inferencex-e2e/infx/evals/minimax_provider_eval.py +++ b/inferencex-e2e/infx/evals/minimax_provider_eval.py @@ -11,7 +11,7 @@ import subprocess import urllib.parse from collections.abc import Callable, Mapping, Sequence -from datetime import UTC, datetime +from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -20,6 +20,9 @@ else: from minimax_m3_full_eval import UPSTREAM_REF, verify_source_tree +# The failure path runs under the serving image's python3, which may be 3.10. +UTC = timezone.utc # noqa: UP017 + TASK_NAME = "minimax_m3_smoke" NATIVE_REPORT_FILENAME = "minimax_vendor_report.json" NATIVE_RESULTS_FILENAME = "minimax_vendor_results.jsonl" diff --git a/inferencex-e2e/infx/launch/drivers/srt/collect.py b/inferencex-e2e/infx/launch/drivers/srt/collect.py index 0324f4e775..8e8771ce4e 100644 --- a/inferencex-e2e/infx/launch/drivers/srt/collect.py +++ b/inferencex-e2e/infx/launch/drivers/srt/collect.py @@ -14,7 +14,8 @@ from pathlib import Path from typing import TYPE_CHECKING -from infx.launch import proc +from infx.bench.env import InputError +from infx.bench.eval import meta as eval_meta from infx.launch.artifacts import ( ArtifactError, bundle_server_logs, @@ -28,7 +29,6 @@ from infx.launch.backends.base import BackendError from infx.launch.drivers.srt.config import EXPORTER_PROVENANCE from infx.launch.drivers.srt.run import SrtRun, require -from infx.launch.request import RequestError if TYPE_CHECKING: from infx.launch.backends.base import Job @@ -186,18 +186,22 @@ def stage_logs_on_exit() -> None: def _write_eval_meta(run: SrtRun) -> int: - """Regenerate meta_env.json with the canonical writer in benchmark_lib.sh.""" - conc = run.request.eval_conc - if conc is None: - raise RequestError.missing("EVAL_CONC") - argv = [ - "bash", "-c", 'source "$1"; _write_lm_eval_meta_json "$2" "" "$3"', "bash", - str(run.workspace / "benchmarks/benchmark_lib.sh"), str(run.workspace / "meta_env.json"), conc, - ] # fmt: skip - rc = proc.run(argv, env={**run.env, "IS_MULTINODE": "true"}).returncode - if rc == 0: - print(f"Wrote meta_env.json (conc={conc}, prefix={run.request.model_prefix})") - return rc + """Refresh the staged meta_env.json's identity and topology from the workflow inputs. + + The eval container does not receive every workflow input (e.g. RECIPE_FINGERPRINT). + The suite, concurrency and batch manifest the eval recorded are kept. + """ + path = run.workspace / "meta_env.json" + if not path.is_file(): + print(f"WARNING: no staged eval metadata to refresh at {path}", file=sys.stderr) + return 0 + try: + eval_meta.refresh(path, {**run.env, "IS_MULTINODE": "true"}) + except (OSError, ValueError, KeyError, InputError) as error: + print(f"ERROR: failed to refresh {path}: {error}", file=sys.stderr) + return 1 + print(f"Refreshed meta_env.json (prefix={run.request.model_prefix})") + return 0 def cleanup_outputs(root: Path, *, sleep: Callable[[float], None] = time.sleep) -> None: diff --git a/inferencex-e2e/infx/launch/request.py b/inferencex-e2e/infx/launch/request.py index f293892211..a601827032 100644 --- a/inferencex-e2e/infx/launch/request.py +++ b/inferencex-e2e/infx/launch/request.py @@ -75,6 +75,7 @@ class LaunchRequest(BaseModel): bench_script_override: str | None = Field(None, alias="BENCH_SCRIPT_OVERRIDE") batch_reentry: OneFlag = Field(False, alias=BATCH_REENTRY_ENV) conc: int | None = Field(None, alias="CONC") + conc_list: IntList = Field(default_factory=list, alias="CONC_LIST") run_eval: TrueFlag = Field(False, alias="RUN_EVAL") eval_only: TrueFlag = Field(False, alias="EVAL_ONLY") salloc_time_limit: int | None = Field(None, alias="SALLOC_TIME_LIMIT") @@ -82,6 +83,25 @@ class LaunchRequest(BaseModel): env: dict[str, str] = Field(default_factory=dict, exclude=True, repr=False) + @model_validator(mode="after") + def _one_agentx_concurrency(self) -> Self: + """A multi-node AgentX throughput point owns a fresh server for its one concurrency.""" + if not self.is_multinode or not self.is_agentic or self.eval_only: + return self + unset = {"CONC": self.conc is None, "CONC_LIST": not self.conc_list} + if missing := [name for name, is_unset in unset.items() if is_unset]: + raise ValueError( + "required environment variables are not set for AgentX throughput: " + + ", ".join(missing) + ) + if self.conc < 1 or self.conc_list != [self.conc]: + raise ValueError( + "AgentX requires exactly one positive concurrency per server deployment; launch a " + f"fresh server for each concurrency (CONC={self.conc}, " + f"CONC_LIST={' '.join(map(str, self.conc_list))})" + ) + return self + @classmethod def from_env(cls, env: Mapping[str, str] | None = None) -> Self: """Parse ``env`` (default ``os.environ``); raise ``RequestError`` naming bad inputs.""" @@ -114,10 +134,8 @@ class SrtRequest(LaunchRequest): run_eval: TrueFlag = Field(alias="RUN_EVAL") eval_only: TrueFlag = Field(alias="EVAL_ONLY") thinking_mode: str | None = Field(None, alias="THINKING_MODE") - conc_list: IntList = Field(default_factory=list, alias="CONC_LIST") require_power: PowerFlag = Field(False, alias="REQUIRE_POWER") inferencex_results_python: str | None = Field(None, alias="INFERENCEX_RESULTS_PYTHON") - eval_conc: str | None = Field(None, alias="EVAL_CONC") @model_validator(mode="after") def _golden_curve_key(self) -> Self: diff --git a/inferencex-e2e/infx/results/power/single_node.py b/inferencex-e2e/infx/results/power/single_node.py index ab4dd9992a..ab29182b96 100644 --- a/inferencex-e2e/infx/results/power/single_node.py +++ b/inferencex-e2e/infx/results/power/single_node.py @@ -839,7 +839,7 @@ def main() -> int: "--csv", type=Path, default=Path("/workspace/gpu_metrics.csv"), - help="Path to gpu_metrics.csv from start_gpu_monitor (default: /workspace/gpu_metrics.csv)", + help="Path to gpu_metrics.csv from infx.bench.gpu_monitor (default: /workspace/gpu_metrics.csv)", ) parser.add_argument( "--bench-result", diff --git a/inferencex-e2e/infx/ruff.toml b/inferencex-e2e/infx/ruff.toml index 11d2196602..21da12e5aa 100644 --- a/inferencex-e2e/infx/ruff.toml +++ b/inferencex-e2e/infx/ruff.toml @@ -1,4 +1,6 @@ target-version = "py312" +# Container-side clients run under serving-image python3 (3.10 on ATOM/ROCm). +per-file-target-version = { "bench/**" = "py310" } line-length = 100 # Tests follow the repository test style, not the package ruleset. extend-exclude = ["tests"] diff --git a/inferencex-e2e/infx/tests/bench/conftest.py b/inferencex-e2e/infx/tests/bench/conftest.py new file mode 100644 index 0000000000..9aca682555 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/conftest.py @@ -0,0 +1,48 @@ +"""Fixtures shared by the ``infx.bench`` tests.""" + +from __future__ import annotations + +import json +import threading +from collections.abc import Callable, Iterator +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +import pytest + +Respond = Callable[[str, str], tuple[int, object]] + + +@pytest.fixture +def http_server() -> Iterator[Callable[[Respond], str]]: + """Start local servers answering with ``respond(method, path) -> (status, json_body)``.""" + started: list[ThreadingHTTPServer] = [] + + def start(respond: Respond) -> str: + class Handler(BaseHTTPRequestHandler): + def do_GET(self) -> None: + self._reply(*respond("GET", self.path)) + + def do_POST(self) -> None: + self.rfile.read(int(self.headers.get("Content-Length", 0))) + self._reply(*respond("POST", self.path)) + + def _reply(self, status: int, body: object) -> None: + payload = json.dumps(body).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *_: object) -> None: + pass + + httpd = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=httpd.serve_forever, args=(0.01,), daemon=True).start() + started.append(httpd) + return f"http://127.0.0.1:{httpd.server_port}" + + yield start + for httpd in started: + httpd.shutdown() + httpd.server_close() diff --git a/inferencex-e2e/infx/tests/bench/stubs.py b/inferencex-e2e/infx/tests/bench/stubs.py new file mode 100644 index 0000000000..0c769d9802 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/stubs.py @@ -0,0 +1,12 @@ +"""Stub executables for the ``infx.bench`` tests.""" + +from __future__ import annotations + +from pathlib import Path + + +def executable(path: Path, text: str) -> Path: + """Write ``text`` to ``path`` as an executable script; return ``path``.""" + path.write_text(text) + path.chmod(0o755) + return path diff --git a/inferencex-e2e/infx/tests/bench/test_agentic_command.py b/inferencex-e2e/infx/tests/bench/test_agentic_command.py new file mode 100644 index 0000000000..e40547b4a1 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_agentic_command.py @@ -0,0 +1,432 @@ +"""The ``agentic`` command: point validation, each power mode's steps and status, and the shim.""" + +from __future__ import annotations + +import contextlib +import json +import os +import re +import shlex +import signal +import subprocess +import sys +import time +from pathlib import Path + +import pytest + +from infx.bench.agentic.run import Plan, execute +from infx.bench.agentic.venv import Runtime +from infx.bench.env import BenchError, InputError +from infx.tests.bench.stubs import executable + +REPO_ROOT = Path(__file__).resolve().parents[3] +SHIM = REPO_ROOT / "benchmarks" / "srt_agentic.sh" + +POINT = { + "RESULT_FILENAME": "agentx", + "EVAL_ONLY": "false", + "IS_MULTINODE": "false", + "ENABLE_AGENTX_POWER": "0", + "REQUIRE_POWER": "0", + "TP": "3", + "PP_SIZE": "2", + "PCP_SIZE": "2", + "KV_OFFLOADING": "none", + "PRECISION": "fp4", + "AIPERF_PYTHON_VERSION": "3.11", + "AIPERF_FAILED_REQUEST_THRESHOLD": "0.10", + "MODEL": "test/model", + "MODEL_PREFIX": "test", + "FRAMEWORK": "vllm", + "CONC": "8", + "DURATION": "3600", + "PORT": "8000", + "AIPERF_LIVE_FAILED_REQUEST_THRESHOLD": "0.10", + "AIPERF_TRACE_IDLE_GAP_CAP_SECONDS": "300", + "AGENTIC_WARMUP_GRACE_PERIOD": "1800", + "AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS": "3600", + "AIPERF_EXPERIMENTAL_FAST": "0", + "AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID": "false", + "AIPERF_UNSAFE_OVERRIDE": "false", + "AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING": "1", + "AIPERF_WARMUP_REQUESTS_PER_LANE": "10", +} +# The runtime's python: each result step logs itself and exits with its *_RC. +FAKE_PYTHON = r"""#!/bin/bash +case "$2" in + infx.results.agentic.process_agentic_result) + echo "aggregate $RESULT_FILENAME" >> "$EVENTS"; exit "${AGGREGATE_RC:-0}" ;; + infx.results.agentic.power_adapter) + echo "adapter ${*:3}" >> "$EVENTS"; exit "${POWER_RC:-0}" ;; + infx.results.agentic.validate_agentic_result) + echo validate >> "$EVENTS"; exit "${VALIDATE_RC:-0}" ;; + infx.results.agentic.analyze_benchmark_distributions) echo analyze >> "$EVENTS" ;; + infx.results.generate_aiperf_plots) echo plots >> "$EVENTS" ;; + *) echo "unexpected $*" >> "$EVENTS"; exit 99 ;; +esac +""" +FAKE_AIPERF = r"""#!/bin/sh +echo replay >> "$EVENTS" +sleep "${REPLAY_SECONDS:-0}" +exit "${REPLAY_RC:-0}" +""" +FAKE_HF = '#!/bin/sh\nexit "${HF_RC:-0}"\n' +# The GPU monitor's identity and final-sample queries log themselves; its 1 s stream runs +# until the monitor stops it. +FAKE_NVIDIA_SMI = r"""#!/bin/sh +case " $* " in + *" -l 1 "*) exec sleep 60 ;; + *noheader*) echo gpu-final-sample >> "$EVENTS" ;; + *) echo gpu-identity >> "$EVENTS" ;; +esac +""" +DRIVER = """ +import os, sys +from pathlib import Path +from infx.bench.agentic.run import Plan, execute +from infx.bench.agentic.venv import Runtime +sys.exit(execute(Plan.from_env(os.environ), Runtime(Path(sys.argv[1])), os.environ)) +""" + + +@pytest.fixture(autouse=True) +def fake_gpu(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """The in-process GPU monitor finds ``nvidia-smi`` through this process's environment.""" + tools = tmp_path / "gpu" + tools.mkdir() + executable(tools / "nvidia-smi", FAKE_NVIDIA_SMI) + monkeypatch.setenv("PATH", f"{tools}{os.pathsep}{os.environ['PATH']}") + monkeypatch.setenv("EVENTS", str(tmp_path / "events.log")) + + +def _point(tmp_path: Path, **overrides: str | None) -> dict[str, str]: + env = { + **POINT, + "PATH": os.environ["PATH"], + "EVENTS": str(tmp_path / "events.log"), + "REAL_PYTHON": sys.executable, + "RESULT_DIR": str(tmp_path / "results"), + "AGENTIC_OUTPUT_DIR": str(tmp_path), + **overrides, + } + return {name: value for name, value in env.items() if value is not None} + + +def _runtime(tmp_path: Path) -> Runtime: + runtime = Runtime(tmp_path / "runtime") + runtime.python.parent.mkdir(parents=True) + stubs = {runtime.python: FAKE_PYTHON, runtime.aiperf: FAKE_AIPERF, runtime.hf: FAKE_HF} + for path, text in stubs.items(): + executable(path, text) + return runtime + + +def _run(tmp_path: Path, **overrides: str | None) -> int: + env = _point(tmp_path, **overrides) + return execute(Plan.from_env(env), _runtime(tmp_path), env) + + +def _events(tmp_path: Path) -> list[str]: + events = tmp_path / "events.log" + return events.read_text().splitlines() if events.exists() else [] + + +@pytest.mark.parametrize( + ("overrides", "message"), + [ + ( + {"RESULT_FILENAME": None, "AIPERF_WARMUP_REQUESTS_PER_LANE": None}, + " - RESULT_FILENAME\n - AIPERF_WARMUP_REQUESTS_PER_LANE", + ), + ({"KV_OFFLOAD_BACKEND": "lmcache"}, "KV_OFFLOAD_BACKEND must be empty"), + ({"KV_OFFLOADING": "dram", "TOTAL_CPU_DRAM_GB": "2400"}, "KV_OFFLOAD_BACKEND is required"), + ( + {"KV_OFFLOADING": "dram", "KV_OFFLOAD_BACKEND": "none", "TOTAL_CPU_DRAM_GB": "2400"}, + "KV_OFFLOAD_BACKEND is required", + ), + ( + {"KV_OFFLOADING": "dram", "KV_OFFLOAD_BACKEND": "lmcache", "TOTAL_CPU_DRAM_GB": "0"}, + "TOTAL_CPU_DRAM_GB must be a positive integer", + ), + ({"KV_OFFLOADING": "cpu"}, "unsupported KV_OFFLOADING value 'cpu'"), + ({"CONC_LIST": "4 8"}, "CONC_LIST='4 8' must equal CONC='8'"), + ({"CONC_LIST": ""}, "CONC_LIST='' must equal CONC='8'"), + ({"CONC": "4 8", "CONC_LIST": "4 8"}, "CONC must be a positive integer"), + ({"IS_MULTINODE": "1"}, "IS_MULTINODE must be true or false"), + ({"EVAL_ONLY": "true"}, " - EVAL_ENDPOINT_READY_TIMEOUT_SECONDS"), + ( + {"ENABLE_AGENTX_POWER": "1", "PP_SIZE": None, "PCP_SIZE": None}, + " - PP_SIZE\n - PCP_SIZE", + ), + ], +) +def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, message): + with pytest.raises(InputError, match=re.escape(message)): + Plan.from_env(_point(tmp_path, **overrides)) + + +WINDOW = {"IS_MULTINODE": "true", "ENABLE_AGENTX_POWER": "1", "SRT_MEASUREMENT_WINDOW_DIR": "/w"} +MARK = "adapter --result-dir {results}/conc_8 --concurrency 8 --write-multinode-window" +OFFSET = "agentic_power_timezone_offset.txt" +REPLAYED = {"benchmark.log", "benchmark_command.txt"} + + +@pytest.mark.parametrize( + ("overrides", "rc", "events", "files"), + [ + pytest.param( + {}, + 0, + ["replay", "aggregate agentx", "plots", "analyze", "validate"], + REPLAYED, + id="power-off", + ), + pytest.param( + # Multi-node recipes that pin IS_MULTINODE=false still run under the multi-node + # workflow, which collects ${RESULT_FILENAME}_conc*.json. + {"CONC_LIST": "8", "KV_OFFLOADING": "dram", "KV_OFFLOAD_BACKEND": "lmcache", + "TOTAL_CPU_DRAM_GB": "2400"}, + 0, + ["replay", "aggregate agentx_conc8", "plots", "analyze", "validate"], + {f"conc_8/{name}" for name in REPLAYED}, + id="conc-list-point", + ), + pytest.param( + {"ENABLE_AGENTX_POWER": "1", "REQUIRE_POWER": "1"}, + 0, + [ + "gpu-identity", "replay", "gpu-final-sample", "aggregate agentx", "plots", + "adapter --result-dir {results} --agg-result {out}/agentx.json" + " --expected-num-gpus 12 --require-power", + "analyze", "validate", + ], + {OFFSET, *REPLAYED, "gpu_metrics.csv", "gpu_metrics_identity.csv"}, + id="single-node-monitor", + ), + pytest.param( + WINDOW, + 0, + [f"{MARK} running", "replay", "aggregate agentx_conc8", "plots", f"{MARK} completed", + "analyze", "validate"], + {f"conc_8/{name}" for name in (OFFSET, *REPLAYED)}, + id="multi-node-window", + ), + pytest.param( + {**WINDOW, "REPLAY_RC": "143"}, + 143, + [f"{MARK} running", "replay", "aggregate agentx_conc8", "plots", "analyze", "validate"], + {f"conc_8/{name}" for name in (OFFSET, *REPLAYED)}, + id="failed-replay-leaves-the-window-running", + ), + pytest.param( + {**WINDOW, "POWER_RC": "4"}, + 4, + [f"{MARK} running"], + {f"conc_8/{OFFSET}"}, + id="unpublished-window-skips-the-replay", + ), + pytest.param( + {"IS_MULTINODE": "true", "ENABLE_AGENTX_POWER": "1"}, + 0, + [ + "replay", "aggregate agentx_conc8", "plots", + "adapter --result-dir {results}/conc_8 --agg-result {out}/agentx_conc8.json" + " --multinode-contract-missing", + "analyze", "validate", + ], + {f"conc_8/{name}" for name in REPLAYED}, + id="multi-node-without-window", + ), + pytest.param({"HF_RC": "6"}, 6, [], set(), id="failed-trace-download"), + ], +) # fmt: skip +def test_point_shape_decides_the_steps_status_and_artifacts(tmp_path, overrides, rc, events, files): + assert _run(tmp_path, **overrides) == rc + + results = tmp_path / "results" + assert _events(tmp_path) == [event.format(results=results, out=tmp_path) for event in events] + written = {str(path.relative_to(results)) for path in results.rglob("*") if path.is_file()} + assert written == files + for offset in results.rglob(OFFSET): + assert re.fullmatch(r"[+-]\d{4}\n", offset.read_text()) + + +@pytest.mark.parametrize( + ("replay", "aggregate", "validate", "power", "expected"), + [ + ("7", "1", "3", "5", 7), + ("0", "2", "3", "5", 2), + ("0", "0", "3", "5", 3), + ("0", "0", "0", "5", 5), + ], +) +def test_every_step_runs_and_the_first_failure_in_precedence_wins( + tmp_path, replay, aggregate, validate, power, expected +): + rc = _run( + tmp_path, + ENABLE_AGENTX_POWER="1", + REPLAY_RC=replay, + AGGREGATE_RC=aggregate, + VALIDATE_RC=validate, + POWER_RC=power, + ) + + assert rc == expected + assert [event.split()[0] for event in _events(tmp_path)] == [ + "gpu-identity", "replay", "gpu-final-sample", "aggregate", "plots", "adapter", "analyze", + "validate", + ] # fmt: skip + + +@pytest.mark.parametrize( + ("csv", "prefix", "error"), + [ + (True, "vllm:", None), + (True, "sglang:", "has no metric with required prefix 'sglang:'"), + (False, "vllm:", "required AIPerf server metrics artifacts are missing or empty"), + ], +) +def test_required_server_metrics_gate_an_otherwise_clean_point(tmp_path, csv, prefix, error): + artifacts = tmp_path / "results" / "aiperf_artifacts" + artifacts.mkdir(parents=True) + # The quoted prefix straddles the scanner's first 1 MiB chunk. + exported = b" " * ((1 << 20) - 3) + b'{"vllm:num_requests_running": 1}' + (artifacts / "server_metrics_export.json").write_bytes(exported) + if csv: + (artifacts / "server_metrics_export.csv").write_text("metric,value\n") + + with pytest.raises(BenchError, match=re.escape(error)) if error else contextlib.nullcontext(): + assert _run(tmp_path, AIPERF_REQUIRED_SERVER_METRIC_PREFIX=prefix) == 0 + + +@pytest.mark.parametrize( + ("sent", "expected_rc"), [(signal.SIGINT, 130), (signal.SIGTERM, 143), (signal.SIGHUP, 129)] +) +def test_signal_during_the_replay_skips_scoring_and_exits_128_plus_n(tmp_path, sent, expected_rc): + runtime = _runtime(tmp_path) + env = _point(tmp_path, ENABLE_AGENTX_POWER="1", REPLAY_SECONDS="60", PYTHONPATH=str(REPO_ROOT)) + driver = subprocess.Popen( + [sys.executable, "-c", DRIVER, str(runtime.root)], + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + start_new_session=True, + ) + try: + deadline = time.monotonic() + 10 + while "replay" not in _events(tmp_path): + assert time.monotonic() < deadline, "the replay did not start" + time.sleep(0.01) + # The whole job gets the signal, like a terminal interrupt or scancel. + os.killpg(driver.pid, sent) + _, stderr = driver.communicate(timeout=10) + finally: + with contextlib.suppress(ProcessLookupError): + os.killpg(driver.pid, signal.SIGKILL) + driver.communicate() + + assert driver.returncode == expected_rc, stderr + # The signal may reach the monitor's sampler first, so its final sample is optional. + assert [e for e in _events(tmp_path) if e != "gpu-final-sample"] == ["gpu-identity", "replay"] + + +# The replay records its argv, writes one profiled request, and prints progress. +FAKE_AIPERF_SCRIPT = r""" +import json, os, pathlib, shlex, sys +argv = sys.argv[1:] +out = pathlib.Path(argv[argv.index("--output-artifact-dir") + 1]) +out.mkdir(parents=True, exist_ok=True) +pathlib.Path(os.environ["REPLAY_ARGV"]).write_text(shlex.join([sys.argv[0], *argv])) +with open(os.environ["EVENTS"], "a") as events: + events.write("replay\n") +record = { + "metadata": { + "conversation_id": "trace-A", "turn_index": 0, "benchmark_phase": "profiling", + "request_start_ns": 1_000_000_000, "request_ack_ns": 1_000_000_100, + "request_end_ns": 2_000_000_000, "was_cancelled": False, + }, + "metrics": { + "input_sequence_length": {"value": 100, "unit": "tokens"}, + "output_sequence_length": {"value": 50, "unit": "tokens"}, + "time_to_first_token": {"value": 30.0, "unit": "ms"}, + "request_latency": {"value": 1000.0, "unit": "ms"}, + "inter_token_latency": {"value": 18.0, "unit": "ms"}, + }, + "error": None, +} +(out / "profile_export.jsonl").write_text(json.dumps(record) + "\n") +(out / "profile_export_aiperf.json").write_text(json.dumps({ + "request_count": {"avg": 1}, "error_request_count": {"avg": 0}, + "completed_request_count": {"avg": 1}, +})) +print("fake aiperf: profiled 1 request", flush=True) +""" +# uv stand-in: `venv` makes a real (pip-less) venv; `pip install` adds AIPerf and hf stubs. +FAKE_UV = r"""#!/bin/sh +case "$1" in + venv) for target; do :; done; exec "$REAL_PYTHON" -m venv --without-pip "$target" ;; + pip) + bin="$(dirname "$4")" + printf '#!%s\n' "$bin/python" > "$bin/aiperf" + cat "$FAKE_AIPERF_SCRIPT" >> "$bin/aiperf" + printf '#!/bin/sh\necho "hf $*" >> "$EVENTS"\n' > "$bin/hf" + chmod +x "$bin/aiperf" "$bin/hf" ;; +esac +""" + + +def test_srt_shim_builds_the_runtime_waits_for_the_server_and_writes_the_point( + tmp_path, http_server +): + tools = tmp_path / "tools" + tools.mkdir() + executable(tools / "uv", FAKE_UV) + # The serving image's python3; it only needs the standard library. + (tools / "python3").symlink_to(sys.executable) + script = tmp_path / "fake_aiperf.py" + script.write_text(FAKE_AIPERF_SCRIPT) + + def frontend(method: str, path: str) -> tuple[int, object]: + with (tmp_path / "events.log").open("a") as events: + events.write(f"{method} {path}\n") + if path == "/v1/models": + return 200, {"data": [{"id": "served/model"}]} + return (405 if path == "/v1/chat/completions" else 200), {} + + env = _point( + tmp_path, + PATH=f"{tools}:/usr/bin:/bin", + HOME=str(tmp_path), + FAKE_AIPERF_SCRIPT=str(script), + REPLAY_ARGV=str(tmp_path / "replay-argv"), + HF_HUB_CACHE=str(tmp_path / "hf"), + AIPERF_RUNTIME_DIR=str(tmp_path / "runtime"), + EVAL_ONLY="true", + AIPERF_SERVER_URL=http_server(frontend), + SERVED_MODEL_NAME="served/model", + EVAL_ENDPOINT_READY_TIMEOUT_SECONDS="30", + EVAL_MODEL_STABILIZATION_SECONDS="0", + ) + + result = subprocess.run( + ["bash", str(SHIM)], env=env, capture_output=True, text=True, timeout=120, check=False + ) + + assert result.returncode == 0, result.stderr + assert _events(tmp_path) == [ + "hf download --repo-type dataset semianalysisai/cc-traces-weka-062126-256k", + "GET /v1/models", + "GET /v1/chat/completions", + "replay", + ] + results = tmp_path / "results" + command = (results / "benchmark_command.txt").read_text() + assert command == (tmp_path / "replay-argv").read_text() + "\n" + assert shlex.split(command)[0] == str(tmp_path / "runtime/venv/bin/aiperf") + assert "fake aiperf: profiled 1 request" in (results / "benchmark.log").read_text() + assert "fake aiperf: profiled 1 request" in result.stdout + aggregate = json.loads((tmp_path / "agentx.json").read_text()) + assert (aggregate["conc"], aggregate["num_requests_total"]) == (8, 1) diff --git a/inferencex-e2e/infx/tests/bench/test_agentic_replay.py b/inferencex-e2e/infx/tests/bench/test_agentic_replay.py new file mode 100644 index 0000000000..a5aa55542b --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_agentic_replay.py @@ -0,0 +1,196 @@ +"""AgentX replay configuration: environment rules, AIPerf argv, and trace corpus choice.""" + +from __future__ import annotations + +import re +from pathlib import Path + +import pytest + +from infx.bench.agentic.replay import ReplayConfig, replay_argv +from infx.bench.agentic.traces import resolve +from infx.bench.env import InputError + +BASE_ENV = { + "MODEL": "deepseek-ai/DeepSeek-V4-Pro", + "MODEL_PREFIX": "dsv4", + "FRAMEWORK": "vllm", + "CONC": "8", + "DURATION": "3600", + "PORT": "8000", + "AIPERF_LIVE_FAILED_REQUEST_THRESHOLD": "0.10", + "AIPERF_TRACE_IDLE_GAP_CAP_SECONDS": "300", + "AGENTIC_WARMUP_GRACE_PERIOD": "1800", + "AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS": "3600", + "AIPERF_EXPERIMENTAL_FAST": "0", + "AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID": "false", + "AIPERF_UNSAFE_OVERRIDE": "false", + "AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING": "1", + "AIPERF_WARMUP_REQUESTS_PER_LANE": "10", +} + + +def _argv(**overrides: str | None) -> list[str]: + env = {name: value for name, value in {**BASE_ENV, **overrides}.items() if value is not None} + return replay_argv(ReplayConfig.from_env(env, Path("/results")), "/venv/bin/aiperf") + + +def _value(argv: list[str], flag: str) -> str: + return argv[argv.index(flag) + 1] + + +def test_default_point_replays_the_uncapped_corpus_with_the_fixed_policy(): + # A workflow MAX_MODEL_LEN is not the server's limit, so it never caps the replay. + assert _argv(MAX_MODEL_LEN="131072") == [ + "/venv/bin/aiperf", "profile", "--scenario", "agentx", + "--url", "http://localhost:8000", "--endpoint", "/v1/chat/completions", + "--model", "deepseek-ai/DeepSeek-V4-Pro", "--tokenizer", "deepseek-ai/DeepSeek-V4-Pro", + "--concurrency", "8", "--benchmark-duration", "3600", + "--failed-request-threshold", "0.10", + "--warmup-requests-per-lane", "10", "--trace-idle-gap-cap-seconds", "300", + "--warmup-grace-period", "1800", "--tokenizer-trust-remote-code", + "--output-artifact-dir", "/results/aiperf_artifacts", + "--public-dataset", "semianalysis_cc_traces_weka_062126", + ] # fmt: skip + + +def test_opt_in_inputs_add_their_flags_in_place(): + argv = _argv( + SERVED_MODEL_NAME="DeepSeek-V4-Pro", + FRAMEWORK="dynamo-vllm", + AIPERF_DYNAMO_SESSION_TIMEOUT_SECONDS="14400", + AIPERF_EXTRA_INPUTS="temperature:0.7 top_p:0.9", + AIPERF_MAX_CONTEXT_LENGTH="262144", + AIPERF_SERVER_METRICS_URLS="http://a:1/metrics,http://b:2/metrics,", + AIPERF_UNSAFE_OVERRIDE="true", + WEKA_LOADER_OVERRIDE="semianalysis_cc_traces_weka_with_subagents_256k", + AIPERF_APPLY_CHAT_TEMPLATE="true", + AIPERF_BENCHMARK_GRACE_PERIOD="1800", + ) + + assert argv == [ + "/venv/bin/aiperf", "profile", "--scenario", "agentx", + "--url", "http://localhost:8000", "--endpoint", "/v1/chat/completions", + "--model", "DeepSeek-V4-Pro", "--tokenizer", "deepseek-ai/DeepSeek-V4-Pro", + "--concurrency", "8", "--benchmark-duration", "3600", + "--failed-request-threshold", "0.10", + "--warmup-requests-per-lane", "10", "--trace-idle-gap-cap-seconds", "300", + "--warmup-grace-period", "1800", + "--extra-inputs", "temperature:0.7", "top_p:0.9", + "--use-dynamo-conv-aware-routing", "--dynamo-session-timeout-seconds", "14400", + "--tokenizer-trust-remote-code", + "--max-context-length", "262144", + "--server-metrics", "http://a:1/metrics", "http://b:2/metrics", + "--output-artifact-dir", "/results/aiperf_artifacts", + "--unsafe-override", + "--public-dataset", "semianalysis_cc_traces_weka_with_subagents_256k", + "--apply-chat-template", "--benchmark-grace-period", "1800", + ] # fmt: skip + + +@pytest.mark.parametrize( + ("overrides", "duration", "warmup", "unsafe"), + [ + ({"DURATION": "899"}, "899", "10", True), + ({"DURATION": "900"}, "900", "10", False), + # agentx-fast's 20-minute profile meets the scenario minimum the caller's 300 s misses. + ({"AIPERF_EXPERIMENTAL_FAST": "1", "DURATION": "300"}, "1200", "1", False), + ], +) +def test_profile_length_decides_the_unsafe_override(overrides, duration, warmup, unsafe): + argv = _argv(**overrides) + + assert _value(argv, "--benchmark-duration") == duration + assert _value(argv, "--warmup-requests-per-lane") == warmup + assert ("--unsafe-override" in argv) is unsafe + + +@pytest.mark.parametrize(("conv_aware", "header_routing"), [("0", "false"), ("1", "true")]) +def test_dynamo_opt_outs_disable_conversation_aware_routing(conv_aware, header_routing): + argv = _argv( + FRAMEWORK="dynamo-sglang", + AIPERF_USE_DYNAMO_CONV_AWARE_ROUTING=conv_aware, + AIPERF_HTTP_X_DYNAMO_SESSION_ID_FROM_CORRELATION_ID=header_routing, + ) + + assert "--use-dynamo-conv-aware-routing" not in argv + assert "--dynamo-session-timeout-seconds" not in argv + + +@pytest.mark.parametrize( + ("overrides", "scraped"), + [ + ( + {"SRT_PREFILL_ENDPOINTS": "p:1", "SRT_DECODE_ENDPOINTS": "d:2,d:3"}, + ["--server-metrics", "http://p:1/metrics", "http://d:2/metrics", "http://d:3/metrics"], + ), + ( + {"SRT_AGG_ENDPOINTS": "w:9,", "SRT_DECODE_ENDPOINTS": "d:2"}, + ["--server-metrics", "http://w:9/metrics"], + ), + ( + {"AIPERF_SERVER_METRICS_URLS": "http://x/metrics", "SRT_AGG_ENDPOINTS": "w:9"}, + ["--server-metrics", "http://x/metrics"], + ), + ({"SRTCTL_FRONTEND_TYPE": "dynamo", "SRT_AGG_ENDPOINTS": "w:9"}, []), + ], +) +def test_server_metrics_come_from_the_caller_or_each_srt_worker(overrides, scraped): + argv = _argv(**overrides) + + assert argv[argv.index("--tokenizer-trust-remote-code") + 1 : argv.index("--output-artifact-dir")] == scraped + + +@pytest.mark.parametrize( + ("overrides", "url"), + [ + ( + {"SRT_FRONTEND_HOST": "10.0.0.5", "SRT_FRONTEND_PORT": "8000", + "AIPERF_SERVER_URL": "http://router:30000"}, + "http://10.0.0.5:8000", + ), + ({"AIPERF_SERVER_URL": "http://router:30000", "PORT": None}, "http://router:30000"), + ], +) # fmt: skip +def test_replay_targets_the_srt_frontend_then_the_explicit_url(overrides, url): + assert _value(_argv(**overrides), "--url") == url + + +@pytest.mark.parametrize( + ("overrides", "message"), + [ + ({"SRT_FRONTEND_HOST": "10.0.0.5"}, " - SRT_FRONTEND_PORT"), + ({"PORT": None}, " - PORT"), + ({"AIPERF_MAX_CONTEXT_LENGTH": "256k"}, "AIPERF_MAX_CONTEXT_LENGTH must be a positive"), + ({"AIPERF_SERVER_METRICS_URLS": "http://a/metrics,,http://b/metrics"}, "non-empty URLs"), + ({"AIPERF_SERVER_METRICS_URLS": "http://a/metrics, http://b/metrics"}, "non-empty URLs"), + ({"AIPERF_SERVER_METRICS_URLS": ","}, "non-empty URLs"), + ( + {"WEKA_LOADER_OVERRIDE": "semianalysis_cc_traces_weka_060226"}, + "unknown WEKA_LOADER_OVERRIDE='semianalysis_cc_traces_weka_060226'", + ), + ], +) +def test_malformed_replay_inputs_are_rejected(overrides, message): + with pytest.raises(InputError, match=re.escape(message)): + _argv(**overrides) + + +@pytest.mark.parametrize( + ("prefix", "loader", "dataset"), + [ + # dsv4 matches by prefix, so DeepSeek-V4-Flash replays the uncapped 1M corpus. + ( + "dsv41flash", + "semianalysis_cc_traces_weka_062126", + "semianalysisai/cc-traces-weka-062126", + ), + ( + "glm5.1", + "semianalysis_cc_traces_weka_062126_256k", + "semianalysisai/cc-traces-weka-062126-256k", + ), + ], +) +def test_default_corpus_follows_the_model_family_context(prefix, loader, dataset): + assert resolve(prefix, None) == (loader, dataset) diff --git a/inferencex-e2e/infx/tests/bench/test_agentic_runtime.py b/inferencex-e2e/infx/tests/bench/test_agentic_runtime.py new file mode 100644 index 0000000000..c964272fd0 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_agentic_runtime.py @@ -0,0 +1,106 @@ +"""The isolated AIPerf venv that the ``agentic`` command builds before re-executing.""" + +from __future__ import annotations + +import shutil +from pathlib import Path + +import pytest + +from infx.bench.agentic.venv import Runtime, bootstrap +from infx.tests.bench.stubs import executable + +FAKE_UV = r"""#!/bin/sh +printf '%s\n' "$@" >> "$UV_CALLS" +printf '%s\n' "$UV_CACHE_DIR" >> "$UV_CACHES" +if [ "$1" = venv ]; then + [ "$FAIL" = venv ] && exit 3 + for target; do :; done + mkdir -p "$target/bin" +elif [ "$1" = pip ]; then + [ "$FAIL" = pip ] && exit 2 + [ "$FAIL" = incomplete ] && exit 0 + bin="$(dirname "$4")" + printf '#!/bin/sh\nexit 0\n' > "$bin/aiperf" + printf '#!/bin/sh\nexit 0\n' > "$bin/hf" + chmod +x "$bin/aiperf" "$bin/hf" +fi +""" + + +@pytest.fixture +def tools(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + """A PATH with only coreutils plus tripwires for git and apt-get.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + for forbidden in ("git", "apt-get"): + executable(bin_dir / forbidden, f'#!/bin/sh\necho {forbidden} >> "$FORBIDDEN"\nexit 97\n') + monkeypatch.setenv("PATH", f"{bin_dir}:/usr/bin:/bin") + monkeypatch.setenv("UV_CALLS", str(tmp_path / "uv-calls")) + monkeypatch.setenv("UV_CACHES", str(tmp_path / "uv-caches")) + monkeypatch.setenv("FORBIDDEN", str(tmp_path / "forbidden")) + monkeypatch.delenv("FAIL", raising=False) + return bin_dir + + +@pytest.mark.parametrize("uv_on_path", [True, False]) +def test_bootstrap_builds_a_pinned_venv_with_editable_aiperf( + tmp_path: Path, tools: Path, uv_on_path: bool +): + uv = executable(tmp_path / "uv", FAKE_UV) + if uv_on_path: + shutil.copy(uv, tools / "uv") + else: + # Astral's installer, piped from curl to sh, drops uv into UV_INSTALL_DIR. + installer = tmp_path / "install.sh" + installer.write_text(f'mkdir -p "$UV_INSTALL_DIR" && cp {uv} "$UV_INSTALL_DIR/uv"\n') + executable(tools / "curl", f"#!/bin/sh\ncat {installer}\n") + runtime = Runtime(tmp_path / "runtime") + runtime.venv.mkdir(parents=True) + (runtime.venv / "stale").write_text("from an earlier point") + source = tmp_path / "checkout with spaces" / "utils" / "aiperf" + + assert bootstrap(runtime, "3.11", source) == 0 + + calls = (tmp_path / "uv-calls").read_text().splitlines() + venv = tmp_path / "runtime" / "venv" + assert calls[:4] == ["venv", "--python", "3.11", str(venv)] + install = ["pip", "install", "--python", str(venv / "bin/python"), "-e", str(source)] + assert calls[4:10] == install + assert not (venv / "stale").exists() + caches = set((tmp_path / "uv-caches").read_text().splitlines()) + assert caches == {str(tmp_path / "runtime" / "uv-cache")} + assert not (tmp_path / "forbidden").exists() + assert (tmp_path / "runtime/uv/bin/uv").is_file() == (not uv_on_path) + + +def test_failed_venv_creation_returns_uv_status_without_installing(tmp_path, tools, monkeypatch): + executable(tools / "uv", FAKE_UV) + monkeypatch.setenv("FAIL", "venv") + + assert bootstrap(Runtime(tmp_path / "runtime"), "3.11", tmp_path / "aiperf") == 3 + + assert "pip" not in (tmp_path / "uv-calls").read_text().splitlines() + + +@pytest.mark.parametrize( + ("failure", "message"), + [ + ("pip", "ERROR: benchmark client dependency bootstrap failed"), + ("incomplete", "ERROR: isolated AIPerf environment is incomplete"), + # No uv on PATH, and Astral's installer does not produce one. + ("installer", "ERROR: uv installation did not create"), + ], +) +def test_unusable_runtime_fails_with_the_reason( + tmp_path, tools, monkeypatch, capsys, failure, message +): + if failure == "installer": + executable(tools / "curl", "#!/bin/sh\nexit 6\n") + else: + executable(tools / "uv", FAKE_UV) + monkeypatch.setenv("FAIL", failure) + + assert bootstrap(Runtime(tmp_path / "runtime"), "3.11", tmp_path / "aiperf") == 1 + + assert message in capsys.readouterr().err diff --git a/inferencex-e2e/infx/tests/bench/test_eval_command.py b/inferencex-e2e/infx/tests/bench/test_eval_command.py new file mode 100644 index 0000000000..e81183204c --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_eval_command.py @@ -0,0 +1,418 @@ +"""The ``eval`` command and its srt-slurm shims against a fake OpenAI server, pip, and lm-eval.""" + +import json +import os +import shutil +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from infx.bench.env import InputError +from infx.bench.eval import FRAMEWORKS, evaluate, lm_eval +from infx.bench.eval.context import EvalOutcome +from infx.evals.validate_scores import validate_batch_manifest +from infx.tests.bench.stubs import executable + +REPO_ROOT = Path(__file__).resolve().parents[3] + +STUBS = { + "pip/__init__.py": "", + "pip/__main__.py": """ +import json, os, sys +with open(os.environ["STUB_PIP_TRACE"], "a") as trace: + trace.write(json.dumps(sys.argv[1:]) + "\\n") +sys.exit(1 if os.environ.get("STUB_PIP_FAIL", "\\0") in " ".join(sys.argv[1:]) else 0) +""", + "lm_eval/__init__.py": "", + "lm_eval/models/__init__.py": "", + "lm_eval/models/openai_completions.py": "class LocalChatCompletion:\n pass\n", + "lm_eval/models/api_models.py": "class JsonChatStr(str):\n pass\n\n\nclass TemplateAPI:\n pass\n", + "lm_eval/__main__.py": """ +import json, os, sys, urllib.request +from pathlib import Path +from lm_eval.models.api_models import TemplateAPI + +args = sys.argv[1:] +model_args = dict(item.split("=", 1) for item in args[args.index("--model_args") + 1].split(",")) +with open(os.environ["STUB_LM_EVAL_TRACE"], "a") as trace: + trace.write(json.dumps({ + "argv": args, + "tasks_found": Path(args[args.index("--tasks") + 1]).is_file(), + "patched": getattr(TemplateAPI, "apply_chat_template", None) is not None, + }) + "\\n") +if model_args["num_concurrent"] == os.environ.get("STUB_LM_EVAL_FAIL_AT"): + sys.exit(3) +body = json.dumps({"model": model_args["model"], "messages": [{"role": "user", "content": "6*7?"}]}) +request = urllib.request.Request( + model_args["base_url"], data=body.encode(), headers={"Content-Type": "application/json"} +) +with urllib.request.urlopen(request, timeout=10) as response: + answer = json.load(response)["choices"][0]["message"]["content"] +output = Path(args[args.index("--output_path") + 1]) / model_args["model"].replace("/", "__") +output.mkdir(parents=True) +(output / "results_stub.json").write_text(json.dumps({"answer": answer})) +(output / "samples_gsm8k_stub.jsonl").write_text(json.dumps({"answer": answer}) + "\\n") +(output / "lm_eval.log").write_text("not an artifact") +""", +} + +# The container's python3; its /logs mount lives under the test's tmp directory. +PYTHON3 = f"""#!/bin/sh +for arg; do + shift + case $arg in /logs/*) arg="$STUB_LOGS${{arg#/logs}}" ;; esac + set -- "$@" "$arg" +done +exec "{sys.executable}" "$@" +""" + +MULTI_NODE = { + "IS_MULTINODE": "true", "EVAL_MAX_MODEL_LEN": "4096", + "PREFILL_TP": "4", "PREFILL_EP": "1", "PREFILL_DP_ATTN": "false", + "DECODE_TP": "8", "DECODE_EP": "1", "DECODE_DP_ATTN": "false", "DECODE_NUM_WORKERS": "2", +} # fmt: skip + + +@pytest.fixture +def openai(http_server): + """Serves one model: listed under /v1/models, answering chat completions by POST only.""" + requests = [] + + def respond(method, path): + requests.append((method, path)) + if method == "POST": + return 200, {"choices": [{"index": 0, "message": {"content": "42"}}]} + if path == "/v1/models": + return 200, {"data": [{"id": "served-model"}]} + return (405 if path == "/v1/chat/completions" else 404), {} + + return SimpleNamespace(url=http_server(respond), requests=requests) + + +@pytest.fixture +def base_env(tmp_path): + """What every eval receives: the stubs first on PYTHONPATH, and no network model lookups.""" + stubs = tmp_path / "stubs" + for name, body in STUBS.items(): + (stubs / name).parent.mkdir(parents=True, exist_ok=True) + (stubs / name).write_text(body) + return { + "PATH": os.environ["PATH"], + "HOME": str(tmp_path), + "PYTHONPATH": str(stubs), + "HF_HUB_OFFLINE": "1", + "STUB_PIP_TRACE": str(tmp_path / "pip.jsonl"), + "STUB_LM_EVAL_TRACE": str(tmp_path / "lm_eval.jsonl"), + "OPENAI_API_KEY": "EMPTY", + "EVAL_ONLY": "false", + "MODEL": "org/test-model", + "MODEL_NAME": "served-model", + } + + +@pytest.fixture +def checkpoint(tmp_path): + """A checkpoint transformers cannot load, whose config.json states an 8192-token context.""" + model = tmp_path / "model" + model.mkdir() + (model / "config.json").write_text( + '{"model_type": "not_registered", "max_position_embeddings": 8192, "seq_length": 4096}' + ) + return model + + +@pytest.fixture +def checkout(tmp_path): + """A checkout whose benchmarks/ are copies of the real shims, so staging stays in tmp.""" + root = tmp_path / "checkout" + for script in ("check_env.sh", "single_node/srt_eval.sh", "multi_node/srt_eval.sh"): + (root / "benchmarks" / script).parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(REPO_ROOT / "benchmarks" / script, root / "benchmarks" / script) + (root / "infx").symlink_to(REPO_ROOT / "infx") + (tmp_path / "bin").mkdir() + executable(tmp_path / "bin/python3", PYTHON3) + return root + + +@pytest.fixture +def single_node_env(base_env, checkpoint): + return { + **base_env, "MODEL_PATH": str(checkpoint), "MAX_MODEL_LEN": "10240", "CONC": "8", + "TP": "8", "EP_SIZE": "8", "DP_ATTENTION": "true", "PP_SIZE": "1", "DCP_SIZE": "1", + "PCP_SIZE": "1", "IS_MULTINODE": "false", "IS_AGENTIC": "0", "FRAMEWORK": "sglang", + "PRECISION": "fp8", "SPEC_DECODING": "none", "MODEL_PREFIX": "dsr1", "RUNNER_TYPE": "h200", + "RECIPE_FINGERPRINT": "recipe-1", "ISL": "1024", "OSL": "1024", + } # fmt: skip + + +def shim(checkout: Path, node: str, env: dict, *args: str) -> subprocess.CompletedProcess: + """Run ``benchmarks/_node/srt_eval.sh`` with the container python3 first on PATH.""" + tmp = checkout.parent + return subprocess.run( + ["bash", str(checkout / f"benchmarks/{node}_node/srt_eval.sh"), *args], + env={**env, "PATH": f"{tmp / 'bin'}{os.pathsep}{env['PATH']}", "STUB_LOGS": str(tmp / "logs")}, + cwd=tmp, + capture_output=True, + text=True, + check=False, + ) + + +def trace(path: Path) -> list: + return [json.loads(line) for line in path.read_text().splitlines()] if path.exists() else [] + + +def model_args(run: dict) -> list[str]: + """The ``--model_args`` items of one recorded lm-eval invocation.""" + return run["argv"][run["argv"].index("--model_args") + 1].split(",") + + +def test_single_node_shim_stages_the_eval_in_the_checkout_and_records_its_status( + checkout, single_node_env, openai, tmp_path +): + status = tmp_path / "infx-eval-exit-code" + + result = shim(checkout, "single", single_node_env, openai.url, str(status)) + + assert result.returncode == 0, result.stdout + result.stderr + assert status.read_text() == "0\n" + staged = sorted(path.name for path in checkout.iterdir() if path.is_file()) + assert staged == ["meta_env.json", "results_stub.json", "samples_gsm8k_stub.jsonl"] + assert json.loads((checkout / "results_stub.json").read_text()) == {"answer": "42"} + assert openai.requests == [("POST", "/v1/chat/completions")] + [run] = trace(tmp_path / "lm_eval.jsonl") + assert run["tasks_found"] and run["patched"] + # MAX_MODEL_LEN 10240 is capped at the checkpoint's 8192; 4096 of it stays for the prompt. + assert {"model=served-model", "num_concurrent=8", "max_length=8192"} <= set(model_args(run)) + assert run["argv"][run["argv"].index("--gen_kwargs") + 1].startswith("max_tokens=4096,") + assert json.loads((checkout / "meta_env.json").read_text()) == { + "is_multinode": False, "framework": "sglang", "precision": "fp8", + "spec_decoding": "none", "eval_suite": "gsm8k", "recipe_fingerprint": "recipe-1", + "tp": 8, "pp": 1, "dcp_size": 1, "pcp_size": 1, "conc": 8, "ep": 8, "dp_attention": True, + "prefill_tp": 8, "prefill_pp": 1, "prefill_dcp_size": 1, "prefill_pcp_size": 1, + "prefill_ep": 8, "prefill_dp_attention": True, "prefill_num_workers": 1, + "decode_tp": 8, "decode_pp": 1, "decode_dcp_size": 1, "decode_pcp_size": 1, + "decode_ep": 8, "decode_dp_attention": True, "decode_num_workers": 1, + "model": "served-model", "infmax_model_prefix": "dsr1", "hw": "h200", + "isl": "1024", "osl": "1024", + } # fmt: skip + + +def test_multi_node_shim_stages_a_batched_eval_in_the_logs_mount( + checkout, base_env, openai, tmp_path +): + env = {**base_env, **MULTI_NODE, "EVAL_CONC": "1 2"} + + result = shim(checkout, "multi", env, openai.url, str(checkout)) + + assert result.returncode == 0, result.stdout + result.stderr + staged = tmp_path / "logs/eval_results" + assert sorted(path.name for path in staged.iterdir()) == [ + "meta_env.json", + "results_stub_conc1.json", + "results_stub_conc2.json", + "samples_gsm8k_stub_conc1.jsonl", + "samples_gsm8k_stub_conc2.jsonl", + ] + meta = json.loads((staged / "meta_env.json").read_text()) + assert {key: meta[key] for key in ("is_multinode", "conc", "eval_concs", "completed_eval_concs", "failed_eval_concs")} == { + "is_multinode": True, "conc": 1, "eval_concs": [1, 2], "completed_eval_concs": [1, 2], "failed_eval_concs": [], + } # fmt: skip + + +@pytest.mark.parametrize(("node", "inputs", "missing"), [ + ("single", {"MAX_MODEL_LEN": ""}, ["MAX_MODEL_LEN"]), + ("multi", {"IS_MULTINODE": "true", "PREFILL_TP": "4", "PREFILL_EP": "1"}, ["EVAL_CONC", "PREFILL_DP_ATTN", "DECODE_DP_ATTN"]), +]) # fmt: skip +def test_shims_name_every_missing_input(checkout, single_node_env, tmp_path, node, inputs, missing): + status = tmp_path / "infx-eval-exit-code" + target = status if node == "single" else checkout + + result = shim(checkout, node, {**single_node_env, **inputs}, "http://127.0.0.1:9", str(target)) + + assert result.returncode == 1 + assert [line[4:] for line in result.stdout.splitlines() if line.startswith(" - ")] == missing + if node == "single": + assert status.read_text() == "1\n" + assert trace(tmp_path / "lm_eval.jsonl") == [] + + +def test_batched_lm_eval_defers_a_failed_concurrency_to_score_validation( + base_env, openai, tmp_path +): + staged = tmp_path / "eval_results" + staged.mkdir() + (staged / "results_stub_conc8.json").write_text("{}") # left behind by an earlier eval + env = {**base_env, **MULTI_NODE, "STUB_LM_EVAL_FAIL_AT": "4"} + + assert evaluate(openai.url, "1 4 8", staged, environ=env) == 0 + + runs = trace(tmp_path / "lm_eval.jsonl") + assert [[arg for arg in model_args(run) if arg.startswith("num_concurrent=")] for run in runs] == [ + ["num_concurrent=1"], ["num_concurrent=4"], ["num_concurrent=8"], + ] # fmt: skip + assert sorted(path.name for path in staged.iterdir()) == [ + "meta_env.json", + "results_stub_conc1.json", + "results_stub_conc8.json", + "results_stub_conc8_2.json", + "samples_gsm8k_stub_conc1.jsonl", + "samples_gsm8k_stub_conc8.jsonl", + ] + meta = json.loads((staged / "meta_env.json").read_text()) + assert {key: meta[key] for key in ("conc", "eval_concs", "completed_eval_concs", "failed_eval_concs")} == { + "conc": 1, "eval_concs": [1, 4, 8], "completed_eval_concs": [1, 8], "failed_eval_concs": [4], + } # fmt: skip + errors = validate_batch_manifest( + str(staged / "meta_env.json"), [str(path) for path in staged.glob("results*.json")] + ) + assert "batched eval failed for concurrency: 4" in errors + + +def recorder(ran: list, *, rc: int = 0, artifacts: tuple[str, ...] = ("results_x.json",)): + """A framework that records the context it ran with and writes ``artifacts``.""" + + def run(ctx): + ran.append(ctx) + for name in artifacts: + (ctx.results_dir / name).write_text("{}") + return EvalOutcome(returncode=rc, suite=ctx.suite or "default_suite") + + return run + + +@pytest.mark.parametrize(("env_framework", "cli_framework", "expected"), [ + (None, None, "lm-eval"), + (None, "kimi-vendor", "kimi-vendor"), + ("bfcl", "kimi-vendor", "bfcl"), +]) # fmt: skip +def test_the_environment_framework_overrides_the_command_line( + monkeypatch, base_env, tmp_path, env_framework, cli_framework, expected +): + ran = {name: [] for name in FRAMEWORKS} + for name in FRAMEWORKS: + monkeypatch.setitem(FRAMEWORKS, name, recorder(ran[name])) + env = {**base_env, "IS_MULTINODE": "false", "EVAL_MAX_MODEL_LEN": "4096"} + if env_framework: + env["EVAL_FRAMEWORK"] = env_framework + + assert evaluate("http://127.0.0.1:9", "2", tmp_path / "out", cli_framework, environ=env) == 0 + assert [name for name, runs in ran.items() if runs] == [expected] + + +@pytest.mark.parametrize(("endpoint", "concurrency", "inputs", "message"), [ + ("http://127.0.0.1:9", "2", {"EVAL_FRAMEWORK": "kimi-vendor", "EVAL_SUITE": 'kimi"suite'}, "EVAL_SUITE may contain only"), + ("http://127.0.0.1:9", "2", {"EVAL_SUITE": "gpqa_diamond"}, "only supported with bfcl, kimi-vendor, minimax-vendor"), + ("http://127.0.0.1:9", "2", {"EVAL_FRAMEWORK": "swebench"}, "unknown eval framework 'swebench'"), + ("http://127.0.0.1:9", "1 4", {"EVAL_FRAMEWORK": "kimi-vendor"}, "batched eval concurrency is only supported for lm-eval"), + ("http://127.0.0.1:9", "4 0", {}, "--concurrency must be a positive integer"), + ("http://127.0.0.1:9", " ", {}, "--concurrency must name at least one concurrency"), + ("127.0.0.1:9", "2", {}, "--endpoint must be an http"), + ("http://127.0.0.1:99999", "2", {}, "--endpoint must be an http"), + ("http://127.0.0.1:9", "2", {"TP": "8x"}, "TP must be an integer"), +]) # fmt: skip +def test_invalid_requests_fail_before_any_eval( + monkeypatch, base_env, tmp_path, endpoint, concurrency, inputs, message +): + ran = [] + for name in FRAMEWORKS: + monkeypatch.setitem(FRAMEWORKS, name, recorder(ran)) + env = {**base_env, "IS_MULTINODE": "false", **inputs} + + with pytest.raises(InputError, match=message): + evaluate(endpoint, concurrency, tmp_path / "out", environ=env) + assert ran == [] + assert not (tmp_path / "out").exists() + + +def test_a_failed_vendor_eval_is_staged_and_returns_its_exit_code(monkeypatch, base_env, tmp_path): + ran = [] + artifacts = ("results_bfcl.json", "bfcl_report.json", "adapter.log") + monkeypatch.setitem(FRAMEWORKS, "bfcl", recorder(ran, rc=7, artifacts=artifacts)) + env = {**base_env, "IS_MULTINODE": "false", "EVAL_FRAMEWORK": "bfcl"} + + assert evaluate("http://127.0.0.1:9/", "7", tmp_path / "out", environ=env) == 7 + + [ctx] = ran + assert (ctx.base_url, ctx.model, ctx.concurrency, ctx.suite) == ( + "http://127.0.0.1:9", "served-model", 7, None, + ) + assert not ctx.results_dir.exists() + staged = sorted(path.name for path in (tmp_path / "out").iterdir()) + assert staged == ["bfcl_report.json", "meta_env.json", "results_bfcl.json"] + meta = json.loads((tmp_path / "out/meta_env.json").read_text()) + assert (meta["eval_suite"], meta["conc"]) == ("default_suite", 7) + + +@pytest.mark.parametrize(("framework", "polled"), [("kimi-vendor", True), ("lm-eval", False)]) +def test_eval_only_vendor_evals_wait_for_the_served_model( + monkeypatch, base_env, openai, tmp_path, framework, polled +): + seen = [] + + def run(ctx): + seen.extend(openai.requests) + return EvalOutcome(returncode=0, suite="suite") + + monkeypatch.setitem(FRAMEWORKS, framework, run) + env = { + **base_env, "IS_MULTINODE": "false", "EVAL_ONLY": "true", "EVAL_FRAMEWORK": framework, + "EVAL_MAX_MODEL_LEN": "4096", "EVAL_ENDPOINT_READY_TIMEOUT_SECONDS": "2", + "EVAL_MODEL_STABILIZATION_SECONDS": "600", + } # fmt: skip + + assert evaluate(openai.url, "1", tmp_path / "out", environ=env) == 0 + # The server lists only MODEL_NAME, so readiness passes only for the name eval requests use. + expected = [("GET", "/v1/models"), ("GET", "/v1/chat/completions")] + assert seen == (expected if polled else []) + + +@pytest.mark.parametrize(("inputs", "context", "max_tokens"), [ + ({"MAX_MODEL_LEN": "2048"}, 2048, 1024), + ({"MAX_MODEL_LEN": "10240"}, 8192, 4096), + ({"MAX_MODEL_LEN": "0"}, 8192, 4096), + ({"MAX_MODEL_LEN": "0", "MODEL_PATH": "/no/such/model"}, 16384, 12288), + ({"MAX_MODEL_LEN": "2048", "EVAL_MAX_MODEL_LEN": "4096"}, 4096, 2048), + ({"EVAL_MAX_MODEL_LEN": "4097"}, 4097, 1), + ({"EVAL_MAX_MODEL_LEN": "30000"}, 30000, 16384), +]) # fmt: skip +def test_request_budget_is_the_benchmark_context_capped_at_the_checkpoint( + checkpoint, inputs, context, max_tokens +): + environ = {"MODEL": "/no/such/model", "MODEL_PATH": str(checkpoint), **inputs} + + assert lm_eval.context_length(environ) == context + assert lm_eval.max_output_tokens(context) == max_tokens + + +def pip_step(argv: list[str]) -> str: + """Name one recorded pip invocation of the lm-eval install.""" + if argv[:3] == ["uninstall", "-y", "torchvision"]: + return "uninstall" + if argv[-1] == "lm-eval[api]": + return "release" + return "git" if argv[-1].startswith("git+") else "archive" + + +@pytest.mark.parametrize(("image", "git", "failing", "expected"), [ + ("rocm/atom:latest", True, "git+", ["uninstall", "release", "git", "archive"]), + ("lmsysorg/sglang:latest", True, None, ["release", "git"]), + ("lmsysorg/sglang:latest", False, None, ["release", "archive"]), +]) # fmt: skip +def test_lm_eval_install_drops_torchvision_on_atom_and_falls_back_to_the_archive( + base_env, tmp_path, image, git, failing, expected +): + bin_dir = tmp_path / "path" + bin_dir.mkdir() + if git: + executable(bin_dir / "git", "#!/bin/sh\n") + env = {**base_env, "PATH": str(bin_dir), "IMAGE": image} + if failing: + env["STUB_PIP_FAIL"] = failing + + lm_eval.install(env) + + assert [pip_step(argv) for argv in trace(tmp_path / "pip.jsonl")] == expected diff --git a/inferencex-e2e/infx/tests/bench/test_eval_meta.py b/inferencex-e2e/infx/tests/bench/test_eval_meta.py new file mode 100644 index 0000000000..f4746f3887 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_eval_meta.py @@ -0,0 +1,41 @@ +"""meta_env.json: the workflow's topology names mapped onto the fields consumers read.""" + +import pytest + +from infx.bench.eval import meta + +TOPOLOGY = ( + "tp", "ep", "dp_attention", + "prefill_tp", "prefill_ep", "prefill_dp_attention", "prefill_num_workers", + "decode_tp", "decode_ep", "decode_dp_attention", "decode_num_workers", +) # fmt: skip + + +@pytest.mark.parametrize(("inputs", "expected"), [ + # srt-slurm names: explicit per-phase DP attention wins over a stale DP_ATTENTION. + ( + {"PREFILL_TP": "4", "PREFILL_EP": "4", "PREFILL_DP_ATTN": "true", + "DECODE_TP": "8", "DECODE_EP": "8", "DECODE_DP_ATTN": "false", "DP_ATTENTION": "false"}, + (4, 4, True, 4, 4, True, 1, 8, 8, False, 1), + ), + # Decode inherits prefill's TP and EP; worker counts default to one each. + ( + {"PREFILL_TP": "8", "PREFILL_EP": "2", "DECODE_DP_ATTN": "true"}, + (8, 2, False, 8, 2, False, 1, 8, 2, True, 1), + ), +]) # fmt: skip +def test_disaggregated_topology_maps_onto_metadata(inputs, expected): + document = meta.build({"IS_MULTINODE": "true", **inputs}, conc=4, suite="gsm8k") + + assert tuple(document[key] for key in TOPOLOGY) == expected + + +@pytest.mark.parametrize(("inputs", "expected"), [ + ({}, ("sglang", "fp8")), + ({"FRAMEWORK": "dynamo-sglang"}, ("dynamo-sglang", "fp8")), +]) # fmt: skip +def test_framework_and_precision_fall_back_to_the_result_filename(inputs, expected): + environ = {"IS_MULTINODE": "false", "RESULT_FILENAME": "dsr1_1k1k_fp8_sglang_tp8-ep1_conc4"} + document = meta.build({**environ, **inputs}, conc=4, suite="gsm8k") + + assert (document["framework"], document["precision"]) == expected diff --git a/inferencex-e2e/infx/tests/bench/test_fixed_seq.py b/inferencex-e2e/infx/tests/bench/test_fixed_seq.py new file mode 100644 index 0000000000..a74b5bb0f7 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_fixed_seq.py @@ -0,0 +1,312 @@ +"""Fixed-sequence lanes against a stub client, pip3, nvidia-smi, and a local frontend.""" + +from __future__ import annotations + +import json +import shlex +import shutil +import subprocess +import sys +from pathlib import Path +from urllib.parse import urlsplit + +import pytest + +from infx.bench.__main__ import main as bench +from infx.tests.bench.stubs import executable + +REPO_ROOT = Path(__file__).resolve().parents[3] +FINAL_SAMPLE = "2026/07/23 12:00:11.000, 0, 500.00 W, 65, 1000, 1000, 90 %, 10 %" +# Client flags every lane passes, whatever the point. +POLICY = { + "--dataset-name": "random", + "--request-rate": "inf", + "--ignore-eos": True, + "--save-result": True, + "--percentile-metrics": "ttft,tpot,itl,e2el", +} + +# Records each client run and writes the result file the real client would. +FAKE_CLIENT = """ +import json, os, sys +from pathlib import Path + +argv = sys.argv[1:] +value = lambda flag: argv[argv.index(flag) + 1] +result = Path(value("--result-dir")) / value("--result-filename") +record = { + "argv": argv, + "monitored": (result.parent / "gpu_metrics.csv").exists(), + "safe_path": os.environ.get("PYTHONSAFEPATH"), +} +with open(os.environ["FAKE_CLIENT_LOG"], "a") as log: + log.write(json.dumps(record) + "\\n") +if value("--max-concurrency") == os.environ.get("FAKE_CLIENT_FAIL_CONC"): + sys.exit(3) +result.write_text(json.dumps( + {"benchmark_start_time_unix": 100.0, "benchmark_end_time_unix": 160.0, "duration": 60.0} +)) +""" + + +@pytest.fixture +def tools(tmp_path: Path) -> Path: + """PATH with a stub benchmark client behind python3, a recording pip3, and nvidia-smi.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + for tool in ("sh", "sleep", "dirname", "env"): + (bin_dir / tool).symlink_to(shutil.which(tool)) + (tmp_path / "fake_client.py").write_text(FAKE_CLIENT) + python, client = shlex.quote(sys.executable), shlex.quote(str(tmp_path / "fake_client.py")) + stubs = { + "python3": f""" +if [ "$1 $2" = "-m infx.bench_serving.benchmark_serving" ]; then + shift 2 + exec {python} {client} "$@" +fi +exec {python} "$@" +""", + "pip3": f"printf '%s\\n' \"$*\" >> {shlex.quote(str(tmp_path / 'pip3.log'))}\n", + "nvidia-smi": f""" +case "$*" in + *" -l 1") printf 'timestamp, index, power.draw [W]\\n'; exec sleep 30 ;; + *noheader*) printf '%s\\n' {shlex.quote(FINAL_SAMPLE)} ;; +esac +""", + } + for name, body in stubs.items(): + executable(bin_dir / name, f"#!/bin/sh\n{body}") + return bin_dir + + +def client_runs(tmp_path: Path) -> list[dict]: + log = tmp_path / "client.log" + return [json.loads(line) for line in log.read_text().splitlines()] if log.exists() else [] + + +def options(argv: list[str]) -> dict[str, str | bool]: + """Client flags as a mapping; a flag without a value maps to True.""" + parsed: dict[str, str | bool] = {} + for index, token in enumerate(argv): + if token.startswith("--"): + following = argv[index + 1] if index + 1 < len(argv) else "--" + parsed[token] = True if following.startswith("--") else following + return parsed + + +def use_env(monkeypatch: pytest.MonkeyPatch, env: dict[str, str | None]) -> None: + for name, value in env.items(): + if value is None: + monkeypatch.delenv(name, raising=False) + else: + monkeypatch.setenv(name, value) + + +def run_shim(shim: str, env: dict[str, str | None], *args: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [shutil.which("bash"), str(REPO_ROOT / "benchmarks" / shim), *args], + env={name: value for name, value in env.items() if value is not None}, + capture_output=True, text=True, timeout=60, check=False, + ) # fmt: skip + + +def single_node_env(tmp_path: Path, tools: Path, **overrides: str | None) -> dict[str, str | None]: + (tmp_path / "logs").mkdir(exist_ok=True) + return { + "PATH": str(tools), + "PYTHONDONTWRITEBYTECODE": "1", + "FAKE_CLIENT_LOG": str(tmp_path / "client.log"), + "MODEL": "org/Model-FP8", + "CONC": "4", + "ISL": "1024", + "OSL": "128", + "RANDOM_RANGE_RATIO": "0.8", + "RESULT_FILENAME": "point_conc4", + "RESULT_DIR": str(tmp_path / "logs"), + "SRT_FRONTEND_HOST": "10.0.0.7", + "SRT_FRONTEND_PORT": "8000", + "RUN_EVAL": "false", + "EVAL_ONLY": "false", + "GPU_MONITOR_INTERVAL": "1", + "USE_CHAT_TEMPLATE": "false", + "FRAMEWORK": "sglang", + **overrides, + } + + +def sweep_env( + tmp_path: Path, tools: Path, url: str, **overrides: str | None +) -> dict[str, str | None]: + return { + "PATH": str(tools), + "PYTHONDONTWRITEBYTECODE": "1", + "FAKE_CLIENT_LOG": str(tmp_path / "client.log"), + "ISL": "1024", + "OSL": "128", + "RANDOM_RANGE_RATIO": "0.8", + "SRT_FRONTEND_HOST": "127.0.0.1", + "SRT_FRONTEND_PORT": str(urlsplit(url).port), + "CONC_LIST": "4 16", + "PREFILL_NUM_WORKERS": "1", + "PREFILL_TP": "2", + "DECODE_NUM_WORKERS": "2", + "DECODE_TP": "4", + "TOKENIZER": None, + "SRT_MEASUREMENT_WINDOW_DIR": None, + **overrides, + } + + +def frontend(*models: str): + """An OpenAI frontend listing ``models``.""" + return lambda _method, path: ( + (200, {"data": [{"id": model, "object": "model"} for model in models]}) + if path == "/v1/models" + else (404, {}) + ) + + +@pytest.mark.parametrize( + ("framework", "chat_template", "args", "flags"), + [ + ("sglang", "true", [], {"--backend": "vllm", "--use-chat-template": True}), + ("trt", "false", ["--trust-remote-code"], {"--backend": "openai", "--trust-remote-code": True}), + ], +) +def test_single_node_shim_runs_one_point_under_the_monitor( + tmp_path, tools, framework, chat_template, args, flags +): + env = single_node_env(tmp_path, tools, FRAMEWORK=framework, USE_CHAT_TEMPLATE=chat_template) + + result = run_shim("single_node/srt_fixed_sequence.sh", env, *args) + + assert result.returncode == 0, result.stderr + [run] = client_runs(tmp_path) + logs = tmp_path / "logs" + assert options(run["argv"]) == { + **POLICY, + "--model": "org/Model-FP8", + "--base-url": "http://10.0.0.7:8000", + "--random-input-len": "1024", + "--random-output-len": "128", + "--random-range-ratio": "0.8", + "--num-prompts": "40", + "--max-concurrency": "4", + "--num-warmups": "8", + "--result-dir": str(logs), + "--result-filename": "point_conc4.json", + **flags, + } + assert (logs / "point_conc4.json").is_file() + # The sampler was already writing when the client started and stopped after it. + assert run["monitored"] + # The shim sets PYTHONSAFEPATH for infx.bench only; Python-script tools break under it. + assert run["safe_path"] is None + assert (logs / "gpu_metrics.csv").read_text().endswith(FINAL_SAMPLE + "\n") + assert (tmp_path / "pip3.log").read_text() == ( + "install --break-system-packages sentencepiece datasets pandas\n" + ) + + +@pytest.mark.parametrize( + ("overrides", "returncode", "reported"), + [ + ({"EVAL_ONLY": "true"}, 0, "EVAL_ONLY mode: skipping throughput benchmark\n"), + ({"MODEL": None, "CONC": ""}, 1, "not set:\n - MODEL\n - CONC\n"), + ({"FRAMEWORK": "vllm"}, 1, "ERROR: unsupported fixed-sequence FRAMEWORK: vllm\n"), + ({"USE_CHAT_TEMPLATE": "yes"}, 1, "ERROR: USE_CHAT_TEMPLATE must be true or false, got 'yes'\n"), + ({"RESULT_DIR": "/nonexistent/logs"}, 1, "ERROR: RESULT_DIR must be an existing"), + ], + ids=["eval-only", "missing", "unsupported-framework", "malformed-flag", "no-result-dir"], +) +def test_single_node_point_runs_nothing_when_eval_only_or_misconfigured( + tmp_path, tools, monkeypatch, capsys, overrides, returncode, reported +): + use_env(monkeypatch, single_node_env(tmp_path, tools, **overrides)) + + assert bench(["fixed-seq", "srt-single"]) == returncode + assert reported in "".join(capsys.readouterr()) + assert client_runs(tmp_path) == [] + assert not (tmp_path / "pip3.log").exists() + assert list((tmp_path / "logs").iterdir()) == [] + + +def test_multi_node_shim_writes_one_result_and_power_window_per_concurrency( + tmp_path, tools, http_server +): + url = http_server(frontend("served/model-id")) + windows = tmp_path / "logs" / "power" / "windows" + windows.mkdir(parents=True) + env = sweep_env(tmp_path, tools, url, TOKENIZER="/model", SRT_MEASUREMENT_WINDOW_DIR=str(windows)) + + # A later --logs-dir overrides the shim's /logs. + result = run_shim("multi_node/srt_fixed_sequence.sh", env, "--logs-dir", str(tmp_path / "logs")) + + assert result.returncode == 0, result.stderr + point_dir = tmp_path / "logs" / "sa-bench_isl_1024_osl_128" + names = [ + "results_concurrency_4_gpus_10_ctx_2_gen_8.json", + "results_concurrency_16_gpus_10_ctx_2_gen_8.json", + ] + first, second = (options(run["argv"]) for run in client_runs(tmp_path)) + assert first == { + **POLICY, + "--model": "served/model-id", + "--backend": "openai", + "--endpoint": "/v1/completions", + "--base-url": url, + "--tokenizer": "/model", + "--random-input-len": "1024", + "--random-output-len": "128", + "--random-range-ratio": "0.8", + "--random-num-workers": "1", + "--num-prompts": "40", + "--max-concurrency": "4", + "--num-warmups": "8", + "--disable-tqdm": True, + "--use-chat-template": True, + "--trust-remote-code": True, + "--result-dir": str(point_dir), + "--result-filename": names[0], + } + assert {key: second[key] for key in ("--num-prompts", "--max-concurrency", "--num-warmups")} == { + "--num-prompts": "160", + "--max-concurrency": "16", + "--num-warmups": "32", + } + assert sorted(path.name for path in point_dir.iterdir()) == sorted(names) + written = {path.name: json.loads(path.read_text()) for path in windows.iterdir()} + assert {name: (window["result_path"], window["concurrency"]) for name, window in written.items()} == { + names[0]: (f"sa-bench_isl_1024_osl_128/{names[0]}", 4), + names[1]: (f"sa-bench_isl_1024_osl_128/{names[1]}", 16), + } + + +def test_sweep_stops_at_the_first_failing_point(tmp_path, tools, http_server, monkeypatch): + url = http_server(frontend("served/model-id")) + # An aggregated row: no decode workers, and no TOKENIZER. + use_env(monkeypatch, sweep_env( + tmp_path, tools, url, CONC_LIST="4 16 32", PREFILL_NUM_WORKERS="2", DECODE_NUM_WORKERS="0", + FAKE_CLIENT_FAIL_CONC="16", + )) # fmt: skip + + assert bench(["fixed-seq", "srt-sweep", "--logs-dir", str(tmp_path / "logs")]) == 3 + runs = [options(run["argv"]) for run in client_runs(tmp_path)] + assert [(run["--max-concurrency"], run["--tokenizer"]) for run in runs] == [ + ("4", "served/model-id"), + ("16", "served/model-id"), + ] + point_dir = tmp_path / "logs" / "sa-bench_isl_1024_osl_128" + assert [path.name for path in point_dir.iterdir()] == ["results_concurrency_4_gpus_4_ctx_4_gen_0.json"] + + +def test_sweep_without_a_served_model_runs_nothing( + tmp_path, tools, http_server, monkeypatch, capsys +): + url = http_server(frontend()) + use_env(monkeypatch, sweep_env(tmp_path, tools, url)) + + assert bench(["fixed-seq", "srt-sweep", "--logs-dir", str(tmp_path / "logs")]) == 1 + assert capsys.readouterr().err == f"ERROR: {url}/v1/models lists no served model\n" + assert client_runs(tmp_path) == [] + assert not (tmp_path / "logs").exists() diff --git a/inferencex-e2e/infx/tests/bench/test_gpu_monitor.py b/inferencex-e2e/infx/tests/bench/test_gpu_monitor.py new file mode 100644 index 0000000000..ebd05ff769 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_gpu_monitor.py @@ -0,0 +1,176 @@ +"""GPU monitor against stub SMI tools: telemetry files, workload status, and signal relay.""" + +from __future__ import annotations + +import json +import os +import shlex +import shutil +import signal +import subprocess +import sys +import time +import types +from pathlib import Path + +import pytest + +from infx.bench import gpu_monitor +from infx.bench.gpu_monitor import GpuMonitor +from infx.tests.bench.stubs import executable + +REPO_ROOT = Path(__file__).resolve().parents[3] +# The columns infx.results.power reads from the NVIDIA stream. +NVIDIA_QUERY = ( + "timestamp,index,power.draw,temperature.gpu,clocks.current.sm,clocks.current.memory," + "utilization.gpu,utilization.memory" +) +NVIDIA_HEADER = ( + "timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], " + "clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]" +) +NVIDIA_SAMPLE = "2026/07/23 12:00:09.000, 0, 490.00 W, 64, 990, 990, 89 %, 9 %" +NVIDIA_FINAL = "2026/07/23 12:00:11.000, 0, 500.00 W, 65, 1000, 1000, 90 %, 10 %" +IDENTITY = ( + "index, uuid, pci.bus_id, name, driver_version\n" + "0, GPU-device-a, 00000000:01:00.0, NVIDIA Test GPU, 590.00\n" +) + + +def tool_dir(tmp_path: Path, **stubs: str) -> Path: + """A PATH with only the shell basics and the given stub tools, so no real SMI is found.""" + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + for tool in ("sh", "sleep"): + (bin_dir / tool).symlink_to(shutil.which(tool)) + for name, body in stubs.items(): + executable(bin_dir / name.replace("_", "-"), f"#!/bin/sh\n{body}") + return bin_dir + + +def nvidia_smi(stream: str, *, interval: int = 1, identity_ok: bool = True, pid_file: Path | None = None) -> str: + """nvidia-smi answering the identity, streaming, and one-shot sample queries.""" + record_pid = f"echo $$ > {shlex.quote(str(pid_file))}; " if pid_file else "" + return f"""case "$*" in + "--query-gpu=index,uuid,pci.bus_id,name,driver_version --format=csv") + printf '%s' {shlex.quote(IDENTITY)} + {'' if identity_ok else 'exit 1'} ;; + "--query-gpu={NVIDIA_QUERY} --format=csv -l {interval}") + {record_pid}printf '%s' {shlex.quote(stream)} + exec sleep 30 ;; + "--query-gpu={NVIDIA_QUERY} --format=csv,noheader") + printf '%s\\n' {shlex.quote(NVIDIA_FINAL)} ;; + *) echo "unexpected nvidia-smi $*"; exit 1 ;; +esac +""" + + +def wait_until(condition, timeout: float = 10) -> None: + deadline = time.monotonic() + timeout + while not condition(): + assert time.monotonic() < deadline, "condition not reached" + time.sleep(0.02) + + +def text(path: Path) -> str: + return path.read_text() if path.is_file() else "" + + +@pytest.mark.parametrize(("identity_ok", "identity"), [(True, IDENTITY), (False, None)]) +def test_nvidia_stream_drops_a_truncated_row_then_appends_the_bracketing_sample( + tmp_path, monkeypatch, identity_ok, identity +): + truncated = "2026/07/23 12:00:10.000, 0, 52" + stream = f"{NVIDIA_HEADER}\n{NVIDIA_SAMPLE}\n{truncated}" + stub = nvidia_smi(stream, interval=2, identity_ok=identity_ok) + monkeypatch.setenv("PATH", str(tool_dir(tmp_path, nvidia_smi=stub))) + (tmp_path / "power").mkdir() + output = tmp_path / "power" / "node0.csv" + + with GpuMonitor(output, interval=2) as monitor: + wait_until(lambda: text(output).endswith(truncated)) + assert monitor.vendor == "nvidia" + + assert output.read_text().splitlines() == [NVIDIA_HEADER, NVIDIA_SAMPLE, NVIDIA_FINAL] + # A failed identity query costs only its sidecar, never the telemetry. + sidecar = tmp_path / "power" / "node0_identity.csv" + assert (sidecar.read_text() if sidecar.exists() else None) == identity + + +def test_amd_stream_keeps_one_header_and_brackets_the_workload(tmp_path, monkeypatch): + first_snapshot = shlex.quote(str(tmp_path / "energy_start_taken")) + amd_smi = f"""case "$*" in + "metric -p -c -t -u -w 3 --csv") + printf "'CTRL' + 'C' to stop watching output:\\n" + printf 'timestamp,gpu,socket_power\\n123,0,400\\ntimestamp,gpu,socket_power\\n124,0,410\\n125,0,4' + exec sleep 30 ;; + "metric -E --csv") + if [ -e {first_snapshot} ]; then energy=200; else : > {first_snapshot}; energy=100; fi + printf 'gpu,total_energy_consumption\\n0,%s\\n' "$energy" ;; + "static --json") printf '{{"gpu_data": []}}\\n' ;; + *) echo "unexpected amd-smi $*"; exit 1 ;; +esac +""" + monkeypatch.setenv("PATH", str(tool_dir(tmp_path, amd_smi=amd_smi))) + drains = [] + monkeypatch.setattr(gpu_monitor, "time", types.SimpleNamespace(sleep=drains.append)) + output = tmp_path / "gpu_metrics.csv" + rows = "timestamp,gpu,socket_power\n123,0,400\n124,0,410\n" + + with GpuMonitor(output, interval=3) as monitor: + # Rows reach the file while amd-smi is still running, not at its exit. + wait_until(lambda: text(output) == rows) + assert monitor.vendor == "amd" + + # Two ticks past the workload: amd-smi stamps whole seconds. + assert drains == [5] + assert output.read_text() == rows + assert (tmp_path / "gpu_metrics_energy_start.csv").read_text() == "gpu,total_energy_consumption\n0,100\n" + assert (tmp_path / "gpu_metrics_energy_end.csv").read_text() == "gpu,total_energy_consumption\n0,200\n" + assert json.loads((tmp_path / "gpu_metrics_identity.json").read_text()) == {"gpu_data": []} + + +@pytest.mark.parametrize("with_gpu", [True, False]) +def test_monitor_returns_the_workload_status(tmp_path, monkeypatch, with_gpu): + stubs = {"nvidia_smi": nvidia_smi(f"{NVIDIA_HEADER}\n")} if with_gpu else {} + monkeypatch.setenv("PATH", str(tool_dir(tmp_path, **stubs))) + output = tmp_path / "gpu_metrics.csv" + + assert gpu_monitor.run(output, 1, ["sh", "-c", "exit 7"]) == 7 + # Without a GPU tool the workload runs unmonitored. + assert output.exists() == with_gpu + + +RUN_MONITOR = ( + "import sys; from pathlib import Path; from infx.bench.gpu_monitor import run; " + "sys.exit(run(Path(sys.argv[1]), 1, sys.argv[2:]))" +) + + +@pytest.mark.parametrize( + ("sent", "rc"), [(signal.SIGTERM, 143), (signal.SIGHUP, 129)], ids=["TERM", "HUP"] +) +def test_monitor_relays_the_signal_and_still_stops_the_sampler(tmp_path, sent, rc): + pid_file = tmp_path / "sampler.pid" + bin_dir = tool_dir(tmp_path, nvidia_smi=nvidia_smi(f"{NVIDIA_HEADER}\n", pid_file=pid_file)) + output, started, relayed = tmp_path / "gpu_metrics.csv", tmp_path / "started", tmp_path / "relayed" + trap = sent.name.removeprefix("SIG") + workload = f'trap "echo {trap} > {relayed}; exit 0" {trap}; : > {started}; while :; do sleep 0.05; done' + process = subprocess.Popen( + [sys.executable, "-c", RUN_MONITOR, str(output), "sh", "-c", workload], + env={"PATH": str(bin_dir), "PYTHONPATH": str(REPO_ROOT)}, + stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, + ) # fmt: skip + try: + wait_until(lambda: started.exists() and pid_file.exists()) + process.send_signal(sent) + _, stderr = process.communicate(timeout=30) + finally: + process.kill() + + # The workload handled the signal and exited 0, but an interrupted run never passes. + assert process.returncode == rc, stderr + assert relayed.read_text() == f"{trap}\n" + assert output.read_text().endswith(NVIDIA_FINAL + "\n") + with pytest.raises(ProcessLookupError): + os.kill(int(pid_file.read_text()), 0) diff --git a/inferencex-e2e/infx/tests/bench/test_server.py b/inferencex-e2e/infx/tests/bench/test_server.py new file mode 100644 index 0000000000..cf069dff92 --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_server.py @@ -0,0 +1,86 @@ +"""Server readiness against a local HTTP server: the ``wait`` command and the OpenAI chat route.""" + +from __future__ import annotations + +import os +import subprocess +import sys +import time +from pathlib import Path + +import pytest + +from infx.bench import server + +REPO_ROOT = Path(__file__).resolve().parents[3] + + +def frontend(models: list[str] | None, chat: int, health: int): + """``models`` listed at /v1/models (None: 404); a bare chat GET answers ``chat``.""" + + def respond(_method: str, path: str) -> tuple[int, object]: + if path == "/v1/models": + return (404, {}) if models is None else (200, {"data": [{"id": m} for m in models]}) + return {"/v1/chat/completions": chat, "/health": health}.get(path, 404), {} + + return respond + + +def test_wait_streams_new_server_log_until_health_answers(http_server, tmp_path, capsys): + log = tmp_path / "server.log" + log.write_text("loading\n") + statuses = iter([503, 200]) + + def respond(method: str, path: str) -> tuple[int, object]: + with log.open("a") as out: + out.write(f"{method} {path}\n") + return next(statuses), {} + + server.wait_ready(f"{http_server(respond)}/health", pid=os.getpid(), log=log, poll_s=0.01) + + # Each line once; the ready poll's own line lands after the last read. + assert capsys.readouterr().out == "loading\nGET /health\n" + + +def test_wait_fails_once_the_server_process_is_gone(): + dead = subprocess.Popen(["true"]) + dead.wait() + url = "http://127.0.0.1:0/health" + + result = subprocess.run( + [sys.executable, "-m", "infx.bench", "wait", "--url", url, "--pid", str(dead.pid)], + env={"PYTHONPATH": str(REPO_ROOT)}, capture_output=True, text=True, timeout=30, check=False, + ) # fmt: skip + + assert result.returncode == 1 + assert result.stderr == f"ERROR: process {dead.pid} died before {url} became ready\n" + + +@pytest.mark.parametrize( + ("models", "health", "stabilization_s", "waiting"), + [(["m"], 503, 0, False), (["other"], 200, 1, True)], + ids=["route-mounted", "health-held"], +) +def test_chat_route_is_ready_once_the_model_is_served_or_health_holds( + http_server, capsys, models, health, stabilization_s, waiting +): + url = http_server(frontend(models, chat=405, health=health)) + chat = f"{url}/v1/chat/completions" + start = time.monotonic() + + server.wait_chat_route(url, "m", (5, stabilization_s), poll_s=0.01) + + assert time.monotonic() - start >= stabilization_s + waited = f"Waiting for {chat} ('m'): 0/5s\n" if waiting else "" + assert capsys.readouterr().out == f"{waited}OpenAI chat endpoint ready for model 'm': {chat}\n" + + +def test_chat_route_gives_up_after_its_timeout(http_server, capsys): + url = http_server(frontend(None, chat=404, health=503)) + chat = f"{url}/v1/chat/completions" + + with pytest.raises(server.NotReadyError) as raised: + server.wait_chat_route(url, "m", (1, 0), poll_s=0.01) + + assert str(raised.value) == f"chat endpoint for model 'm' not ready within 1s: {chat}" + assert capsys.readouterr().out == f"Waiting for {chat} ('m'): 0/1s\n" diff --git a/inferencex-e2e/infx/tests/bench/test_vendor_eval.py b/inferencex-e2e/infx/tests/bench/test_vendor_eval.py new file mode 100644 index 0000000000..bfbfc9533c --- /dev/null +++ b/inferencex-e2e/infx/tests/bench/test_vendor_eval.py @@ -0,0 +1,419 @@ +"""``infx.bench.eval.vendor``: the real runner with stub interpreters and adapters.""" + +from __future__ import annotations + +import json +import os +import sys +import tarfile +import time +from dataclasses import replace +from pathlib import Path +from typing import Any + +import pytest + +from infx.bench.env import InputError +from infx.bench.eval import FRAMEWORKS, vendor +from infx.bench.eval.context import EvalContext, EvalOutcome +from infx.tests.bench.stubs import executable + +REPO_ROOT = Path(__file__).resolve().parents[3] +BASE_URL = "http://127.0.0.1:8888" + +PYTHON_STUB = r''' +import json +import os +import shutil +import sys +from pathlib import Path + +argv = sys.argv[1:] +with open(os.environ["STUB_LOG"], "a") as log: + log.write(json.dumps({"python": sys.argv[0], "argv": argv}) + "\n") +if argv[0] == "-c": + sys.exit(int(os.environ["STUB_VERSION_RC"])) +if argv[:2] == ["-m", "pip"]: + returncode = int(os.environ.get("STUB_PIP_RC", "0")) + if returncode == 0 and "--prefix" in argv: + uv = Path(argv[argv.index("--prefix") + 1], "bin", "uv") + uv.parent.mkdir(parents=True) + shutil.copy(os.environ["STUB_UV"], uv) + sys.exit(returncode) +if argv[:2] == ["-m", "venv"]: + python = Path(argv[-1], "bin", "python") + python.parent.mkdir(parents=True) + shutil.copy(__file__, python) + sys.exit(0) +env = dict(os.environ) +if "STUB_SITE" in env: + env["PYTHONPATH"] = os.pathsep.join(filter(None, [env["STUB_SITE"], env.get("PYTHONPATH")])) +os.execve(sys.executable, [sys.executable, *argv], env) +''' + +UV_STUB = r''' +import json +import os +import shutil +import sys +from pathlib import Path + +argv = sys.argv[1:] +record = {"uv": argv, **{name: os.environ.get(name) for name in ("UV_CACHE_DIR", "UV_PYTHON_INSTALL_DIR")}} +with open(os.environ["STUB_LOG"], "a") as log: + log.write(json.dumps(record) + "\n") +python = Path(argv[-1], "bin", "python") +python.parent.mkdir(parents=True) +shutil.copy(os.environ["STUB_PYTHON"], python) +''' + +ADAPTER_STUB = r''' +import json +import os +import sys +import time +from pathlib import Path + +script = Path(sys.argv[0]).name +argv = sys.argv[1:] +if "--integration-error" in argv or argv[:1] == ["failure"]: + mode = "failure" +elif script.startswith("_") or argv[:1] == ["prepare-source"] or "--install-runtime" in argv: + mode = "prepare" +else: + mode = "run" +with open(os.environ["STUB_LOG"], "a") as log: + record = {"script": script, "mode": mode, "argv": argv, "PYTHONPATH": os.environ.get("PYTHONPATH")} + log.write(json.dumps(record) + "\n") +plans = json.loads(os.environ["STUB_PLAN"]) +if mode == "failure" and f"{script} failure" not in plans: + real = os.path.join(os.environ["STUB_REAL_EVALS"], script) + os.execv(sys.executable, [sys.executable, real, *argv]) +plan = plans.get(f"{script} {mode}", {}) +time.sleep(plan.get("sleep", 0)) +for name in plan.get("writes", ()): + Path(argv[argv.index("--output-dir") + 1], name).write_text(json.dumps({"argv": argv})) +if "--bfcl-project-root" in argv: + project = Path(argv[argv.index("--bfcl-project-root") + 1]) + for name, content in plan.get("project", {}).items(): + (project / name).parent.mkdir(parents=True, exist_ok=True) + (project / name).write_text(content) + if "project_symlink" in plan: + (project / plan["project_symlink"]).symlink_to(os.environ["STUB_LOG"]) +sys.exit(plan.get("rc", 0)) +''' + +# What each real adapter publishes on success. +PUBLISHES = { + "kimi_vendor_eval.py": ["results_kimi_vendor_2026-01-01.json", "kimi_vendor_report.json"], + "minimax_provider_eval.py": ["results_minimax_vendor_2026-01-01.json", "minimax_vendor_report.json"], + "minimax_m3_full_eval.py": [ + "results_minimax_vendor_full_2026-01-01.json", + "minimax_vendor_report.json", + "minimax_vendor_results.jsonl", + ], + "bfcl_adapter.py": ["results_bfcl.json", "bfcl_report.json"], +} +# The image python3 is too old, and pip cannot install uv. +PROVISIONING_FAILS = {"STUB_VERSION_RC": "1", "STUB_PIP_RC": "7"} + + +def _flag(argv: list[str], flag: str) -> str: + return argv[argv.index(flag) + 1] + + +class Stub: + def __init__(self, root: Path) -> None: + self.evals = root / "evals" + self.evals.mkdir() + for name in (*PUBLISHES, "_kimi_verifier_archive.py"): + (self.evals / name).write_text(ADAPTER_STUB) + bin_dir = root / "bin" + bin_dir.mkdir() + self.log = root / "calls.jsonl" + self.results = root / "results" + self.results.mkdir() + python = executable(bin_dir / "python3", f"#!{sys.executable}\n{PYTHON_STUB}") + uv = executable(root / "uv", f"#!{sys.executable}\n{UV_STUB}") + self.env = { + "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", + "STUB_LOG": str(self.log), + "STUB_PYTHON": str(python), + "STUB_UV": str(uv), + "STUB_REAL_EVALS": str(REPO_ROOT / "infx" / "evals"), + "STUB_VERSION_RC": "0", + } + + def context( + self, suite: str | None = None, *, plan: dict[str, dict[str, Any]] | None = None, **env: str + ) -> EvalContext: + plans = {f"{script} run": {"writes": names} for script, names in PUBLISHES.items()} + return EvalContext( + base_url=BASE_URL, + model="served-model", + concurrency=1, + context_length=0, + results_dir=self.results, + suite=suite, + env={**self.env, "STUB_PLAN": json.dumps({**plans, **(plan or {})}), **env}, + ) + + def calls(self) -> list[dict[str, Any]]: + if not self.log.exists(): + return [] + return [json.loads(line) for line in self.log.read_text().splitlines()] + + def adapter_calls(self, script: str, mode: str) -> list[dict[str, Any]]: + return [c for c in self.calls() if c.get("script") == script and c["mode"] == mode] + + def interpreter_calls(self, argv_prefix: list[str]) -> list[dict[str, Any]]: + n = len(argv_prefix) + return [c for c in self.calls() if "python" in c and c["argv"][:n] == argv_prefix] + + def result(self, pattern: str) -> dict[str, Any]: + [path] = self.results.glob(pattern) + return json.loads(path.read_text()) + + +@pytest.fixture +def stub(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Stub: + stub = Stub(tmp_path) + monkeypatch.setattr(vendor, "EVALS", stub.evals) + return stub + + +def test_kimi_runs_the_verified_checkout_with_an_isolated_runtime(stub: Stub) -> None: + ctx = stub.context( + MODEL_PREFIX="dsv4", PYTHONPATH="/image/site", OPENAI_API_KEY="sk-caller-key-4711" + ) + + assert FRAMEWORKS["kimi-vendor"](ctx) == EvalOutcome(0, "kimi_tool_call_schema") + + [pip] = stub.interpreter_calls(["-m", "pip"]) + [checkout] = stub.adapter_calls("_kimi_verifier_archive.py", "prepare") + [run] = stub.adapter_calls("kimi_vendor_eval.py", "run") + runtime = _flag(pip["argv"], "--target") + assert run["PYTHONPATH"] == os.pathsep.join([runtime, str(REPO_ROOT), "/image/site"]) + assert _flag(run["argv"], "--verifier-dir") == checkout["argv"][-1] + assert _flag(run["argv"], "--base-url") == f"{BASE_URL}/v1" + assert _flag(run["argv"], "--model") == "served-model" + assert _flag(run["argv"], "--output-dir") == str(stub.results) + assert _flag(run["argv"], "--task-name") == "kimi_tool_call_schema" + assert _flag(run["argv"], "--model-prefix") == "dsv4" + assert "sk-caller-key-4711" not in json.dumps([call["argv"] for call in stub.calls()]) + assert not stub.adapter_calls("kimi_vendor_eval.py", "failure") + assert not Path(checkout["argv"][-1]).exists() + + +@pytest.mark.parametrize( + ("suite", "adapter"), + [(None, "minimax_provider_eval.py"), ("minimax_m3_full", "minimax_m3_full_eval.py")], +) +def test_minimax_runs_stock_sources_with_pinned_dependencies( + stub: Stub, suite: str | None, adapter: str +) -> None: + outcome = FRAMEWORKS["minimax-vendor"](stub.context(suite)) + + assert outcome == EvalOutcome(0, suite or "minimax_m3_smoke") + [source] = stub.adapter_calls("minimax_m3_full_eval.py", "prepare") + [pip] = stub.interpreter_calls(["-m", "pip"]) + [run] = stub.adapter_calls(adapter, "run") + assert run["argv"][0] == "run" + assert _flag(run["argv"], "--source-dir") == _flag(source["argv"], "--source-dir") + assert _flag(run["argv"], "--dependency-dir") == _flag(pip["argv"], "--target") + assert _flag(run["argv"], "--python") == "python3" + + +def test_bfcl_runs_in_a_venv_that_sees_the_image_site_packages(stub: Stub) -> None: + assert FRAMEWORKS["bfcl"](stub.context()) == EvalOutcome(0, "bfcl_smoke") + + [venv] = stub.interpreter_calls(["-m", "venv"]) + assert "--system-site-packages" in venv["argv"] + adapter = str(stub.evals / "bfcl_adapter.py") + interpreters = [call["python"] for call in stub.interpreter_calls([adapter])] + assert interpreters == [str(Path(venv["argv"][-1], "bin", "python"))] * 2 + [install] = stub.adapter_calls("bfcl_adapter.py", "prepare") + [run] = stub.adapter_calls("bfcl_adapter.py", "run") + assert install["argv"][0] == "--install-runtime" + assert _flag(run["argv"], "--suite") == "bfcl_smoke" + assert not Path(_flag(run["argv"], "--bfcl-project-root")).exists() + assert not Path(venv["argv"][-1]).exists() + assert not (stub.results / vendor.BFCL_ARCHIVE).exists() + + +def test_too_old_image_python_runs_the_adapter_in_a_pinned_uv_venv(stub: Stub) -> None: + outcome = FRAMEWORKS["kimi-vendor"](stub.context(STUB_VERSION_RC="1")) + + assert outcome.returncode == 0 + [uv] = [call for call in stub.calls() if "uv" in call] + venv = Path(uv["uv"][-1]) + assert "--system-site-packages" not in uv["uv"] + adapter = str(stub.evals / "kimi_vendor_eval.py") + [run] = stub.interpreter_calls([adapter]) + assert run["python"] == str(venv / "bin" / "python") + assert Path(uv["UV_CACHE_DIR"]).parent == venv.parent + assert Path(uv["UV_PYTHON_INSTALL_DIR"]).parent == venv.parent + assert not venv.parent.exists() + + +@pytest.mark.parametrize(("framework", "suite", "inputs", "rc", "message"), [ + ("kimi-vendor", "kimi_tool_call_schema", PROVISIONING_FAILS, 7, + "Kimi Vendor Verifier Python runtime preparation failed with exit code 7"), + ("minimax-vendor", "minimax_m3_smoke", PROVISIONING_FAILS, 7, + "MiniMax Provider Verifier Python runtime preparation failed with exit code 7"), + ("minimax-vendor", "minimax_m3_full", PROVISIONING_FAILS, 7, + "MiniMax M3 full Python runtime preparation failed with exit code 7"), + ("bfcl", "bfcl_vllm_minimax_m3", PROVISIONING_FAILS, 7, + "BFCL Python runtime preparation failed with exit code 7"), + ("kimi-vendor", "kimi_tool_call_schema_full", {"STUB_PIP_RC": "12"}, 12, + "Kimi Vendor Verifier dependency installation failed with exit code 12"), + ("bfcl", "bfcl_vllm_kimi", {"plan": {"bfcl_adapter.py prepare": {"rc": 6}}}, 6, + "BFCL dependency installation failed with exit code 6"), +]) # fmt: skip +def test_setup_failure_is_recorded_by_the_real_adapter_on_python310_and_skips_the_run( + stub: Stub, + tmp_path: Path, + framework: str, + suite: str, + inputs: dict[str, Any], + rc: int, + message: str, +) -> None: + # Python 3.10 has no datetime.UTC; the real adapters must still write the failure. + python310 = tmp_path / "python310" + python310.mkdir() + (python310 / "sitecustomize.py").write_text( + "import datetime\nvars(datetime).pop('UTC', None)\n" + ) + + outcome = FRAMEWORKS[framework](stub.context(suite, STUB_SITE=str(python310), **inputs)) + + assert outcome == EvalOutcome(rc, suite) + failure = stub.result("results*.json") + assert failure["integration_error"]["message"] == message + assert failure["results"] and all(task.startswith(suite) for task in failure["results"]) + assert not [call for call in stub.calls() if call.get("mode") == "run"] + assert not (stub.results / vendor.BFCL_ARCHIVE).exists() + + +@pytest.mark.parametrize(("framework", "adapter", "writes", "message"), [ + ("kimi-vendor", "kimi_vendor_eval.py", [], "Kimi Vendor Verifier evaluation failed with exit code 3"), + ("minimax-vendor", "minimax_provider_eval.py", [], + "MiniMax Provider Verifier evaluation failed with exit code 3"), + # Half a BFCL publication is not a published result. + ("bfcl", "bfcl_adapter.py", ["results_bfcl.json"], "BFCL evaluation failed with exit code 3"), + # A complete publication is the adapter's own verdict. + ("kimi-vendor", "kimi_vendor_eval.py", PUBLISHES["kimi_vendor_eval.py"], None), +]) # fmt: skip +def test_a_failed_run_is_recorded_as_an_integration_error_unless_the_adapter_published_it( + stub: Stub, framework: str, adapter: str, writes: list[str], message: str | None +) -> None: + plan = {f"{adapter} run": {"rc": 3, "writes": writes}} + + outcome = FRAMEWORKS[framework](stub.context(plan=plan)) + + assert outcome.returncode == 3 + assert stub.result("results*.json").get("integration_error", {}).get("message") == message + + +def test_unwritable_failure_result_is_reported_and_keeps_the_setup_code( + stub: Stub, capfd: pytest.CaptureFixture[str] +) -> None: + ctx = stub.context(plan={"kimi_vendor_eval.py failure": {"rc": 5}}, STUB_PIP_RC="12") + + assert FRAMEWORKS["kimi-vendor"](ctx).returncode == 12 + + assert "failed to write the Kimi Vendor Verifier failure artifact (exit code 5)" in ( + capfd.readouterr().err + ) + assert list(stub.results.iterdir()) == [] + + +def test_suite_deadline_kills_the_adapter_and_records_exit_124(stub: Stub) -> None: + bfcl = vendor.PROVIDERS["bfcl"] + provider = replace(bfcl, suites={"bfcl_smoke": replace(bfcl.suites["bfcl_smoke"], timeout_s=1)}) + ctx = stub.context(plan={"bfcl_adapter.py run": {"sleep": 60}}) + started = time.monotonic() + + assert vendor.run(provider, ctx) == EvalOutcome(124, "bfcl_smoke") + + assert time.monotonic() - started < 30 + assert stub.result("results_bfcl.json")["integration_error"]["message"] == ( + "BFCL evaluation failed with exit code 124" + ) + + +def test_bfcl_full_suite_archives_the_upstream_tree_after_a_failed_run(stub: Stub) -> None: + project = {"result/run/generation.json": "{}\n", "score/run/score.json": "{}\n"} + run = {"rc": 2, "writes": PUBLISHES["bfcl_adapter.py"], "project": project} + + outcome = FRAMEWORKS["bfcl"](stub.context("bfcl_vllm_kimi", plan={"bfcl_adapter.py run": run})) + + assert outcome == EvalOutcome(2, "bfcl_vllm_kimi") + with tarfile.open(stub.results / vendor.BFCL_ARCHIVE) as archive: + assert archive.getnames() == [ + "result", + "result/run", + "result/run/generation.json", + "score", + "score/run", + "score/run/score.json", + ] + assert not stub.adapter_calls("bfcl_adapter.py", "failure") + + +def test_bfcl_archive_failure_fails_a_passing_run_and_keeps_its_scores( + stub: Stub, capfd: pytest.CaptureFixture[str] +) -> None: + run = {"writes": PUBLISHES["bfcl_adapter.py"], "project_symlink": "escape.json"} + + outcome = FRAMEWORKS["bfcl"]( + stub.context("bfcl_vllm_minimax_m3", plan={"bfcl_adapter.py run": run}) + ) + + assert outcome == EvalOutcome(1, "bfcl_vllm_minimax_m3") + assert "refusing to archive symbolic link: escape.json" in capfd.readouterr().err + # Neither the archive nor its temporary file is left behind. + assert sorted(path.name for path in stub.results.iterdir()) == sorted( + PUBLISHES["bfcl_adapter.py"] + ) + + +def test_unknown_suite_is_rejected_before_any_work(stub: Stub) -> None: + with pytest.raises(InputError, match="unsupported BFCL suite 'minimax_m3_smoke'"): + FRAMEWORKS["bfcl"](stub.context("minimax_m3_smoke")) + + assert stub.calls() == [] + assert list(stub.results.iterdir()) == [] + + +def test_archive_tree_is_byte_reproducible(tmp_path: Path) -> None: + root = tmp_path / "project" + for name, content in ( + ("result/run/BFCL_v4_simple_python_result.json", '{"id":"simple_python_0"}\n'), + ("score/run/BFCL_v4_simple_python_score.json", '{"accuracy":1.0}\n'), + ("test_case_ids_to_generate.json", '{"simple_python":["simple_python_0"]}\n'), + ): + (root / name).parent.mkdir(parents=True, exist_ok=True) + (root / name).write_text(content) + vendor.archive_tree(root, tmp_path / "first.tar.gz") + for path in root.rglob("*"): + os.utime(path, (1_000_000, 1_000_000)) + vendor.archive_tree(root, tmp_path / "second.tar.gz") + + first = (tmp_path / "first.tar.gz").read_bytes() + assert first == (tmp_path / "second.tar.gz").read_bytes() + with tarfile.open(tmp_path / "first.tar.gz") as archive: + assert archive.getnames() == [ + "result", + "result/run", + "result/run/BFCL_v4_simple_python_result.json", + "score", + "score/run", + "score/run/BFCL_v4_simple_python_score.json", + "test_case_ids_to_generate.json", + ] + member = archive.getmember("score/run/BFCL_v4_simple_python_score.json") + assert (member.uid, member.gid, member.uname, member.gname, member.mtime) == ( + 0, 0, "", "", 0, + ) # fmt: skip diff --git a/inferencex-e2e/infx/tests/bench_serving/test_server_watch.py b/inferencex-e2e/infx/tests/bench_serving/test_server_watch.py deleted file mode 100644 index 11de140146..0000000000 --- a/inferencex-e2e/infx/tests/bench_serving/test_server_watch.py +++ /dev/null @@ -1,123 +0,0 @@ -"""Client lifecycle regressions, using short real processes without a GPU server.""" -import os -import signal -import subprocess -import sys -import time -from pathlib import Path - -import pytest - -from infx.bench_serving import server_watch - - -@pytest.mark.parametrize('state', [None, ('Z', 'original'), ('S', 'reused')]) -def test_dead_zombie_and_reused_pids_are_not_healthy(monkeypatch, state): - monkeypatch.setattr(server_watch, 'process_state', lambda _: state) - assert not server_watch.healthy({'123': 'original'}) - - -def test_client_exit_code_preserved_and_healthy_server_left_alone(): - with subprocess.Popen([sys.executable, '-c', 'import time; time.sleep(30)']) as server: - try: - state = server_watch.snapshot(server.pid) - assert server_watch.run(state, [sys.executable, '-c', 'raise SystemExit(7)'], 0.02) == 7 - assert server.poll() is None - finally: - server.terminate() - - -def test_required_worker_death_stops_client_while_wrapper_lives(tmp_path: Path): - pidfile = tmp_path / 'client.pid' - with subprocess.Popen([sys.executable, '-c', 'import time; time.sleep(30)']) as wrapper, \ - subprocess.Popen([sys.executable, '-c', 'import sys; sys.stdin.read()'], stdin=subprocess.PIPE) as worker: - try: - state = {**server_watch.snapshot(wrapper.pid), **server_watch.snapshot(worker.pid)} - command = [sys.executable, '-c', - 'import os,time,pathlib,sys,signal; pathlib.Path(sys.argv[1]).write_text(str(os.getpid())); ' - 'os.kill(int(sys.argv[2]), signal.SIGTERM); time.sleep(30)', - str(pidfile), str(worker.pid)] - started = time.monotonic() - assert server_watch.run(state, command, 0.02) == 1 - assert time.monotonic() - started < 5 - assert wrapper.poll() is None - assert pidfile.exists() - observed = server_watch.process_state(int(pidfile.read_text())) - assert observed is None or observed[0].startswith('Z') - finally: - wrapper.terminate() - if worker.poll() is None: - worker.terminate() - - -def test_exited_client_leader_does_not_leave_owned_worker(tmp_path: Path): - pidfile = tmp_path / 'worker.pid' - with subprocess.Popen([sys.executable, '-c', 'import time; time.sleep(30)']) as server: - try: - command = [sys.executable, '-c', - 'import pathlib,subprocess,sys; p=subprocess.Popen([sys.executable,"-c","import time; time.sleep(30)"]); pathlib.Path(sys.argv[1]).write_text(str(p.pid))', - str(pidfile)] - assert server_watch.run(server_watch.snapshot(server.pid), command, 0.02) == 0 - deadline = time.monotonic() + 3 - while time.monotonic() < deadline: - state = server_watch.process_state(int(pidfile.read_text())) - if state is None or state[0].startswith('Z'): - break - time.sleep(0.02) - else: - pytest.fail('Client exited but its owned worker survived') - assert server.poll() is None - finally: - server.terminate() - - -def test_cleanup_accepts_a_group_with_only_an_exited_process(): - with subprocess.Popen([sys.executable, '-c', 'pass'], start_new_session=True) as client: - deadline = time.monotonic() + 3 - while time.monotonic() < deadline: - state = server_watch.process_state(client.pid) - if state and state[0].startswith('Z'): - break - time.sleep(0.01) - else: - pytest.fail('Client did not exit') - server_watch.stop(client) - assert client.wait() == 0 - - -@pytest.mark.parametrize('leader_exits', [False, True]) -def test_cleanup_does_not_hide_permission_denial_for_a_live_group(monkeypatch, leader_exits): - command = ('import subprocess,sys; subprocess.Popen([sys.executable,"-c","import time; time.sleep(30)"])' - if leader_exits else 'import time; time.sleep(30)') - killpg = os.killpg - with subprocess.Popen([sys.executable, '-c', command], start_new_session=True) as client: - def denied(*args): - raise PermissionError('denied') - try: - if leader_exits: - client.wait(timeout=3) - monkeypatch.setattr(os, 'killpg', denied) - with pytest.raises(PermissionError, match='denied'): - server_watch.stop(client) - if not leader_exits: - assert client.poll() is None - finally: - killpg(client.pid, signal.SIGKILL) - - -def test_cleanup_kills_a_client_that_ignores_termination(monkeypatch, tmp_path): - ready = tmp_path / 'ready' - command = 'import signal,pathlib,sys,time; signal.signal(signal.SIGTERM,signal.SIG_IGN); pathlib.Path(sys.argv[1]).touch(); time.sleep(30)' - with subprocess.Popen([sys.executable, '-c', command, str(ready)], start_new_session=True) as client: - try: - deadline = time.monotonic() + 3 - while not ready.exists(): - assert time.monotonic() < deadline, 'Client did not become ready' - time.sleep(0.01) - wait = client.wait - monkeypatch.setattr(client, 'wait', lambda timeout=None: wait(0.05 if timeout is not None else None)) - server_watch.stop(client) - assert client.returncode == -signal.SIGKILL - finally: - if client.poll() is None: - client.kill() diff --git a/inferencex-e2e/infx/tests/evals/test_batched_eval.py b/inferencex-e2e/infx/tests/evals/test_batched_eval.py index df827815ce..1316204ea6 100644 --- a/inferencex-e2e/infx/tests/evals/test_batched_eval.py +++ b/inferencex-e2e/infx/tests/evals/test_batched_eval.py @@ -1,8 +1,6 @@ -"""Tests for batched multi-node eval runtime and validation.""" +"""Tests for batched eval score and manifest validation.""" import json -import os -import subprocess import sys from pathlib import Path @@ -10,116 +8,6 @@ from infx.evals.validate_scores import validate_batch_manifest -def _run_batched_eval( - tmp_path: Path, - *, - failing_conc: str = "", -) -> dict: - benchmark_lib = ( - Path(__file__).resolve().parents[3] / "benchmarks" / "benchmark_lib.sh" - ) - trace_path = tmp_path / "eval_concs.txt" - env = { - **os.environ, - "BENCHMARK_LIB": str(benchmark_lib), - "TRACE_PATH": str(trace_path), - "FAILING_CONC": failing_conc, - "IS_AGENTIC": "0", - } - script = r''' -source "$BENCHMARK_LIB" - -run_lm_eval() { - local results_dir="" - while [[ $# -gt 0 ]]; do - case "$1" in - --results-dir) results_dir="$2"; shift 2 ;; - *) shift ;; - esac - done - - mkdir -p "$results_dir/nested" - printf '%s\n' "$EVAL_CONCURRENT_REQUESTS" >> "$TRACE_PATH" - printf '{"lm_eval_version":"0.4.0"}' \ - > "$results_dir/nested/results_test.json" - printf '{"sample":true}\n' \ - > "$results_dir/nested/samples_test.jsonl" - if [ "$EVAL_CONCURRENT_REQUESTS" = "$FAILING_CONC" ]; then - return 7 - fi -} - -export EVAL_CONCURRENT_REQUESTS="1 4 8" -export EVAL_MAX_MODEL_LEN=4096 -export EVAL_ONLY=true -export MODEL=test-model -export MODEL_NAME=test-model -export MODEL_PREFIX=test -export RUNNER_TYPE=gb200 -export FRAMEWORK=dynamo-sglang -export PRECISION=fp8 -export SPEC_DECODING=none -export IS_MULTINODE=true -export ISL=8192 -export OSL=1024 -export PREFILL_TP=4 -export PREFILL_EP=1 -export PREFILL_NUM_WORKERS=1 -export DECODE_TP=8 -export DECODE_EP=1 -export DECODE_NUM_WORKERS=2 - -run_eval --framework lm-eval --port 30000 -export CONC="$EVAL_CONCURRENT_REQUESTS" -append_lm_eval_summary -''' - subprocess.run( - ["bash", "-c", script], - cwd=tmp_path, - env=env, - check=True, - text=True, - capture_output=True, - ) - - assert trace_path.read_text().splitlines() == ["1", "4", "8"] - return json.loads((tmp_path / "meta_env.json").read_text()) - - -def test_batched_eval_runs_every_concurrency_and_stages_results( - tmp_path: Path, -) -> None: - meta = _run_batched_eval(tmp_path) - - assert meta["eval_concs"] == [1, 4, 8] - assert meta["completed_eval_concs"] == [1, 4, 8] - assert meta["failed_eval_concs"] == [] - assert sorted(path.name for path in tmp_path.glob("results*.json")) == [ - "results_test_conc1.json", - "results_test_conc4.json", - "results_test_conc8.json", - ] - assert validate_batch_manifest( - str(tmp_path / "meta_env.json"), - [str(path) for path in tmp_path.glob("results*.json")], - ) == [] - - -def test_batched_eval_preserves_partial_results_and_records_failure( - tmp_path: Path, -) -> None: - meta = _run_batched_eval(tmp_path, failing_conc="4") - - assert meta["completed_eval_concs"] == [1, 8] - assert meta["failed_eval_concs"] == [4] - errors = validate_batch_manifest( - str(tmp_path / "meta_env.json"), - [str(path) for path in tmp_path.glob("results*.json")], - ) - assert any("failed for concurrency: 4" in error for error in errors) - assert any("missing completed concurrency: 4" in error for error in errors) - - def test_batched_eval_requires_a_valid_manifest(tmp_path: Path) -> None: result_path = tmp_path / "results_test_conc4.json" result_path.write_text('{"lm_eval_version":"0.4.0"}') diff --git a/inferencex-e2e/infx/tests/evals/test_bfcl_eval.py b/inferencex-e2e/infx/tests/evals/test_bfcl_eval.py index e03b0742da..1990616cf3 100644 --- a/inferencex-e2e/infx/tests/evals/test_bfcl_eval.py +++ b/inferencex-e2e/infx/tests/evals/test_bfcl_eval.py @@ -1,10 +1,13 @@ import builtins +import hashlib import json import os import shutil import subprocess import sys +import threading from dataclasses import replace +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path from types import ModuleType from typing import Any @@ -660,6 +663,59 @@ def test_integration_error_cli_is_stdlib_only_and_returns_nonzero( assert native["integration_error"] == compatibility["integration_error"] +@pytest.mark.parametrize("verified", [True, False]) +def test_install_runtime_installs_only_the_verified_wheel( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], + verified: bool, +) -> None: + payload = b"pinned wheel bytes" + + class Wheel(BaseHTTPRequestHandler): + def do_GET(self) -> None: + self.send_response(200) + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args: object) -> None: + pass + + installs: list[tuple[list[str], bytes]] = [] + + def pip(argv: list[str], **kwargs: Any) -> None: + [wheel] = [arg for arg in argv if arg.endswith(".whl")] + installs.append((argv, Path(wheel).read_bytes())) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Wheel) + thread = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.01}) + thread.start() + url = f"http://127.0.0.1:{server.server_port}/bfcl_eval-2026.3.23-py3-none-any.whl" + monkeypatch.setattr(be, "BFCL_WHEEL_URL", url) + expected = payload if verified else b"different bytes" + monkeypatch.setattr(be, "BFCL_WHEEL_SHA256", hashlib.sha256(expected).hexdigest()) + monkeypatch.setattr(be.subprocess, "run", pip) + download_dir = tmp_path / "download" + try: + return_code = be.main(["--install-runtime", str(download_dir)]) + finally: + server.shutdown() + thread.join() + server.server_close() + + if verified: + assert return_code == 0 + [(argv, installed)] = installs + assert argv[:4] == [sys.executable, "-m", "pip", "install"] + assert installed == payload + else: + assert return_code == 1 + assert installs == [] + assert list(download_dir.iterdir()) == [] + assert "SHA256 mismatch" in capsys.readouterr().err + + def test_full_suite_sets_project_root_before_dataset_import( monkeypatch, tmp_path: Path ) -> None: diff --git a/inferencex-e2e/infx/tests/evals/test_kimi_verifier_archive.py b/inferencex-e2e/infx/tests/evals/test_kimi_verifier_archive.py new file mode 100644 index 0000000000..70d05dc9f3 --- /dev/null +++ b/inferencex-e2e/infx/tests/evals/test_kimi_verifier_archive.py @@ -0,0 +1,164 @@ +"""``_kimi_verifier_archive.py`` against a local archive server.""" + +from __future__ import annotations + +import hashlib +import io +import sys +import tarfile +import threading +from collections.abc import Iterator +from contextlib import contextmanager +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +import pytest + +from infx.evals import _kimi_verifier_archive as archive + +REF = "1" * 40 +REQUIRED_FILES = { + "pyproject.toml", + "tests/conftest.py", + "tests/__init__.py", + "tests/tool_call_json_schema/conftest.py", + "tests/tool_call_json_schema/__init__.py", + "tests/tool_call_json_schema/test_tool_call_json_schema.py", + "tests/tool_call_json_schema/validator.py", + *{ + f"testdata/walle_validator_cases/validator_cases/{case}/valid.jsonl" + for case in ( + "TestAdditionalProperties", + "TestAnyOf", + "TestBasicTypes", + "TestDefs", + "TestDescription", + "TestEnforcerCases", + "TestID", + "TestKeywordsValidation", + "TestNestedDefsDepth", + "TestNumberFormat", + "TestRangeConstraints", + "TestRefInProperties", + "TestReferences", + "TestRequired", + "TestSingleTypeInArray", + "TestTypeLocation", + ) + }, +} + + +def _archive(*, missing: str | None = None, extra: tarfile.TarInfo | None = None) -> bytes: + output = io.BytesIO() + with tarfile.open(fileobj=output, mode="w:gz") as tar: + for relative_path in sorted(REQUIRED_FILES - {missing}): + payload = relative_path.encode() + member = tarfile.TarInfo(f"verifier-pinned/{relative_path}") + member.size = len(payload) + tar.addfile(member, io.BytesIO(payload)) + readme = tarfile.TarInfo("verifier-pinned/README.md") + readme.size = len(b"not selected") + tar.addfile(readme, io.BytesIO(b"not selected")) + if extra is not None: + tar.addfile(extra, io.BytesIO(b"x" * extra.size)) + return output.getvalue() + + +@contextmanager +def _serve(payload: bytes, *, transient_failures: int = 0) -> Iterator[tuple[str, list[str]]]: + paths: list[str] = [] + + class Handler(BaseHTTPRequestHandler): + def do_GET(self) -> None: + paths.append(self.path) + if len(paths) <= transient_failures: + self.send_response(503) + self.end_headers() + return + self.send_response(200) + self.send_header("Content-Length", str(len(payload))) + self.end_headers() + self.wfile.write(payload) + + def log_message(self, *args: object) -> None: + pass + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.01}) + thread.start() + try: + yield f"http://127.0.0.1:{server.server_port}/owner/verifier.git", paths + finally: + server.shutdown() + thread.join() + server.server_close() + + +def _fetch( + monkeypatch: pytest.MonkeyPatch, + checkout: Path, + payload: bytes, + *, + sha256: str | None = None, + transient_failures: int = 0, +) -> list[str]: + """Fetch ``payload`` from a local archive server into ``checkout``; return the paths asked.""" + checkout.mkdir() + monkeypatch.setattr(archive.time, "sleep", lambda _: None) + with _serve(payload, transient_failures=transient_failures) as (repo_url, paths): + digest = sha256 or hashlib.sha256(payload).hexdigest() + monkeypatch.setattr(sys, "argv", ["archive", repo_url, REF, digest, str(checkout)]) + archive.main() + return paths + + +def test_extracts_only_the_required_subset_after_a_transient_server_error( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + checkout = tmp_path / "checkout" + + paths = _fetch(monkeypatch, checkout, _archive(), transient_failures=1) + + assert paths == [f"/owner/verifier/archive/{REF}.tar.gz"] * 2 + assert "archive download attempt 1/3 failed" in capsys.readouterr().err + extracted = {p.relative_to(checkout).as_posix() for p in checkout.rglob("*") if p.is_file()} + assert extracted == REQUIRED_FILES + assert (checkout / "pyproject.toml").read_text() == "pyproject.toml" + + +def _escaping_member() -> tarfile.TarInfo: + member = tarfile.TarInfo("verifier-pinned/../../escaped") + member.size = 6 + return member + + +@pytest.mark.parametrize( + ("payload", "sha256", "error"), + [ + (_archive(), "0" * 64, "archive SHA256 mismatch"), + ( + _archive(missing="tests/tool_call_json_schema/validator.py"), + None, + "tests/tool_call_json_schema/validator.py", + ), + (_archive(extra=_escaping_member()), None, "unsafe archive member path"), + ], +) +def test_rejected_archive_extracts_nothing( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], + payload: bytes, + sha256: str | None, + error: str, +) -> None: + checkout = tmp_path / "checkout" + + with pytest.raises(SystemExit) as exited: + _fetch(monkeypatch, checkout, payload, sha256=sha256) + + assert exited.value.code == 1 + assert error in capsys.readouterr().err + assert list(checkout.iterdir()) == [] + assert not (tmp_path / "escaped").exists() diff --git a/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py b/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py deleted file mode 100644 index 05095caab5..0000000000 --- a/inferencex-e2e/infx/tests/evals/test_run_eval_dispatch.py +++ /dev/null @@ -1,2572 +0,0 @@ -from __future__ import annotations - -import hashlib -import io -import json -import os -import signal -import stat -import subprocess -import sys -import tarfile -import threading -from contextlib import contextmanager -from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from pathlib import Path - -import pytest -import yaml - -REPO_ROOT = Path(__file__).resolve().parents[3] -BENCHMARK_LIB = REPO_ROOT / "benchmarks" / "benchmark_lib.sh" -MULTINODE_AGENTIC_SCRIPT = REPO_ROOT / "benchmarks/srt_agentic.sh" - - -@pytest.mark.parametrize("use_model_path", [False, True]) -def test_local_context_does_not_require_registered_transformers_model(tmp_path, use_model_path): - model = tmp_path / "new model's weights" - model.mkdir() - (model / "config.json").write_text( - json.dumps( - { - "model_type": "not_registered_yet", - "max_position_embeddings": 1048576, - "seq_length": 4096, - } - ) - ) - (tmp_path / "transformers.py").write_text('raise RuntimeError("model not registered")\n') - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "PYTHONPATH": str(tmp_path), - "MODEL_PATH": str(model) if use_model_path else "", - "MODEL_ARG": "served-alias" if use_model_path else str(model), - "KV_OFFLOADING": "none", - } - result = subprocess.run( - [ - "bash", - "-c", - 'source "$BENCHMARK_LIB"; get_native_max_context_length "$MODEL_ARG"', - ], - env=env, - capture_output=True, - text=True, - check=True, - ) - assert result.stdout.strip() == "1048576" - - -@pytest.fixture(autouse=True) -def explicit_runtime_inputs(monkeypatch: pytest.MonkeyPatch) -> None: - """Provide explicit caller inputs before each case applies its overrides.""" - for name, value in { - "EVAL_ONLY": "false", - "IS_AGENTIC": "0", - "IS_MULTINODE": "false", - "PORT": "8888", - "CONC": "64", - "VENDOR_VERIFIER_PYTHON": "python3", - "INFMAX_CONTAINER_WORKSPACE": str(REPO_ROOT), - "AIPERF_PYTHON_VERSION": "3.11", - "AIPERF_DRAIN_TIMEOUT_SECONDS": "120", - "AIPERF_DRAIN_POLL_SECONDS": "1", - "SGLANG_TORCH_PROFILER_DIR": "/workspace", - "VLLM_TORCH_PROFILER_DIR": "/workspace", - "EVAL_ENDPOINT_READY_TIMEOUT_SECONDS": "1800", - "EVAL_MODEL_STABILIZATION_SECONDS": "0", - "OPENAI_API_KEY": "EMPTY", - }.items(): - monkeypatch.setenv(name, value) - - -_SCRIPT = r""" -source "$BENCHMARK_LIB" -_wait_for_openai_chat_route() { echo "READY=$*"; } -run_lm_eval() { echo "DISPATCH=lm-eval"; } -run_kimi_vendor_eval() { echo "DISPATCH=kimi-vendor"; } -run_minimax_vendor_eval() { echo "DISPATCH=minimax-vendor"; } -run_bfcl_eval() { echo "DISPATCH=bfcl"; } -append_lm_eval_summary() { echo "STAGED=summary"; } -export EVAL_MAX_MODEL_LEN=16384 -export EVAL_CONCURRENT_REQUESTS="" -run_eval ${CLI_FW:+--framework "$CLI_FW"} --port 8888 -""" - - -def _dispatch( - *, - is_agentic: str = "0", - eval_only: str = "false", - cli_fw=None, - env_fw=None, -) -> str: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "IS_AGENTIC": is_agentic, - "EVAL_ONLY": eval_only, - "KV_OFFLOADING": "none", - } - env.pop("EVAL_FRAMEWORK", None) - env.pop("CLI_FW", None) - env.pop("KV_OFFLOAD_BACKEND", None) - env.pop("EVAL_SUITE", None) - if cli_fw is not None: - env["CLI_FW"] = cli_fw - if env_fw is not None: - env["EVAL_FRAMEWORK"] = env_fw - res = subprocess.run( - ["bash", "-c", _SCRIPT], env=env, text=True, capture_output=True, check=True - ) - return res.stdout - - -def test_agentic_dependency_install_is_rootless_without_git(tmp_path: Path) -> None: - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - fake_uv = fake_bin / "uv" - fake_uv.write_text( - "#!/bin/bash\n" - 'printf "%s\\n" "$@" >> "$UV_CALLS"\n' - "if [[ \"$1\" == \"venv\" ]]; then\n" - " target=\"${@: -1}\"\n" - " mkdir -p \"$target/bin\"\n" - " printf '#!/bin/sh\\nexit 0\\n' > \"$target/bin/aiperf\"\n" - " printf '#!/bin/sh\\nexit 0\\n' > \"$target/bin/hf\"\n" - " chmod +x \"$target/bin/aiperf\" \"$target/bin/hf\"\n" - "fi\n" - ) - fake_uv.chmod(0o755) - fake_apt = fake_bin / "apt-get" - fake_apt.write_text("#!/bin/sh\nexit 97\n") - fake_apt.chmod(0o755) - fake_git = fake_bin / "git" - fake_git.write_text("#!/bin/sh\nexit 97\n") - fake_git.chmod(0o755) - - env = { - **os.environ, - "INFMAX_CONTAINER_WORKSPACE": str(tmp_path / "checkout with spaces"), - "UV_CALLS": str(tmp_path / "uv-calls"), - "AIPERF_RUNTIME_DIR": str(tmp_path / "runtime"), - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "FAKE_UV": str(fake_uv), - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONPYCACHEPREFIX": str(tmp_path / "pycache"), - } - result = subprocess.run( - [ - "/bin/bash", - "-c", - 'source "$BENCHMARK_LIB"; ' - 'ensure_agentic_uv() { AIPERF_UV_BIN="$FAKE_UV"; }; ' - "install_agentic_deps", - ], - env=env, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert (tmp_path / "runtime/venv/bin/aiperf").is_file() - assert (tmp_path / "runtime/venv/bin/hf").is_file() - install_args = (tmp_path / "uv-calls").read_text().splitlines() - assert install_args[:4] == [ - "venv", - "--python", - "3.11", - str(tmp_path / "runtime/venv"), - ] - assert install_args[4:10] == [ - "pip", - "install", - "--python", - str(tmp_path / "runtime/venv/bin/python"), - "-e", - str(tmp_path / "checkout with spaces/utils/aiperf"), - ] - assert set(install_args[10:]) == { - "numpy>=1.24", - "pandas>=2.0.0", - "aiohttp>=3.10", - "transformers>=4.46", - "xlsxwriter>=3.2.1", - "tqdm>=4.66", - "datasets>=4.7.0", - "tiktoken", - "huggingface_hub[cli]>=0.25.0", - "urllib3", - "requests", - } - - -def test_agentic_scenario_defaults_to_gsm8k_lm_eval(): - assert "DISPATCH=lm-eval" in _dispatch(is_agentic="1") - - -def test_agentic_eval_only_stages_summary(): - output = _dispatch(is_agentic="1", eval_only="true") - assert "DISPATCH=lm-eval" in output - assert "STAGED=summary" in output - - -def test_fixed_seqlen_eval_only_leaves_staging_to_recipe(): - assert "STAGED=summary" not in _dispatch(is_agentic="0", eval_only="true") - - -def test_fixed_seqlen_provider_leaves_staging_to_recipe() -> None: - output = _dispatch( - is_agentic="0", - eval_only="true", - env_fw="minimax-vendor", - ) - assert "DISPATCH=minimax-vendor" in output - assert "STAGED=summary" not in output - - -def test_environment_framework_overrides_legacy_recipe_argument() -> None: - assert "DISPATCH=kimi-vendor" in _dispatch( - is_agentic="1", - cli_fw="bfcl", - env_fw="kimi-vendor", - ) - - -def test_kimi_vendor_skips_unused_model_context_loading() -> None: - script = r""" -source "$BENCHMARK_LIB" -unset EVAL_MAX_MODEL_LEN -compute_eval_context_length() { echo "UNEXPECTED_CONTEXT_LOAD"; return 99; } -run_kimi_vendor_eval() { echo "DISPATCH=kimi-vendor"; } -export EVAL_FRAMEWORK=kimi-vendor -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_ONLY=false -export IS_AGENTIC=0 -run_eval --port 8888 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "DISPATCH=kimi-vendor" in result.stdout - assert "UNEXPECTED_CONTEXT_LOAD" not in result.stdout - - -def test_kimi_failure_preserves_rc_when_eval_only_is_false() -> None: - script = r""" -source "$BENCHMARK_LIB" -run_kimi_vendor_eval() { return 7; } -export EVAL_FRAMEWORK=kimi-vendor -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_MAX_MODEL_LEN=16384 -export IS_AGENTIC=0 -export EVAL_ONLY=false -run_eval --port 8888 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 7 - assert "unbound variable" not in result.stderr - - -def test_run_eval_rejects_missing_eval_mode() -> None: - result = subprocess.run( - ["bash", "-c", 'source "$BENCHMARK_LIB"; unset EVAL_ONLY; run_eval'], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - ) - assert result.returncode == 1 - assert " - EVAL_ONLY" in result.stdout - - -def test_validation_only_source_does_not_initialize_runtime(tmp_path: Path) -> None: - cache_dir = tmp_path / "pycache" - result = subprocess.run( - ["bash", "-c", 'source "$BENCHMARK_LIB" --validation-only; ' - 'unset MISSING_INPUT; check_env_vars MISSING_INPUT'], - env={ - "PATH": os.environ["PATH"], - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "PYTHONPYCACHEPREFIX": str(cache_dir), - }, - text=True, - capture_output=True, - ) - assert result.returncode == 1 - assert " - MISSING_INPUT" in result.stdout - assert not cache_dir.exists() - - -def _run_invalid_call(call: str) -> subprocess.CompletedProcess: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - } - return subprocess.run( - ["bash", "-c", f'source "$BENCHMARK_LIB"; {call}'], - env=env, - text=True, - capture_output=True, - ) - - -def test_run_eval_rejects_missing_framework_value(): - result = _run_invalid_call("run_eval --framework") - assert result.returncode == 2 - assert "--framework requires a value" in result.stderr - - -def test_run_eval_rejects_unsafe_suite_name() -> None: - result = _run_invalid_call( - "EVAL_SUITE='kimi\"suite' run_eval --framework kimi-vendor" - ) - - assert result.returncode == 2 - assert "EVAL_SUITE may contain only" in result.stderr - - -def test_run_eval_rejects_suite_override_for_lm_eval() -> None: - result = _run_invalid_call("EVAL_SUITE=gpqa_diamond run_eval --framework lm-eval") - - assert result.returncode == 2 - assert "only supported with kimi-vendor, minimax-vendor, or bfcl" in result.stderr - - -def test_run_eval_scopes_runner_selected_suite_to_one_call() -> None: - script = r""" -source "$BENCHMARK_LIB" -run_kimi_vendor_eval() { - export EVAL_SUITE=kimi_tool_call_schema - echo "DISPATCH=kimi-vendor SUITE=$EVAL_SUITE" -} -run_lm_eval() { - echo "DISPATCH=lm-eval SUITE=${EVAL_SUITE:-unset} COMPLETED=${EVAL_COMPLETED_SUITE:-unset}" -} -export EVAL_MAX_MODEL_LEN=16384 -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_ONLY=false -export IS_AGENTIC=0 -unset EVAL_SUITE -export EVAL_FRAMEWORK=kimi-vendor -run_eval --port 8888 -printf 'KIMI_COMPLETED=%s\n' "${EVAL_COMPLETED_SUITE:-unset}" -export EVAL_FRAMEWORK=lm-eval -run_eval --port 8888 -printf 'LM_COMPLETED=%s\n' "${EVAL_COMPLETED_SUITE:-unset}" -printf 'FINAL_SUITE=%s\n' "${EVAL_SUITE-unset}" -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "DISPATCH=kimi-vendor SUITE=kimi_tool_call_schema" in result.stdout - assert "KIMI_COMPLETED=kimi_tool_call_schema" in result.stdout - assert "DISPATCH=lm-eval SUITE=unset COMPLETED=unset" in result.stdout - assert "LM_COMPLETED=unset" in result.stdout - assert "FINAL_SUITE=unset" in result.stdout - - -def test_kimi_default_suite_reaches_eval_only_metadata() -> None: - script = r""" -source "$BENCHMARK_LIB" -_wait_for_openai_chat_route() { :; } -run_kimi_vendor_eval() { echo "DISPATCH=$EVAL_SUITE"; } -append_lm_eval_summary() { echo "METADATA=$EVAL_COMPLETED_SUITE"; } -export EVAL_FRAMEWORK=kimi-vendor -export EVAL_ONLY=true -export IS_AGENTIC=1 -export EVAL_CONCURRENT_REQUESTS="" -unset EVAL_SUITE -run_eval --port 8888 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "DISPATCH=kimi_tool_call_schema" in result.stdout - assert "METADATA=kimi_tool_call_schema" in result.stdout - - -def test_agentic_eval_propagates_artifact_staging_failure() -> None: - script = r""" -source "$BENCHMARK_LIB" -_wait_for_openai_chat_route() { :; } -run_kimi_vendor_eval() { :; } -append_lm_eval_summary() { return 73; } -export EVAL_FRAMEWORK=kimi-vendor -export EVAL_ONLY=true -export IS_AGENTIC=1 -export EVAL_CONCURRENT_REQUESTS="" -unset EVAL_SUITE -run_eval --port 8888 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 73 - assert "eval artifact staging failed with exit code 73" in result.stderr - - -def test_minimax_full_suite_dispatches_to_full_runner() -> None: - script = r""" -source "$BENCHMARK_LIB" -_run_minimax_m3_full_eval() { - printf 'DISPATCH=%s ARGS=<%s>\n' "$EVAL_SUITE" "$*" -} -EVAL_SUITE=minimax_m3_full run_minimax_vendor_eval --port 9999 -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "DISPATCH=minimax_m3_full ARGS=<--port 9999>" in result.stdout - - -def test_kimi_vendor_rejects_batched_concurrency() -> None: - result = _run_invalid_call( - "EVAL_MAX_MODEL_LEN=16384 " - "EVAL_CONCURRENT_REQUESTS='1 4' " - "run_eval --framework kimi-vendor" - ) - assert result.returncode == 1 - assert "batched eval concurrency is only supported for lm-eval" in result.stderr - - -def test_kimi_vendor_rejects_unsupported_suite() -> None: - result = _run_invalid_call("EVAL_SUITE=gsm8k run_kimi_vendor_eval") - assert result.returncode == 2 - assert "unsupported Kimi Vendor Verifier suite 'gsm8k'" in result.stderr - - -def _run_minimax_dispatch(*, suite: str | None = None, concurrency: str = "") -> str: - script = r""" -source "$BENCHMARK_LIB" -unset EVAL_MAX_MODEL_LEN -compute_eval_context_length() { echo "UNEXPECTED_CONTEXT_LOAD"; return 99; } -MINIMAX_DISPATCH_COUNT=0 -run_minimax_vendor_eval() { - MINIMAX_DISPATCH_COUNT=$((MINIMAX_DISPATCH_COUNT + 1)) - printf 'DISPATCH=minimax-vendor SUITE=%s ARGS=<%s>\n' "$EVAL_SUITE" "$*" -} -export EVAL_CONCURRENT_REQUESTS="$TEST_EVAL_CONCURRENCY" -export EVAL_ONLY=false -export IS_AGENTIC=0 -run_eval --framework minimax-vendor --port 9999 -printf 'DISPATCH_COUNT=%s\n' "$MINIMAX_DISPATCH_COUNT" -printf 'COMPLETED_SUITE=%s\n' "$EVAL_COMPLETED_SUITE" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "MODEL": "served-model", - "MODEL_PREFIX": "minimaxm3", - "TEST_EVAL_CONCURRENCY": concurrency, - } - for key in ("EVAL_FRAMEWORK", "EVAL_SUITE", "EVAL_COMPLETED_SUITE"): - env.pop(key, None) - if suite is not None: - env["EVAL_SUITE"] = suite - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - assert "UNEXPECTED_CONTEXT_LOAD" not in result.stdout - return result.stdout - - -def test_minimax_vendor_defaults_suite_dispatches_once_and_records_completion() -> None: - output = _run_minimax_dispatch() - - assert "DISPATCH=minimax-vendor SUITE=minimax_m3_smoke" in output - assert "ARGS=<--port 9999>" in output - assert "DISPATCH_COUNT=1" in output - assert "COMPLETED_SUITE=minimax_m3_smoke" in output - - -def test_minimax_vendor_accepts_explicit_supported_suite() -> None: - output = _run_minimax_dispatch(suite="minimax_m3_smoke") - - assert "DISPATCH=minimax-vendor SUITE=minimax_m3_smoke" in output - assert "DISPATCH_COUNT=1" in output - assert "COMPLETED_SUITE=minimax_m3_smoke" in output - - -def test_minimax_vendor_rejects_unsupported_suite() -> None: - result = _run_invalid_call( - "MODEL_PREFIX=minimaxm3 EVAL_SUITE=gsm8k run_eval --framework minimax-vendor" - ) - - assert result.returncode == 2 - assert "unsupported MiniMax Provider Verifier suite 'gsm8k'" in result.stderr - - -def test_run_eval_rejects_unknown_framework() -> None: - result = _run_invalid_call( - "EVAL_MAX_MODEL_LEN=16384 run_eval --framework not-a-framework" - ) - - assert result.returncode == 1 - assert "Unknown framework 'not-a-framework'" in result.stdout - - -def test_minimax_vendor_rejects_concurrency_sweep_for_sequential_smoke() -> None: - result = _run_invalid_call( - "MODEL_PREFIX=minimaxm3 " - "EVAL_CONCURRENT_REQUESTS='1 4' " - "run_eval --framework minimax-vendor" - ) - - assert result.returncode == 1 - assert "batched eval concurrency is only supported for lm-eval" in result.stderr - - -def test_minimax_vendor_ignores_single_launcher_concurrency_value() -> None: - output = _run_minimax_dispatch(concurrency="128") - - assert "DISPATCH=minimax-vendor SUITE=minimax_m3_smoke" in output - assert "DISPATCH_COUNT=1" in output - - -def test_minimax_vendor_accepts_non_m3_model() -> None: - script = r""" -source "$BENCHMARK_LIB" -_run_minimax_m3_smoke_eval() { echo "DISPATCH=$EVAL_SUITE"; } -unset EVAL_SUITE EVAL_RESULT_DIR -MODEL=moonshotai/Kimi-K3 MODEL_PREFIX=kimik3 run_minimax_vendor_eval -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=True, - ) - - assert "DISPATCH=minimax_m3_smoke" in result.stdout - - -def test_minimax_vendor_setup_failure_uses_integration_error_and_stages( - tmp_path: Path, -) -> None: - script = r""" -source "$BENCHMARK_LIB" -unset EVAL_SUITE EVAL_RESULT_DIR EVAL_COMPLETED_SUITE -unset VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -_prepare_vendor_verifier_python() { - if compgen -G "$RESULTS_DIR/results_minimax_vendor_*.json" >/dev/null \ - || [ -e "$RESULTS_DIR/minimax_vendor_report.json" ]; then - echo "STALE_MINIMAX_ARTIFACT" - return 99 - fi - mkdir -p "$PYTHON_DIR" - cat >"$PYTHON_DIR/python3" <<'PY' -#!/bin/bash -printf 'ADAPTER_ARG=<%s>\n' "$@" -PY - chmod +x "$PYTHON_DIR/python3" - export VENDOR_VERIFIER_PYTHON="$PYTHON_DIR/python3" - export VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" -} -_prepare_minimax_m3_full_runtime() { return 12; } -append_lm_eval_summary() { - printf 'STAGED=<%s>\n' "$EVAL_RESULT_DIR" - printf 'STAGED_CONC=<%s>\n' "$CONC" -} -export MODEL_PREFIX=minimaxm3 -export MODEL=test-model -export EVAL_CONCURRENT_REQUESTS=7 -export EVAL_ONLY=false -export IS_AGENTIC=0 -run_eval --framework minimax-vendor --results-dir "$RESULTS_DIR" -eval_rc=$? -printf 'EVAL_RC=%s\n' "$eval_rc" -""" - results_dir = tmp_path / "results" - results_dir.mkdir() - (results_dir / "results_minimax_vendor_stale.json").write_text("{}") - (results_dir / "minimax_vendor_report.json").write_text("{}") - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "PYTHON_DIR": str(tmp_path / "python"), - }, - text=True, - capture_output=True, - check=True, - ) - output = result.stdout + result.stderr - - assert "EVAL_RC=12" in output - assert "STALE_MINIMAX_ARTIFACT" not in output - assert not (tmp_path / "python").exists() - assert ( - f"ADAPTER_ARG=<{REPO_ROOT / 'infx/evals/minimax_provider_eval.py'}>" in output - ) - assert "ADAPTER_ARG=" in output - assert f"ADAPTER_ARG=<{results_dir}>" in output - assert "ADAPTER_ARG=" in output - assert "ADAPTER_ARG=<--message>" in output - assert ( - "ADAPTER_ARG=" - ) in output - assert f"STAGED=<{results_dir}>" in output - assert output.count("STAGED=<") == 1 - assert "STAGED_CONC=<7>" in output - - -def test_minimax_full_dependency_install_is_isolated( - tmp_path: Path, -) -> None: - script = r""" -source "$BENCHMARK_LIB" -selected_python() { printf 'PYTHON_ARG=<%s>\n' "$@"; } -VENDOR_VERIFIER_PYTHON=selected_python -_install_minimax_m3_full_deps "$RUNTIME_DIR" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RUNTIME_DIR": str(tmp_path / "runtime"), - }, - text=True, - capture_output=True, - check=True, - ) - - args = [ - line.removeprefix("PYTHON_ARG=<").removesuffix(">") - for line in result.stdout.splitlines() - if line.startswith("PYTHON_ARG=<") - ] - assert args[:3] == ["-m", "pip", "install"] - assert args[args.index("--target") + 1] == str(tmp_path / "runtime") - assert "--break-system-packages" not in result.stdout - - -def test_minimax_runtime_prepares_source_with_full_adapter(tmp_path: Path) -> None: - runtime_dir = tmp_path / "runtime" - calls_path = tmp_path / "calls" - script = r""" -source "$BENCHMARK_LIB" -mktemp() { - mkdir -p "$RUNTIME_DIR" - printf '%s\n' "$RUNTIME_DIR" -} -selected_python() { - printf 'PYTHON_ARG=<%s>\n' "$@" >> "$CALLS_PATH" - mkdir -p "$RUNTIME_DIR/source" -} -_install_minimax_m3_full_deps() { mkdir -p "$1"; } -VENDOR_VERIFIER_PYTHON=selected_python -prepared_runtime=$(_prepare_minimax_m3_full_runtime) -printf 'RUNTIME=<%s>\n' "$prepared_runtime" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RUNTIME_DIR": str(runtime_dir), - "CALLS_PATH": str(calls_path), - }, - text=True, - capture_output=True, - check=True, - ) - calls = calls_path.read_text() - - assert f"PYTHON_ARG=<{REPO_ROOT / 'infx/evals/minimax_m3_full_eval.py'}>" in calls - assert "PYTHON_ARG=" in calls - assert f"PYTHON_ARG=<{runtime_dir / 'source'}>" in calls - assert "minimax_provider_eval.py" not in calls - assert f"RUNTIME=<{runtime_dir}>" in result.stdout - - -def test_minimax_vendor_runner_uses_fixed_adapter_contract(tmp_path: Path) -> None: - results_dir = tmp_path / "results" - runtime_dir = tmp_path / "runtime" - python_dir = tmp_path / "python" - script = r""" -source "$BENCHMARK_LIB" -selected_python() { - printf 'PYTHONPATH=<%s>\n' "$PYTHONPATH" >&2 - printf 'PYTHON_ARG=<%s>\n' "$@" >&2 -} -_prepare_vendor_verifier_python() { - mkdir "$PYTHON_DIR" - VENDOR_VERIFIER_PYTHON=selected_python - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_minimax_m3_full_runtime() { - mkdir -p "$RUNTIME_DIR/source" "$RUNTIME_DIR/deps" - printf '%s\n' "$RUNTIME_DIR" -} -mktemp() { echo "UNEXPECTED_DEFAULT_RESULTS_DIR" >&2; return 99; } -run_minimax_vendor_eval --port 9999 --results-dir "$RESULTS_DIR" -printf 'EVAL_SUITE=%s\n' "$EVAL_SUITE" -printf 'EVAL_RESULT_DIR=%s\n' "$EVAL_RESULT_DIR" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "RUNTIME_DIR": str(runtime_dir), - "PYTHON_DIR": str(python_dir), - "MODEL": "test-model", - "MODEL_PREFIX": "minimaxm3", - "OPENAI_API_KEY": "must-not-be-forwarded", - } - for key in ("EVAL_SUITE", "EVAL_RESULT_DIR", "MODEL_NAME"): - env.pop(key, None) - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - output = result.stdout + result.stderr - adapter = REPO_ROOT / "infx/evals/minimax_provider_eval.py" - fixture = REPO_ROOT / "infx/evals/minimax_m3_smoke.json" - - for value in ( - adapter, - "run", - "selected_python", - runtime_dir / "source", - runtime_dir / "deps", - "http://127.0.0.1:9999/v1", - "test-model", - results_dir, - fixture, - ): - assert f"PYTHON_ARG=<{value}>" in output - for option in ( - "--python", - "--source-dir", - "--dependency-dir", - "--base-url", - "--model", - "--output-dir", - "--fixture", - ): - assert f"PYTHON_ARG=<{option}>" in output - assert "must-not-be-forwarded" not in output - assert "UNEXPECTED_DEFAULT_RESULTS_DIR" not in output - assert "EVAL_SUITE=minimax_m3_smoke" in output - assert f"EVAL_RESULT_DIR={results_dir}" in output - assert not runtime_dir.exists() - assert not python_dir.exists() - - -def test_kimi_vendor_setup_failure_writes_compatibility_result( - tmp_path: Path, -) -> None: - results_dir = tmp_path / "results" - results_dir.mkdir() - (results_dir / "results_kimi_vendor_stale.json").write_text("{}") - (results_dir / "kimi_vendor_report.json").write_text("{}") - python_dir = tmp_path / "python" - script = r""" -source "$BENCHMARK_LIB" -_prepare_vendor_verifier_python() { - if compgen -G "$RESULTS_DIR/results_kimi_vendor_*.json" >/dev/null \ - || [ -e "$RESULTS_DIR/kimi_vendor_report.json" ]; then - echo "STALE_KIMI_ARTIFACT" - return 99 - fi - mkdir "$PYTHON_DIR" - cat >"$PYTHON_DIR/python3" <<'PY' -#!/bin/bash -exec /usr/bin/env python3 "$@" -PY - chmod +x "$PYTHON_DIR/python3" - VENDOR_VERIFIER_PYTHON="$PYTHON_DIR/python3" - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_kimi_vendor_runtime() { return 12; } -run_kimi_vendor_eval --results-dir "$RESULTS_DIR" -printf 'SETUP_RC=%s\n' "$?" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "PYTHON_DIR": str(python_dir), - "MODEL": "test-model", - "IS_MULTINODE": "false", - "KV_OFFLOADING": "none", - } - for key in ( - "EVAL_SUITE", - "EVAL_RESULT_DIR", - "MODEL_NAME", - ): - env.pop(key, None) - - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - message = "Kimi Vendor Verifier dependency installation failed with exit code 12" - score_files = list(results_dir.glob("results*.json")) - - assert "SETUP_RC=12" in result.stdout - assert "STALE_KIMI_ARTIFACT" not in result.stdout + result.stderr - assert message in result.stderr - assert "failed to write Kimi verifier failure artifact" not in result.stderr - assert len(score_files) == 1 - score_result = json.loads(score_files[0].read_text()) - assert ( - score_result["results"]["kimi_tool_call_schema"]["exact_match,strict-match"] - == 0.0 - ) - assert score_result["integration_error"]["message"] == message - native_result = json.loads((results_dir / "kimi_vendor_report.json").read_text()) - assert native_result["completed"] is False - assert native_result["integration_error"]["message"] == message - assert not python_dir.exists() - - -def test_preclear_failure_cannot_stage_stale_provider_result(tmp_path: Path) -> None: - results_dir = tmp_path / "results" - results_dir.mkdir() - stale_result = results_dir / "results_kimi_vendor_stale.json" - stale_result.write_text('{"stale": true}\n') - work_dir = tmp_path / "work" - work_dir.mkdir() - script = r""" -source "$BENCHMARK_LIB" -rm() { return 73; } -append_lm_eval_summary() { - if [ -n "${EVAL_RESULT_DIR:-}" ]; then - echo "UNEXPECTED_STAGING" - fi - return 1 -} -unset EVAL_FRAMEWORK EVAL_SUITE EVAL_RESULT_DIR EVAL_COMPLETED_SUITE -export MODEL=test-model -cd "$WORK_DIR" -run_eval --framework kimi-vendor --results-dir "$RESULTS_DIR" -printf 'EVAL_RC=%s\n' "$?" -printf 'EVAL_RESULT_DIR=<%s>\n' "${EVAL_RESULT_DIR:-}" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "WORK_DIR": str(work_dir), - }, - text=True, - capture_output=True, - check=True, - ) - - assert "EVAL_RC=73" in result.stdout - assert "EVAL_RESULT_DIR=<>" in result.stdout - assert "UNEXPECTED_STAGING" not in result.stdout - assert "failed to remove stale eval artifact" in result.stderr - assert stale_result.is_file() - assert list(work_dir.iterdir()) == [] - - -_KIMI_VERIFIER_REQUIRED_FILES = { - "pyproject.toml", - "tests/conftest.py", - "tests/__init__.py", - "tests/tool_call_json_schema/conftest.py", - "tests/tool_call_json_schema/__init__.py", - "tests/tool_call_json_schema/test_tool_call_json_schema.py", - "tests/tool_call_json_schema/validator.py", - *{ - f"testdata/walle_validator_cases/validator_cases/{case}/valid.jsonl" - for case in ( - "TestAdditionalProperties", - "TestAnyOf", - "TestBasicTypes", - "TestDefs", - "TestDescription", - "TestEnforcerCases", - "TestID", - "TestKeywordsValidation", - "TestNestedDefsDepth", - "TestNumberFormat", - "TestRangeConstraints", - "TestRefInProperties", - "TestReferences", - "TestRequired", - "TestSingleTypeInArray", - "TestTypeLocation", - ) - }, -} - - -def _kimi_verifier_archive( - *, - missing: str | None = None, - unsafe_member: tarfile.TarInfo | None = None, -) -> bytes: - output = io.BytesIO() - with tarfile.open(fileobj=output, mode="w:gz") as archive: - for relative_path in sorted(_KIMI_VERIFIER_REQUIRED_FILES - {missing}): - payload = relative_path.encode() - member = tarfile.TarInfo(f"verifier-pinned/{relative_path}") - member.size = len(payload) - archive.addfile(member, io.BytesIO(payload)) - extra = b"must not be extracted" - member = tarfile.TarInfo("verifier-pinned/README.md") - member.size = len(extra) - archive.addfile(member, io.BytesIO(extra)) - if unsafe_member is not None: - archive.addfile( - unsafe_member, - io.BytesIO(b"unsafe") if unsafe_member.isfile() else None, - ) - return output.getvalue() - - -@contextmanager -def _serve_archive(payload: bytes, *, transient_failures: int = 0): - request_paths = [] - request_count = 0 - - class ArchiveHandler(BaseHTTPRequestHandler): - def do_GET(self): - nonlocal request_count - request_paths.append(self.path) - request_count += 1 - if request_count <= transient_failures: - self.send_response(503) - self.end_headers() - return - self.send_response(200) - self.send_header("Content-Length", str(len(payload))) - self.end_headers() - self.wfile.write(payload) - - def log_message(self, *args): - pass - - server = ThreadingHTTPServer(("127.0.0.1", 0), ArchiveHandler) - thread = threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.01}) - thread.start() - try: - yield ( - f"http://127.0.0.1:{server.server_port}/owner/verifier.git", - request_paths, - ) - finally: - server.shutdown() - thread.join() - server.server_close() - - -def _prepare_local_kimi_verifier( - tmp_path: Path, - payload: bytes, - verifier_ref: str = "1" * 40, - transient_failures: int = 0, - archive_sha256: str | None = None, -) -> tuple[subprocess.CompletedProcess[str], Path, list[str]]: - checkout = tmp_path / "checkout" - script = r""" -source "$BENCHMARK_LIB" -git() { echo "git must not be invoked" >&2; return 127; } -mktemp() { mkdir "$CHECKOUT"; printf '%s\n' "$CHECKOUT"; } -_prepare_kimi_vendor_verifier "$REPO_URL" "$VERIFIER_REF" "$ARCHIVE_SHA256" -""" - with _serve_archive( - payload, - transient_failures=transient_failures, - ) as (repo_url, request_paths): - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "CHECKOUT": str(checkout), - "REPO_URL": repo_url, - "VERIFIER_REF": verifier_ref, - "ARCHIVE_SHA256": archive_sha256 or hashlib.sha256(payload).hexdigest(), - }, - cwd=tmp_path, - text=True, - capture_output=True, - ) - return result, checkout, request_paths - - -def test_kimi_vendor_verifier_fetches_expected_subset_without_git( - tmp_path: Path, -) -> None: - result, checkout, request_paths = _prepare_local_kimi_verifier( - tmp_path, - _kimi_verifier_archive(), - ) - verifier_ref = "1" * 40 - - assert result.returncode == 0, result.stderr - assert result.stdout.strip() == str(checkout) - assert request_paths == [ - f"/owner/verifier/archive/{verifier_ref}.tar.gz", - ] - assert { - path.relative_to(checkout).as_posix() - for path in checkout.rglob("*") - if path.is_file() - } == _KIMI_VERIFIER_REQUIRED_FILES - assert "git must not be invoked" not in result.stderr - - -def test_kimi_vendor_verifier_retries_transient_archive_failure( - tmp_path: Path, monkeypatch, capsys, -) -> None: - from infx.evals import _kimi_verifier_archive as archive - - payload = _kimi_verifier_archive() - verifier_ref = "1" * 40 - monkeypatch.setattr(archive.time, 'sleep', lambda _: None) - with _serve_archive(payload, transient_failures=1) as (repo_url, request_paths): - monkeypatch.setattr(sys, 'argv', ['archive', repo_url, verifier_ref, - hashlib.sha256(payload).hexdigest(), str(tmp_path)]) - archive.main() - assert request_paths == [ - f"/owner/verifier/archive/{verifier_ref}.tar.gz", - f"/owner/verifier/archive/{verifier_ref}.tar.gz", - ] - assert "archive download attempt 1/3 failed" in capsys.readouterr().err - assert (tmp_path / 'pyproject.toml').read_text() == 'pyproject.toml' - - -def test_kimi_vendor_verifier_rejects_archive_hash_mismatch( - tmp_path: Path, -) -> None: - result, checkout, _ = _prepare_local_kimi_verifier( - tmp_path, - _kimi_verifier_archive(), - archive_sha256="0" * 64, - ) - - assert result.returncode == 1 - assert "archive SHA256 mismatch" in result.stderr - assert not checkout.exists() - - -def test_kimi_vendor_verifier_removes_partial_checkout_when_member_missing( - tmp_path: Path, -) -> None: - missing = "tests/tool_call_json_schema/validator.py" - result, checkout, _ = _prepare_local_kimi_verifier( - tmp_path, - _kimi_verifier_archive(missing=missing), - ) - - assert result.returncode == 1 - assert missing in result.stderr - assert not checkout.exists() - - -def test_kimi_vendor_verifier_rejects_unsafe_archive_members(tmp_path: Path) -> None: - unsafe = tarfile.TarInfo("verifier-pinned/../../escaped") - unsafe.size = len(b"unsafe") - result, checkout, _ = _prepare_local_kimi_verifier( - tmp_path, - _kimi_verifier_archive(unsafe_member=unsafe), - ) - - assert result.returncode == 1 - assert "unsafe archive member path" in result.stderr - assert not checkout.exists() - assert not (tmp_path / "escaped").exists() - - -def test_kimi_vendor_uses_system_python_fast_path() -> None: - script = r""" -source "$BENCHMARK_LIB" -python3() { - printf 'SYSTEM_PYTHON_ARG=<%s>\n' "$@" - [ "$1" = "-c" ] -} -mktemp() { echo "UNEXPECTED_MKTEMP"; return 99; } -VENDOR_VERIFIER_PYTHON=/previous/python -VENDOR_VERIFIER_PYTHON_CLEANUP_DIR=/previous/runtime -_prepare_vendor_verifier_python "Kimi Vendor Verifier" "kimi-vendor-python" -printf 'SELECTED_PYTHON=<%s>\n' "$VENDOR_VERIFIER_PYTHON" -printf 'PYTHON_CLEANUP=<%s>\n' "$VENDOR_VERIFIER_PYTHON_CLEANUP_DIR" -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=True, - ) - - assert "SYSTEM_PYTHON_ARG=<-c>" in result.stdout - assert "SELECTED_PYTHON=" in result.stdout - assert "PYTHON_CLEANUP=<>" in result.stdout - assert "UNEXPECTED_MKTEMP" not in result.stdout - - -def test_kimi_vendor_bootstraps_isolated_python_and_cleans_it( - tmp_path: Path, -) -> None: - log_path = tmp_path / "bootstrap.log" - fake_uv = tmp_path / "fake-uv" - fake_uv.write_text( - r"""#!/usr/bin/env bash -printf 'UV_CACHE_DIR=<%s>\n' "$UV_CACHE_DIR" >> "$KIMI_LOG" -printf 'UV_PYTHON_INSTALL_DIR=<%s>\n' "$UV_PYTHON_INSTALL_DIR" >> "$KIMI_LOG" -printf 'UV_ARG=<%s>\n' "$@" >> "$KIMI_LOG" -venv_dir="${!#}" -mkdir -p "$venv_dir/bin" -cat > "$venv_dir/bin/python" <<'PYTHON' -#!/usr/bin/env bash -printf 'SELECTED_PYTHON_ARG=<%s>\n' "$@" >> "$KIMI_LOG" -PYTHON -chmod +x "$venv_dir/bin/python" -""" - ) - fake_uv.chmod(0o755) - script = r""" -source "$BENCHMARK_LIB" -python3() { - if [ "$1" = "-c" ]; then - printf 'VERSION_CHECK\n' >> "$KIMI_LOG" - return 1 - fi - printf 'SYSTEM_PYTHON_ARG=<%s>\n' "$@" >> "$KIMI_LOG" - local prefix="" - while [[ $# -gt 0 ]]; do - if [ "$1" = "--prefix" ]; then - prefix="$2" - break - fi - shift - done - mkdir -p "$prefix/bin" - cp "$FAKE_UV" "$prefix/bin/uv" - chmod +x "$prefix/bin/uv" -} -_prepare_vendor_verifier_python "Kimi Vendor Verifier" "kimi-vendor-python" -cleanup_dir="$VENDOR_VERIFIER_PYTHON_CLEANUP_DIR" -printf 'SELECTED_PYTHON=<%s>\n' "$VENDOR_VERIFIER_PYTHON" -printf 'PYTHON_CLEANUP=<%s>\n' "$cleanup_dir" -runtime_dir="$TEST_ROOT/runtime" -mkdir "$runtime_dir" -_install_kimi_vendor_eval_deps "$runtime_dir" -_cleanup_vendor_eval "$runtime_dir" "$cleanup_dir" -[ ! -e "$runtime_dir" ] && [ ! -e "$cleanup_dir" ] && printf 'CLEANED\n' -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "FAKE_UV": str(fake_uv), - "KIMI_LOG": str(log_path), - "TEST_ROOT": str(tmp_path), - }, - text=True, - capture_output=True, - check=True, - ) - log = log_path.read_text() - - assert "VERSION_CHECK" in log - assert "SYSTEM_PYTHON_ARG=<--prefix>" in log - assert "SYSTEM_PYTHON_ARG=<--break-system-packages>" in log - assert "UV_ARG=" in log - assert "UV_ARG=<--python>" in log - assert "UV_ARG=<--seed>" in log - assert "UV_CACHE_DIR=" in log - assert "SELECTED_PYTHON= None: - runtime_dir = tmp_path / "runtime" - script = r""" -source "$BENCHMARK_LIB" -selected_python() { printf 'PYTHON_ARG=<%s>\n' "$@"; } -VENDOR_VERIFIER_PYTHON=selected_python -_install_kimi_vendor_eval_deps "$RUNTIME_DIR" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RUNTIME_DIR": str(runtime_dir), - }, - text=True, - capture_output=True, - check=True, - ) - - assert "PYTHON_ARG=<--target>" in result.stdout - assert f"PYTHON_ARG=<{runtime_dir}>" in result.stdout - assert "--break-system-packages" not in result.stdout - - -def test_kimi_vendor_surfaces_failure_artifact_error(tmp_path: Path) -> None: - script = r""" -source "$BENCHMARK_LIB" -_prepare_vendor_verifier_python() { return 12; } -_write_kimi_vendor_integration_error() { return 23; } -run_kimi_vendor_eval --results-dir "$RESULTS_DIR" -printf 'EVAL_RC=%s\n' "$?" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(tmp_path / "results"), - "MODEL": "test-model", - "IS_MULTINODE": "false", - }, - text=True, - capture_output=True, - check=True, - ) - - assert "EVAL_RC=12" in result.stdout - assert "failed to write Kimi verifier failure artifact" in result.stderr - - -def test_kimi_vendor_multinode_runner_uses_fixed_upstream_contract( - tmp_path: Path, -) -> None: - results_dir = tmp_path / "results" - verifier_dir = tmp_path / "verifier" - runtime_dir = tmp_path / "runtime" - python_dir = tmp_path / "python" - verifier_dir.mkdir() - script = r""" -source "$BENCHMARK_LIB" -_prepare_vendor_verifier_python() { - mkdir "$PYTHON_DIR" - VENDOR_VERIFIER_PYTHON=selected_python - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_kimi_vendor_runtime() { - mkdir "$RUNTIME_DIR" - _install_kimi_vendor_eval_deps "$RUNTIME_DIR" >&2 - printf '%s\n' "$RUNTIME_DIR" -} -_prepare_kimi_vendor_verifier() { - printf 'CHECKOUT=%s@%s\n' "$1" "$2" >&2 - printf 'CHECKOUT_SHA=%s\n' "$3" >&2 - "$VENDOR_VERIFIER_PYTHON" - "$1" "$2" "$3" "$VERIFIER_DIR" <<'PY' >&2 -archive extraction -PY - printf '%s\n' "$VERIFIER_DIR" -} -selected_python() { - printf 'PYTHONPATH=<%s>\n' "$PYTHONPATH" >&2 - printf 'PYTHON_ARG=<%s>\n' "$@" >&2 -} -python3() { echo "SYSTEM_PYTHON_UNEXPECTED" >&2; return 99; } -run_kimi_vendor_eval --port 9999 --results-dir "$RESULTS_DIR" -printf 'EVAL_SUITE=%s\n' "$EVAL_SUITE" -printf 'EVAL_RESULT_DIR=%s\n' "$EVAL_RESULT_DIR" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "VERIFIER_DIR": str(verifier_dir), - "MODEL": "test-model", - "MODEL_PREFIX": "dsv4", - "RUNTIME_DIR": str(runtime_dir), - "PYTHON_DIR": str(python_dir), - "OPENAI_API_KEY": "must-not-be-forwarded", - "KV_OFFLOADING": "none", - "IS_MULTINODE": "true", - } - for key in ( - "EVAL_SUITE", - "EVAL_RESULT_DIR", - "MODEL_NAME", - ): - env.pop(key, None) - - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - output = result.stdout + result.stderr - adapter = BENCHMARK_LIB.parents[1] / "infx/evals/kimi_vendor_eval.py" - - assert f"PYTHONPATH=<{tmp_path / 'runtime'}" in output - assert "PYTHON_ARG=<->" in output - for value in ( - adapter, - verifier_dir, - "http://127.0.0.1:9999/v1", - "EMPTY", - "test-model", - results_dir, - ): - assert f"PYTHON_ARG=<{value}>" in output - assert "PYTHON_ARG=<--model-prefix>" in output - assert "PYTHON_ARG=" in output - assert "PYTHON_ARG=<--task-name>" in output - assert "PYTHON_ARG=" in output - assert "PYTHON_ARG=<--timeout-seconds>" not in output - assert "must-not-be-forwarded" not in output - assert "SYSTEM_PYTHON_UNEXPECTED" not in output - assert "EVAL_SUITE=kimi_tool_call_schema" in output - assert f"EVAL_RESULT_DIR={results_dir}" in output - assert not (tmp_path / "runtime").exists() - assert not verifier_dir.exists() - assert not python_dir.exists() - - -def test_kimi_full_runner_installs_xdist_without_deadline_and_cleans_runtimes( - tmp_path: Path, -) -> None: - results_dir = tmp_path / "results" - verifier_dir = tmp_path / "verifier" - runtime_dir = tmp_path / "runtime" - python_dir = tmp_path / "python" - script = r""" -source "$BENCHMARK_LIB" -_prepare_vendor_verifier_python() { - mkdir "$PYTHON_DIR" - VENDOR_VERIFIER_PYTHON=selected_python - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_kimi_vendor_runtime() { - printf 'RUNTIME_SUITE=<%s>\n' "$1" >&2 - mkdir "$RUNTIME_DIR" - _install_kimi_vendor_eval_deps "$RUNTIME_DIR" "$1" >&2 - printf '%s\n' "$RUNTIME_DIR" -} -_prepare_kimi_vendor_verifier() { - mkdir "$VERIFIER_DIR" - printf '%s\n' "$VERIFIER_DIR" -} -selected_python() { - printf 'PYTHONPATH=<%s>\n' "$PYTHONPATH" >&2 - printf 'PYTHON_ARG=<%s>\n' "$@" >&2 -} -run_kimi_vendor_eval --port 9999 --results-dir "$RESULTS_DIR" -printf 'EVAL_SUITE=%s\n' "$EVAL_SUITE" -printf 'EVAL_RESULT_DIR=%s\n' "$EVAL_RESULT_DIR" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "EVAL_SUITE": "kimi_tool_call_schema_full", - "RESULTS_DIR": str(results_dir), - "VERIFIER_DIR": str(verifier_dir), - "MODEL": "test-model", - "RUNTIME_DIR": str(runtime_dir), - "PYTHON_DIR": str(python_dir), - } - for key in ("EVAL_RESULT_DIR", "MODEL_NAME"): - env.pop(key, None) - - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - output = result.stdout + result.stderr - - assert "RUNTIME_SUITE=" in output - assert "PYTHON_ARG=" in output - assert "PYTHON_ARG=" in output - assert "PYTHON_ARG=<--timeout-seconds>" not in output - assert "EVAL_SUITE=kimi_tool_call_schema_full" in output - assert f"EVAL_RESULT_DIR={results_dir}" in output - assert not runtime_dir.exists() - assert not verifier_dir.exists() - assert not python_dir.exists() - - -def test_run_lm_eval_rejects_missing_option_value(): - result = _run_invalid_call("run_lm_eval --port") - assert result.returncode == 2 - assert "--port requires a value" in result.stderr - - -def test_lm_patch_copy_resolves_outside_repo(tmp_path): - script = r""" -source "$BENCHMARK_LIB" -cd "$OTHER_CWD" -_patch_lm_eval -patch_dir=${PYTHONPATH%%:*} -cmp "$(_eval_patches_dir)/lm_eval_sitecustomize.py" "$patch_dir/sitecustomize.py" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "OTHER_CWD": str(tmp_path), - "KV_OFFLOADING": "none", - } - subprocess.run(["bash", "-c", script], env=env, check=True) - - -_EVAL_LIMIT_SCRIPT = r""" -set -e -SHIM_DIR=$(mktemp -d) -cat > "$SHIM_DIR/python3" <<'PY' -#!/usr/bin/env bash -echo "PYTHON_ARGS: $*" -exit 0 -PY -chmod +x "$SHIM_DIR/python3" - -source "$BENCHMARK_LIB" - -export EVAL_MAX_MODEL_LEN=16384 -export MODEL_NAME=test-model -export OPENAI_API_KEY=EMPTY -export INFERENCEX_LM_EVAL_RUNTIME_READY=true - -_install_lm_eval_deps() { :; } -_patch_lm_eval() { :; } - -PATH="$SHIM_DIR:$PATH" run_lm_eval --port 9999 2>&1 -""" - - -def _run_lm_eval_cmdline(*, eval_limit=None) -> str: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - } - env.pop("EVAL_LIMIT", None) - if eval_limit is not None: - env["EVAL_LIMIT"] = str(eval_limit) - res = subprocess.run( - ["bash", "-c", _EVAL_LIMIT_SCRIPT], - env=env, - text=True, - capture_output=True, - check=True, - ) - return res.stdout + res.stderr - - -def test_eval_limit_appended_when_set(): - out = _run_lm_eval_cmdline(eval_limit=10) - assert "--limit 10" in out, f"Expected '--limit 10' in output:\n{out}" - - -def test_eval_limit_absent_when_unset(): - out = _run_lm_eval_cmdline(eval_limit=None) - assert "--limit" not in out, f"Expected no '--limit' in output:\n{out}" - - -def _summary_metadata(tmp_path: Path, **overrides: str) -> dict: - work_dir = tmp_path / "work" - results_dir = tmp_path / "results" - work_dir.mkdir(parents=True) - results_dir.mkdir() - script = r""" -source "$BENCHMARK_LIB" -cd "$WORK_DIR" -append_lm_eval_summary >/dev/null -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "WORK_DIR": str(work_dir), - "EVAL_RESULT_DIR": str(results_dir), - "MODEL": "test-model", - "CONC": "7", - "KV_OFFLOADING": "none", - } - for key in ( - "EVAL_COMPLETED_SUITE", "EVAL_SUITE", "EVAL_TASKS_DIR", - "DP_ATTENTION", - "PREFILL_DP_ATTN", "PREFILL_DP_ATTENTION", "PREFILL_ENABLE_DP", - "DECODE_DP_ATTN", "DECODE_DP_ATTENTION", "DECODE_ENABLE_DP", - ): - env.pop(key, None) - env.update(overrides) - subprocess.run(["bash", "-c", script], env=env, check=True) - return json.loads((work_dir / "meta_env.json").read_text()) - - -@pytest.mark.parametrize("dp_attention, expected", [("true", True), ("false", False)]) -def test_summary_preserves_single_node_dp_attention( - tmp_path: Path, dp_attention: str, expected: bool, -) -> None: - meta = _summary_metadata( - tmp_path, IS_MULTINODE="false", TP="8", EP_SIZE="8", - DP_ATTENTION=dp_attention, - ) - assert meta["dp_attention"] is expected - assert meta["prefill_dp_attention"] is expected - assert meta["decode_dp_attention"] is expected - assert meta["tp"] == 8 - assert meta["ep"] == 8 - - -def test_summary_preserves_asymmetric_multinode_dp_attention(tmp_path: Path) -> None: - meta = _summary_metadata( - tmp_path, IS_MULTINODE="true", DP_ATTENTION="false", - PREFILL_TP="4", PREFILL_EP="4", DECODE_TP="8", DECODE_EP="8", - PREFILL_DP_ATTN="true", DECODE_DP_ATTN="false", - ) - assert meta["dp_attention"] is True - assert meta["prefill_dp_attention"] is True - assert meta["decode_dp_attention"] is False - assert meta["prefill_tp"] == 4 - assert meta["decode_tp"] == 8 - - -def test_summary_stages_bfcl_upstream_archive_before_cleanup(tmp_path: Path) -> None: - work_dir = tmp_path / "work" - results_dir = tmp_path / "results" - work_dir.mkdir() - results_dir.mkdir() - archive = results_dir / "bfcl_upstream_artifacts.tar.gz" - archive.write_bytes(b"bfcl-archive") - script = r""" -source "$BENCHMARK_LIB" -cd "$WORK_DIR" -append_lm_eval_summary >/dev/null -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "WORK_DIR": str(work_dir), - "EVAL_RESULT_DIR": str(results_dir), - "MODEL": "test-model", - "CONC": "7", - "KV_OFFLOADING": "none", - } - - subprocess.run(["bash", "-c", script], env=env, check=True) - - assert (work_dir / archive.name).read_bytes() == b"bfcl-archive" - assert not results_dir.exists() - - -def test_stage_eval_artifacts_copies_eval_outputs_only(tmp_path: Path) -> None: - source_one = tmp_path / "source-one" - source_two = tmp_path / "source-two" - destination = tmp_path / "destination" - source_one.mkdir() - source_two.mkdir() - expected = { - "meta_env.json", - "results_bfcl.json", - "kimi_vendor_report.json", - "kimi_vendor_results.jsonl", - "bfcl_report.json", - "bfcl_upstream_artifacts.tar.gz", - "sample_eval.jsonl", - } - for filename in expected: - source = source_one if filename.endswith(".json") else source_two - (source / filename).write_text(filename) - (source_one / "unrelated.log").write_text("skip") - script = r""" -source "$BENCHMARK_LIB" -stage_eval_artifacts "$DESTINATION" "$SOURCE_ONE" "$SOURCE_TWO" -""" - subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "DESTINATION": str(destination), - "SOURCE_ONE": str(source_one), - "SOURCE_TWO": str(source_two), - "KV_OFFLOADING": "none", - }, - check=True, - ) - - assert {path.name for path in destination.iterdir()} == expected - - -def test_stage_eval_artifacts_propagates_copy_failure(tmp_path: Path) -> None: - source = tmp_path / "source" - source.mkdir() - (source / "bfcl_report.json").write_text("{}") - script = r""" -source "$BENCHMARK_LIB" -cp() { return 73; } -stage_eval_artifacts "$DESTINATION" "$SOURCE" -""" - - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "DESTINATION": str(tmp_path / "destination"), - "SOURCE": str(source), - "KV_OFFLOADING": "none", - }, - check=False, - ) - - assert result.returncode == 73 - - -def test_stage_eval_artifacts_fails_when_no_artifacts_exist(tmp_path: Path) -> None: - source = tmp_path / "source" - source.mkdir() - script = r""" -source "$BENCHMARK_LIB" -stage_eval_artifacts "$DESTINATION" "$SOURCE" -""" - - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "DESTINATION": str(tmp_path / "destination"), - "SOURCE": str(source), - "KV_OFFLOADING": "none", - }, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode != 0 - assert "no eval artifacts found to stage" in result.stderr - - -def test_summary_propagates_artifact_staging_failure(tmp_path: Path) -> None: - work_dir = tmp_path / "work" - results_dir = tmp_path / "results" - work_dir.mkdir() - results_dir.mkdir() - (results_dir / "results_eval.json").write_text("{}") - script = r""" -source "$BENCHMARK_LIB" -cp() { return 73; } -cd "$WORK_DIR" -append_lm_eval_summary -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "WORK_DIR": str(work_dir), - "EVAL_RESULT_DIR": str(results_dir), - "MODEL": "test-model", - "CONC": "7", - "KV_OFFLOADING": "none", - }, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == 73 - - -def test_summary_metadata_preserves_lm_eval_gsm8k_defaults(tmp_path: Path) -> None: - meta = _summary_metadata(tmp_path) - - assert meta["eval_suite"] == "gsm8k" - assert meta["conc"] == 7 - - -def test_summary_metadata_preserves_single_node_expert_parallelism( - tmp_path: Path, -) -> None: - meta = _summary_metadata(tmp_path, TP="8", EP_SIZE="8") - - assert meta["ep"] == 8 - assert meta["prefill_ep"] == 8 - assert meta["decode_ep"] == 8 - - -def test_run_lm_eval_exports_cli_task_path(tmp_path: Path) -> None: - script = r""" -source "$BENCHMARK_LIB" -python3() { :; } -export EVAL_MAX_MODEL_LEN=16384 -export INFERENCEX_LM_EVAL_RUNTIME_READY=true -run_lm_eval --task custom.yaml --results-dir "$RESULTS_DIR" -printf 'EVAL_TASKS_DIR=%s\n' "$EVAL_TASKS_DIR" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(tmp_path / "results"), - "MODEL_NAME": "test-model", - "OPENAI_API_KEY": "EMPTY", - "KV_OFFLOADING": "none", - } - env.pop("EVAL_TASKS_DIR", None) - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - - assert "EVAL_TASKS_DIR=custom.yaml" in result.stdout - - -def test_summary_metadata_prefers_explicit_suite_then_task_basename( - tmp_path: Path, -) -> None: - from_task = _summary_metadata( - tmp_path / "task", - EVAL_TASKS_DIR="/tmp/custom_reasoning.yaml", - ) - explicit = _summary_metadata( - tmp_path / "explicit", - EVAL_SUITE="kimi_tool_call_schema", - EVAL_TASKS_DIR="/tmp/ignored.yaml", - ) - - assert from_task["eval_suite"] == "custom_reasoning" - assert explicit["eval_suite"] == "kimi_tool_call_schema" - - -def test_summary_metadata_prefers_completed_eval_identity(tmp_path: Path) -> None: - meta = _summary_metadata( - tmp_path, - EVAL_COMPLETED_SUITE="kimi_tool_call_schema", - EVAL_SUITE="stale_input_selector", - EVAL_TASKS_DIR="/tmp/ignored.yaml", - ) - - assert meta["eval_suite"] == "kimi_tool_call_schema" - - -def test_env_is_true_is_case_insensitive_and_unset_safe() -> None: - script = r""" -source "$BENCHMARK_LIB" -for value in TrUe yEs oN 1 false 0; do - if _env_is_true "$value"; then - echo true - else - echo false - fi -done -for empty_call in with-argument without-argument; do - if [ "$empty_call" = "with-argument" ]; then - _env_is_true "" - else - _env_is_true - fi - if [ "$?" -eq 0 ]; then - echo true - else - echo false - fi -done -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB)}, - text=True, - capture_output=True, - check=True, - ) - - assert result.stdout.splitlines() == [ - "true", - "true", - "true", - "true", - "false", - "false", - "false", - "false", - ] - - -_INCLUDE_PATH_SCRIPT = r""" -set -e -SHIM_DIR=$(mktemp -d) -cat > "$SHIM_DIR/python3" <<'PY' -#!/usr/bin/env bash -echo "PYTHON_ARGS: $*" -exit 0 -PY -chmod +x "$SHIM_DIR/python3" - -source "$BENCHMARK_LIB" - -export EVAL_MAX_MODEL_LEN=16384 -export MODEL_NAME=test-model -export OPENAI_API_KEY=EMPTY -export INFERENCEX_LM_EVAL_RUNTIME_READY=true - -_install_lm_eval_deps() { :; } -_patch_lm_eval() { :; } - -PATH="$SHIM_DIR:$PATH" run_lm_eval --port 9999 2>&1 -""" - - -def _run_lm_eval_with_include_path( - *, - eval_include_path: str | None = None, - eval_tasks_dir: str | None = None, -) -> str: - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "KV_OFFLOADING": "none", - } - env.pop("EVAL_INCLUDE_PATH", None) - env.pop("EVAL_TASKS_DIR", None) - if eval_include_path is not None: - env["EVAL_INCLUDE_PATH"] = eval_include_path - if eval_tasks_dir is not None: - env["EVAL_TASKS_DIR"] = eval_tasks_dir - res = subprocess.run( - ["bash", "-c", _INCLUDE_PATH_SCRIPT], - cwd=REPO_ROOT, - env=env, - text=True, - capture_output=True, - check=True, - ) - return res.stdout + res.stderr - - -def test_include_path_injected_when_eval_include_path_set(): - out = _run_lm_eval_with_include_path( - eval_include_path="infx/evals", - eval_tasks_dir="custom_task", - ) - assert "--include_path infx/evals" in out, ( - f"Expected '--include_path infx/evals' in output:\n{out}" - ) - assert "--tasks custom_task" in out, ( - f"Expected '--tasks custom_task' in output:\n{out}" - ) - assert ".yaml" not in out.split("--tasks")[1].split()[0], ( - f"--tasks must not contain a .yaml path when include_path is set:\n{out}" - ) - - -def test_include_path_absent_when_eval_include_path_unset(): - out = _run_lm_eval_with_include_path() - assert "--include_path" not in out, ( - f"Expected no '--include_path' in output:\n{out}" - ) - assert "--tasks infx/evals/gsm8k.yaml" in out, ( - f"Expected '--tasks infx/evals/gsm8k.yaml' in output:\n{out}" - ) - - -# Advance the watchdog's clock by requested waits. The small real yield lets -# child processes run; their sleep, signal handling, and exit status stay real. - - -def test_chat_route_readiness_requires_model_and_active_route(tmp_path: Path) -> None: - bin_dir = tmp_path / "bin" - events_path = tmp_path / "events" - bin_dir.mkdir() - curl = bin_dir / "curl" - curl.write_text( - """#!/usr/bin/env bash -printf 'curl %s\n' "$*" >> "$EVENTS" -case "$*" in - */v1/models*) printf '{"data":[{"id":"test-model"}]}\n' ;; - */v1/chat/completions*) printf '405' ;; -esac -""", - encoding="utf-8", - ) - curl.chmod(curl.stat().st_mode | stat.S_IXUSR) - script = r""" -source "$BENCHMARK_LIB" -MODEL=test-model -_wait_for_openai_chat_route --port 8765 -""" - - subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "PATH": f"{bin_dir}{os.pathsep}{os.environ['PATH']}", - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "EVENTS": str(events_path), - }, - text=True, - capture_output=True, - check=True, - ) - - events = events_path.read_text().splitlines() - assert events[0].endswith("http://localhost:8765/health") - assert events[1].endswith("http://localhost:8765/v1/models") - assert events[2].endswith("http://localhost:8765/v1/chat/completions") - assert "--data" not in events[2] - - -def test_multinode_agentic_waits_only_for_eval_openai_endpoint( - tmp_path: Path, -) -> None: - workspace = tmp_path / "workspace" - events_path = tmp_path / "events" - (workspace / "benchmarks").mkdir(parents=True) - (workspace / "benchmarks/benchmark_lib.sh").write_text( - """ -source "$BENCHMARK_LIB" --validation-only -PORT=8765 -check_env_vars() { :; } -resolve_trace_source() { echo resolve >> "$EVENTS"; } -AIPERF_PYTHON=python3 -install_agentic_deps() { echo deps >> "$EVENTS"; } -_wait_for_openai_chat_route() { echo "ready $*" >> "$EVENTS"; } -build_replay_cmd() { echo build >> "$EVENTS"; } -run_agentic_replay_and_write_outputs() { echo replay >> "$EVENTS"; } -""", - encoding="utf-8", - ) - - base_env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "INFMAX_CONTAINER_WORKSPACE": str(workspace), - "EVENTS": str(events_path), - "MODEL": "test-model", - "MODEL_PREFIX": "test-prefix", - "FRAMEWORK": "dynamo-vllm", - "PRECISION": "fp4", - "CONC": "1", - "RESULT_FILENAME": "result", - "RESULT_DIR": str(tmp_path / "results"), - "DURATION": "1", - } - expected_without_readiness = ["resolve", "deps", "build", "replay"] - - for eval_only, expected in ( - ("false", expected_without_readiness), - ("true", [*expected_without_readiness[:2], "ready --port 8765", *expected_without_readiness[2:]]), - ): - events_path.unlink(missing_ok=True) - subprocess.run( - ["bash", str(MULTINODE_AGENTIC_SCRIPT)], - env={**base_env, "EVAL_ONLY": eval_only}, - text=True, - capture_output=True, - check=True, - ) - assert events_path.read_text().splitlines() == expected - - -def test_env_can_force_bfcl_on_agentic_eval() -> None: - output = _dispatch(is_agentic="1", eval_only="true", env_fw="bfcl") - - assert "DISPATCH=bfcl" in output - assert "STAGED=summary" in output - - -def test_bfcl_defaults_suite_dispatches_once_without_context_loading() -> None: - script = r""" -source "$BENCHMARK_LIB" -unset EVAL_MAX_MODEL_LEN -compute_eval_context_length() { echo "UNEXPECTED_CONTEXT_LOAD"; return 99; } -BFCL_DISPATCH_COUNT=0 -run_bfcl_eval() { - BFCL_DISPATCH_COUNT=$((BFCL_DISPATCH_COUNT + 1)) - printf 'DISPATCH=bfcl SUITE=%s ARGS=<%s>\n' "$EVAL_SUITE" "$*" -} -append_lm_eval_summary() { printf 'STAGED=%s\n' "$EVAL_COMPLETED_SUITE"; } -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_ONLY=false -export IS_AGENTIC=0 -run_eval --framework bfcl --port 9999 -printf 'DISPATCH_COUNT=%s\n' "$BFCL_DISPATCH_COUNT" -printf 'COMPLETED_SUITE=%s\n' "$EVAL_COMPLETED_SUITE" -""" - env = {**os.environ, "BENCHMARK_LIB": str(BENCHMARK_LIB), "MODEL": "served-model"} - for key in ("EVAL_FRAMEWORK", "EVAL_SUITE", "EVAL_COMPLETED_SUITE"): - env.pop(key, None) - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=True, - ) - - assert "DISPATCH=bfcl SUITE=bfcl_smoke ARGS=<--port 9999>" in result.stdout - assert "DISPATCH_COUNT=1" in result.stdout - assert "COMPLETED_SUITE=bfcl_smoke" in result.stdout - assert "STAGED=bfcl_smoke" not in result.stdout - assert "UNEXPECTED_CONTEXT_LOAD" not in result.stdout - - -def test_bfcl_rejects_suite_from_another_provider() -> None: - result = _run_invalid_call( - "EVAL_CONCURRENT_REQUESTS='' " - "EVAL_SUITE=minimax_m3_smoke " - "run_eval --framework bfcl" - ) - - assert result.returncode == 2 - assert "unsupported BFCL suite 'minimax_m3_smoke'" in result.stderr - - -def test_bfcl_dependency_timeout_uses_integration_error_and_stages( - tmp_path: Path, -) -> None: - results_dir = tmp_path / "results" - python_dir = tmp_path / "python" - results_dir.mkdir() - stale_result = results_dir / "results_bfcl_previous.json" - stale_result.write_text('{"stale": true}\n') - script = r""" -source "$BENCHMARK_LIB" -export EVAL_SUITE=bfcl_vllm_kimi -unset VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -selected_python() { - printf 'ADAPTER_ARG=<%s>\n' "$@" - touch "$RESULTS_DIR/bfcl_report.json" "$RESULTS_DIR/results_bfcl.json" - return 1 -} -_prepare_vendor_verifier_python() { - mkdir "$PYTHON_DIR" - VENDOR_VERIFIER_PYTHON=selected_python - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_bfcl_runtime() { return 124; } -python3() { echo "UNEXPECTED_SYSTEM_PYTHON"; return 99; } -append_lm_eval_summary() { printf 'STAGED=<%s>\n' "$EVAL_RESULT_DIR"; } -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_ONLY=false -export IS_AGENTIC=0 -eval_rc=0 -run_eval --framework bfcl --results-dir "$RESULTS_DIR" || eval_rc=$? -printf 'EVAL_RC=%s\n' "$eval_rc" -exit "$eval_rc" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "PYTHON_DIR": str(python_dir), - "MODEL": "test-model", - } - env.pop("EVAL_FRAMEWORK", None) - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=False, - ) - output = result.stdout + result.stderr - - assert result.returncode == 124 - assert "EVAL_RC=124" in output - assert f"ADAPTER_ARG=<{REPO_ROOT / 'infx/evals/bfcl_adapter.py'}>" in output - assert "ADAPTER_ARG=" in output - assert f"ADAPTER_ARG=<{results_dir}>" in output - assert "ADAPTER_ARG=<--integration-error>" in output - assert "ADAPTER_ARG=<--suite>" in output - assert "ADAPTER_ARG=" in output - assert ( - "ADAPTER_ARG=" in output - ) - assert output.count(f"STAGED=<{results_dir}>") == 1 - assert (results_dir / "bfcl_report.json").exists() - assert (results_dir / "results_bfcl.json").exists() - assert not stale_result.exists() - assert "failed to write BFCL failure artifact" not in output - assert "UNEXPECTED_SYSTEM_PYTHON" not in output - assert not (results_dir / "bfcl_upstream_artifacts.tar.gz").exists() - assert not python_dir.exists() - - -def _run_bfcl_adapter_command( - tmp_path: Path, - *, - adapter_rc: int = 0, - suite: str = "", - archive_rc: int = 0, -) -> tuple[subprocess.CompletedProcess[str], tuple[Path, Path, Path, Path]]: - results_dir = tmp_path / "results" - runtime_dir = tmp_path / "runtime" - python_dir = tmp_path / "python" - project_root = tmp_path / "bfcl-project" - script = r""" -source "$BENCHMARK_LIB" -selected_python() { - printf 'ADAPTER_ARG=<%s>\n' "$@" - local arg - for arg in "$@"; do - if [ "$arg" = "--integration-error" ]; then - touch "$RESULTS_DIR/bfcl_report.json" "$RESULTS_DIR/results_bfcl.json" - return 1 - fi - done - if [ "$TEST_ADAPTER_RC" -eq 0 ]; then - touch "$RESULTS_DIR/bfcl_report.json" "$RESULTS_DIR/results_bfcl.json" - fi - return "$TEST_ADAPTER_RC" -} -_prepare_vendor_verifier_python() { - printf 'PREPARE_ARG=<%s>\n' "$@" - mkdir "$PYTHON_DIR" - VENDOR_VERIFIER_PYTHON=selected_python - VENDOR_VERIFIER_PYTHON_CLEANUP_DIR="$PYTHON_DIR" - export VENDOR_VERIFIER_PYTHON VENDOR_VERIFIER_PYTHON_CLEANUP_DIR -} -_prepare_bfcl_runtime() { - mkdir "$RUNTIME_DIR" - printf '%s\n' "$RUNTIME_DIR" -} -mktemp() { - printf 'MKTEMP_ARG=<%s>\n' "$@" >&2 - mkdir "$PROJECT_ROOT" - printf '%s\n' "$PROJECT_ROOT" -} -_archive_bfcl_upstream_artifacts() { - printf 'ARCHIVE_PROJECT_ROOT=<%s>\n' "$1" - printf 'ARCHIVE_PATH=<%s>\n' "$2" - if [ "$TEST_ARCHIVE_RC" -eq 0 ]; then - touch "$2" - fi - return "$TEST_ARCHIVE_RC" -} -timeout() { - printf 'TIMEOUT_ARG=<%s>\n' "$1" - shift - "$@" -} -append_lm_eval_summary() { printf 'STAGED=<%s>\n' "$EVAL_RESULT_DIR"; } -if [ -n "$TEST_SUITE" ]; then - export EVAL_SUITE="$TEST_SUITE" -else - unset EVAL_SUITE -fi -unset EVAL_RESULT_DIR EVAL_COMPLETED_SUITE EVAL_MAX_MODEL_LEN -compute_eval_context_length() { echo "UNEXPECTED_CONTEXT_LOAD"; return 99; } -export EVAL_CONCURRENT_REQUESTS="" -export EVAL_ONLY=false -export IS_AGENTIC=0 -eval_rc=0 -run_eval --framework bfcl --port 9999 --results-dir "$RESULTS_DIR" || eval_rc=$? -printf 'EVAL_RC=%s\n' "$eval_rc" -printf 'EVAL_COMPLETED_SUITE=%s\n' "$EVAL_COMPLETED_SUITE" -printf 'EVAL_RESULT_DIR=%s\n' "$EVAL_RESULT_DIR" -exit "$eval_rc" -""" - env = { - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "RESULTS_DIR": str(results_dir), - "RUNTIME_DIR": str(runtime_dir), - "PYTHON_DIR": str(python_dir), - "PROJECT_ROOT": str(project_root), - "MODEL": "repository/model", - "MODEL_NAME": "served-model", - "OPENAI_API_KEY": "must-not-be-forwarded", - "TEST_ADAPTER_RC": str(adapter_rc), - "TEST_SUITE": suite, - "TEST_ARCHIVE_RC": str(archive_rc), - } - for key in ( - "EVAL_FRAMEWORK", - "EVAL_SUITE", - "EVAL_RESULT_DIR", - "EVAL_COMPLETED_SUITE", - "VENDOR_VERIFIER_PYTHON", - "VENDOR_VERIFIER_PYTHON_CLEANUP_DIR", - ): - env.pop(key, None) - result = subprocess.run( - ["bash", "-c", script], - env=env, - text=True, - capture_output=True, - check=False, - ) - return result, (results_dir, runtime_dir, python_dir, project_root) - - -def test_bfcl_runner_uses_fixed_adapter_contract_and_cleans_runtime( - tmp_path: Path, -) -> None: - result, paths = _run_bfcl_adapter_command(tmp_path) - results_dir, runtime_dir, python_dir, project_root = paths - output = result.stdout + result.stderr - - assert result.returncode == 0, result.stderr - for value in ( - str(REPO_ROOT / "infx/evals/bfcl_adapter.py"), - "--base-url", - "http://127.0.0.1:9999/v1", - "--api-key", - "EMPTY", - "--model", - "served-model", - "--output-dir", - str(results_dir), - "--bfcl-project-root", - str(project_root), - "--num-threads", - "4", - ): - assert f"ADAPTER_ARG=<{value}>" in output - assert "PREPARE_ARG=" in output - assert "PREPARE_ARG=" in output - assert "PREPARE_ARG=" in output - assert "PREPARE_ARG=<10>" in output - assert "TIMEOUT_ARG=<900>" in output - assert "ADAPTER_ARG=<--suite>" not in output - assert "ARCHIVE_PROJECT_ROOT=" not in output - assert not (results_dir / "bfcl_upstream_artifacts.tar.gz").exists() - assert "ADAPTER_ARG=" not in output - assert "UNEXPECTED_CONTEXT_LOAD" not in output - assert f"STAGED=<{results_dir}>" not in output - assert "EVAL_RC=0" in output - assert "EVAL_COMPLETED_SUITE=bfcl_smoke" in output - assert f"EVAL_RESULT_DIR={results_dir}" in output - assert results_dir.exists() - assert not runtime_dir.exists() - assert not python_dir.exists() - assert not project_root.exists() - - -def test_bfcl_full_suites_use_suite_specific_runtime_and_archive_before_cleanup( - tmp_path: Path, -) -> None: - suite_contracts = ( - ("bfcl_vllm_minimax_m3", "8", "7200"), - ("bfcl_vllm_kimi", "16", "14400"), - ) - - for suite, expected_threads, expected_timeout in suite_contracts: - suite_tmp_path = tmp_path / suite - suite_tmp_path.mkdir() - result, paths = _run_bfcl_adapter_command( - suite_tmp_path, - suite=suite, - ) - results_dir, runtime_dir, python_dir, project_root = paths - output = result.stdout + result.stderr - - assert result.returncode == 0, result.stderr - assert f"TIMEOUT_ARG=<{expected_timeout}>" in output - assert "ADAPTER_ARG=<--suite>" in output - assert f"ADAPTER_ARG=<{suite}>" in output - assert f"EVAL_COMPLETED_SUITE={suite}" in output - assert "ADAPTER_ARG=<--num-threads>" in output - assert f"ADAPTER_ARG=<{expected_threads}>" in output - assert f"ARCHIVE_PROJECT_ROOT=<{project_root}>" in output - assert ( - f"ARCHIVE_PATH=<{results_dir / 'bfcl_upstream_artifacts.tar.gz'}>" in output - ) - assert (results_dir / "bfcl_upstream_artifacts.tar.gz").exists() - assert not runtime_dir.exists() - assert not python_dir.exists() - assert not project_root.exists() - - -def test_bfcl_full_suite_archive_failure_preserves_scores_and_cleans_runtime( - tmp_path: Path, -) -> None: - result, paths = _run_bfcl_adapter_command( - tmp_path, - suite="bfcl_vllm_kimi", - archive_rc=73, - ) - results_dir, runtime_dir, python_dir, project_root = paths - output = result.stdout + result.stderr - - assert result.returncode == 73 - assert "failed to archive BFCL upstream artifacts (exit code 73)" in output - assert (results_dir / "bfcl_report.json").exists() - assert (results_dir / "results_bfcl.json").exists() - assert not (results_dir / "bfcl_upstream_artifacts.tar.gz").exists() - assert not runtime_dir.exists() - assert not python_dir.exists() - assert not project_root.exists() - - -def test_bfcl_adapter_timeout_writes_reports_stages_and_propagates( - tmp_path: Path, -) -> None: - result, paths = _run_bfcl_adapter_command(tmp_path, adapter_rc=124) - results_dir, runtime_dir, python_dir, project_root = paths - output = result.stdout + result.stderr - - assert result.returncode == 124 - assert "EVAL_RC=124" in output - assert output.count(f"STAGED=<{results_dir}>") == 1 - assert "ADAPTER_ARG=<--integration-error>" in output - assert "ADAPTER_ARG=<--suite>" in output - assert "ADAPTER_ARG=" in output - assert "ADAPTER_ARG=" in output - assert (results_dir / "bfcl_report.json").exists() - assert (results_dir / "results_bfcl.json").exists() - assert "failed to write BFCL failure artifact" not in output - assert "run_eval failed with exit code 124" in result.stderr - assert not runtime_dir.exists() - assert not python_dir.exists() - assert not project_root.exists() - - -def test_bfcl_upstream_archive_is_deterministic_and_survives_cleanup( - tmp_path: Path, -) -> None: - project_root = tmp_path / "project" - result_path = project_root / "result/run/BFCL_v4_simple_python_result.json" - score_path = project_root / "score/run/BFCL_v4_simple_python_score.json" - id_path = project_root / "test_case_ids_to_generate.json" - for path, content in ( - (result_path, '{"id":"simple_python_0"}\n'), - (score_path, '{"accuracy":1.0}\n'), - (id_path, '{"simple_python":["simple_python_0"]}\n'), - ): - path.parent.mkdir(parents=True, exist_ok=True) - path.write_text(content) - first_archive = tmp_path / "first.tar.gz" - second_archive = tmp_path / "second.tar.gz" - script = r""" -source "$BENCHMARK_LIB" -VENDOR_VERIFIER_PYTHON="$PYTHON" -_archive_bfcl_upstream_artifacts "$PROJECT_ROOT" "$FIRST_ARCHIVE" -_archive_bfcl_upstream_artifacts "$PROJECT_ROOT" "$SECOND_ARCHIVE" -_cleanup_vendor_eval "$PROJECT_ROOT" -""" - subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "PYTHON": sys.executable, - "PROJECT_ROOT": str(project_root), - "FIRST_ARCHIVE": str(first_archive), - "SECOND_ARCHIVE": str(second_archive), - }, - text=True, - capture_output=True, - check=True, - ) - - assert first_archive.read_bytes() == second_archive.read_bytes() - with tarfile.open(first_archive, "r:gz") as archive: - assert archive.getnames() == [ - "result", - "result/run", - "result/run/BFCL_v4_simple_python_result.json", - "score", - "score/run", - "score/run/BFCL_v4_simple_python_score.json", - "test_case_ids_to_generate.json", - ] - assert not project_root.exists() - - -def test_bfcl_upstream_archive_rejects_symbolic_links(tmp_path: Path) -> None: - project_root = tmp_path / "project" - project_root.mkdir() - outside = tmp_path / "outside.json" - outside.write_text('{"secret":true}\n') - (project_root / "escape.json").symlink_to(outside) - archive = tmp_path / "unsafe.tar.gz" - script = r""" -source "$BENCHMARK_LIB" -VENDOR_VERIFIER_PYTHON="$PYTHON" -_archive_bfcl_upstream_artifacts "$PROJECT_ROOT" "$ARCHIVE" -""" - - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "PYTHON": sys.executable, - "PROJECT_ROOT": str(project_root), - "ARCHIVE": str(archive), - }, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode != 0 - assert "refusing to archive symbolic link: escape.json" in result.stderr - assert not archive.exists() - assert not (tmp_path / ".unsafe.tar.gz.tmp").exists() - - -@pytest.mark.parametrize("verification_rc", (0, 23)) -def test_bfcl_installer_requires_verified_wheel_before_installing( - tmp_path: Path, - verification_rc: int, -) -> None: - script = r""" -source "$BENCHMARK_LIB" -selected_python() { - if [ "$1" = "-" ]; then - printf 'VERIFY_PATH=<%s>\n' "$4" - [ "$VERIFICATION_RC" -eq 0 ] || return "$VERIFICATION_RC" - printf 'verified wheel' > "$4" - else - printf 'INSTALL_ARG=<%s>\n' "$@" - for arg in "$@"; do - if [ -f "$arg" ]; then - printf 'INSTALLED_CONTENT=<%s>\n' "$(cat "$arg")" - fi - done - fi -} -timeout() { - printf 'TIMEOUT_ARG=<%s>\n' "$1" - shift - "$@" -} -VENDOR_VERIFIER_PYTHON=selected_python -_install_bfcl_eval_deps "$DOWNLOAD_DIR" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "DOWNLOAD_DIR": str(tmp_path), - "VERIFICATION_RC": str(verification_rc), - }, - text=True, - capture_output=True, - check=False, - ) - - assert result.returncode == verification_rc, result.stderr - if verification_rc: - assert "INSTALL_ARG=" not in result.stdout - else: - args = [ - line.removeprefix("INSTALL_ARG=<").removesuffix(">") - for line in result.stdout.splitlines() - if line.startswith("INSTALL_ARG=<") - ] - assert args[:3] == ["-m", "pip", "install"] - assert "INSTALLED_CONTENT=" in result.stdout - assert "--break-system-packages" not in args - assert "--target" not in args - - -def test_bfcl_python_preparation_exposes_system_site_packages( - tmp_path: Path, -) -> None: - python_root = tmp_path / "bfcl-python" - script = r""" -source "$BENCHMARK_LIB" -python3() { - if [ "$1" = "-c" ]; then - printf 'VERSION_CHECK_ARG=<%s>\n' "$@" - return 0 - fi - printf 'SYSTEM_PYTHON_ARG=<%s>\n' "$@" - venv_dir="${!#}" - mkdir -p "$venv_dir/bin" - printf '#!/usr/bin/env bash\n' > "$venv_dir/bin/python" - chmod +x "$venv_dir/bin/python" -} -mktemp() { - mkdir "$PYTHON_ROOT" - printf '%s\n' "$PYTHON_ROOT" -} -_prepare_vendor_verifier_python "BFCL" "bfcl-python" true 10 -cleanup_dir="$VENDOR_VERIFIER_PYTHON_CLEANUP_DIR" -printf 'SELECTED_PYTHON=<%s>\n' "$VENDOR_VERIFIER_PYTHON" -_cleanup_vendor_eval "$cleanup_dir" -""" - result = subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "BENCHMARK_LIB": str(BENCHMARK_LIB), - "PYTHON_ROOT": str(python_root), - }, - text=True, - capture_output=True, - check=True, - ) - - assert "SYSTEM_PYTHON_ARG=<-m>" in result.stdout - assert "SYSTEM_PYTHON_ARG=" in result.stdout - assert "SYSTEM_PYTHON_ARG=<--system-site-packages>" in result.stdout - assert "VERSION_CHECK_ARG=<10>" in result.stdout - assert f"SELECTED_PYTHON=<{python_root / 'venv/bin/python'}>" in result.stdout - assert "--prefix" not in result.stdout - assert not python_root.exists() diff --git a/inferencex-e2e/infx/tests/launch/fake_slurm.py b/inferencex-e2e/infx/tests/launch/fake_slurm.py index 851a62b06d..35a4e1525d 100644 --- a/inferencex-e2e/infx/tests/launch/fake_slurm.py +++ b/inferencex-e2e/infx/tests/launch/fake_slurm.py @@ -130,6 +130,8 @@ if os.environ.get("RUN_EVAL") == "true" or os.environ.get("EVAL_ONLY") == "true": (logs / "eval_results").mkdir() (logs / "eval_results" / "results_gsm8k.json").write_text("{}") + if os.environ.get("FAKE_EVAL_META"): + (logs / "eval_results" / "meta_env.json").write_text(os.environ["FAKE_EVAL_META"]) (logs / "infx-eval-exit-code").write_text("0\n") if os.environ.get("FAKE_ACTIVE"): pathlib.Path(os.environ["FAKE_ACTIVE"]).touch() @@ -217,7 +219,7 @@ def runner_for(cluster_id: str) -> str: def make_workspace(workspace: Path) -> Path: - """A GITHUB_WORKSPACE with the recipe mirror, one patch beside the patches README, and a stub benchmark_lib.""" + """A GITHUB_WORKSPACE with the recipe mirror and one patch beside the patches README.""" recipes = workspace / "benchmarks/multi_node/srt-slurm-recipes" (recipes / "configs").mkdir(parents=True) (recipes / "configs/setup.sh").write_text("true\n") @@ -225,9 +227,6 @@ def make_workspace(workspace: Path) -> Path: patches.mkdir(parents=True) (patches / "README.md").write_text("# srt-slurm patches\n") (patches / "001-fixture.patch").write_text("fixture\n") - (workspace / "benchmarks/benchmark_lib.sh").write_text( - '_write_lm_eval_meta_json() { printf \'{"conc": "%s"}\\n\' "$3" > "$1"; }\n' - ) return workspace diff --git a/inferencex-e2e/infx/tests/launch/test_launch_request.py b/inferencex-e2e/infx/tests/launch/test_launch_request.py new file mode 100644 index 0000000000..715c8c92bf --- /dev/null +++ b/inferencex-e2e/infx/tests/launch/test_launch_request.py @@ -0,0 +1,35 @@ +"""Launch-request validation that rejects a job before any work.""" + +import pytest + +from infx.launch.request import LaunchRequest, RequestError + +MULTINODE_AGENTX = {"RUNNER_NAME": "r_0", "IS_MULTINODE": "true", "IS_AGENTIC": "1"} + + +@pytest.mark.parametrize( + ("overrides", "concurrencies"), + [ + ({"CONC": "4", "CONC_LIST": "4"}, [4]), + ({"EVAL_ONLY": "true", "CONC": "4", "CONC_LIST": "4 8"}, [4, 8]), + ({"IS_AGENTIC": "0", "CONC_LIST": "4 8"}, [4, 8]), + ({"IS_MULTINODE": "false", "CONC": "4"}, []), + ], + ids=["one-concurrency", "eval-only-batches", "fixed-sequence-batches", "single-node"], +) +def test_one_agentx_deployment_or_batched_points_are_accepted(overrides, concurrencies): + assert LaunchRequest.from_env({**MULTINODE_AGENTX, **overrides}).conc_list == concurrencies + + +@pytest.mark.parametrize( + ("overrides", "error"), + [ + ({"CONC": "4", "CONC_LIST": "4 8"}, "exactly one positive concurrency"), + ({"CONC": "0", "CONC_LIST": "0"}, "exactly one positive concurrency"), + ({"CONC_LIST": "4"}, "not set for AgentX throughput: CONC$"), + ], + ids=["two-concurrencies", "zero", "no-conc"], +) +def test_multinode_agentx_throughput_needs_one_concurrency_per_deployment(overrides, error): + with pytest.raises(RequestError, match=error): + LaunchRequest.from_env({**MULTINODE_AGENTX, **overrides}) diff --git a/inferencex-e2e/infx/tests/launch/test_srt_driver.py b/inferencex-e2e/infx/tests/launch/test_srt_driver.py index bbaac995d7..826efb99b2 100644 --- a/inferencex-e2e/infx/tests/launch/test_srt_driver.py +++ b/inferencex-e2e/infx/tests/launch/test_srt_driver.py @@ -185,7 +185,7 @@ def test_single_node_failed_allocation_fails_the_launch(harness): ), "lab-b": dict( lane=SrtLane(shared_run_root=(Match(),)), - env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", FAKE_RESULTS="agentic"), + env=dict(FRAMEWORK="dynamo-vllm", IS_AGENTIC="1", ISL="0", OSL="0", CONC="4", FAKE_RESULTS="agentic"), model="models/model", preflight=True, tag=None, setup_script=None, served=None, dist_timeout=False, time="10", mounts=(), staging="registry", shared_checkout=True, ), @@ -303,7 +303,7 @@ def test_power_lane_stages_provenance_and_validates_each_concurrency( env = lane_env( harness, "h200-dgxc", LANE_RECIPE + POWER_TELEMETRY, MODEL_PREFIX=model_prefix, PRECISION=precision, FRAMEWORK=framework, MODEL=model, - IS_AGENTIC="1", ISL="0", OSL="0", CONC_LIST="4 8", FAKE_RESULTS="agentic", + IS_AGENTIC="1", ISL="0", OSL="0", CONC="4", CONC_LIST="4", FAKE_RESULTS="agentic", REQUIRE_POWER=require_power, INFERENCEX_RESULTS_PYTHON=str(adapter), ) # fmt: skip assert_ok(launch(env, harness.config, harness.workspace)) @@ -318,10 +318,10 @@ def test_power_lane_stages_provenance_and_validates_each_concurrency( assert (workspace / "LOGS/power/exporter-image.sha256").read_text() == provenance assert (workspace / "LOGS/power/power-producer-sha.txt").read_text() == commit + "\n" staged = yaml.safe_load((checkout / "recipes/test/lane.yaml").read_text()) - assert staged["benchmark"]["concurrencies"] == [4, 8] + assert staged["benchmark"]["concurrencies"] == [4] runs = lines(harness.logs, "adapter") - assert [run.split("--result-dir ")[1].split()[0].rsplit("/", 1)[1] for run in runs] == ["conc_4", "conc_8"] + assert [run.split("--result-dir ")[1].split()[0].rsplit("/", 1)[1] for run in runs] == ["conc_4"] assert all(f"--expected-producer-sha {commit}" in run for run in runs) required = lane == "agentx" or require_power == "1" assert all(run.endswith("--require-power") is required for run in runs) @@ -420,11 +420,17 @@ def test_eval_only_runs_the_eval_recipe_with_real_verification(harness): "roles:\n decode:\n env:\n TLLM_SPEC_DECODE_FORCE_NUM_ACCEPTED_TOKENS: 2\n KEEP: 1\n" ) (mirror / "eval.yaml").write_text(LANE_RECIPE) + # What the in-container eval staged: its own batch record, but not every workflow input. + staged = { + "eval_suite": "kimi_tool_call_schema", "recipe_fingerprint": "", "conc": 4, + "eval_concs": [4, 8], "completed_eval_concs": [8], "failed_eval_concs": [4], + "infmax_model_prefix": "unknown", + } # fmt: skip env = lane_env( harness, "gb300-nv", MODEL_PREFIX="dsv4", PRECISION="fp4", FRAMEWORK="dynamo-trt", MODEL="deepseek-ai/DeepSeek-V4-Pro", IS_AGENTIC="1", SPEC_DECODING="mtp", ISL="0", OSL="0", EVAL_ONLY="true", EVAL_CONFIG_FILE="recipes/test/eval.yaml", FAKE_RESULTS="eval", - EVAL_CONC="4 8", + EVAL_CONC="4 8", RECIPE_FINGERPRINT="recipe-fixture", FAKE_EVAL_META=json.dumps(staged), ) # fmt: skip assert_ok(launch(env, harness.config, harness.workspace)) @@ -439,7 +445,16 @@ def test_eval_only_runs_the_eval_recipe_with_real_verification(harness): assert srtslurm(checkout)["default_time_limit"] == "8:00:00" workspace = harness.workspace assert (workspace / "results_gsm8k.json").read_text() == "{}" - assert json.loads((workspace / "meta_env.json").read_text()) == {"conc": "4 8"} + # The host refresh keeps the eval's record and restates identity from the workflow. + meta = json.loads((workspace / "meta_env.json").read_text()) + kept = ("eval_suite", "conc", "eval_concs", "completed_eval_concs", "failed_eval_concs") + assert {key: meta[key] for key in kept} == { + "eval_suite": "kimi_tool_call_schema", "conc": 4, "eval_concs": [4, 8], + "completed_eval_concs": [8], "failed_eval_concs": [4], + } # fmt: skip + assert (meta["infmax_model_prefix"], meta["recipe_fingerprint"], meta["framework"]) == ( + "dsv4", "recipe-fixture", "dynamo-trt", + ) assert list(workspace.glob("point-identity_*.json")) == [] diff --git a/inferencex-e2e/infx/tests/results/agentic/test_power_lifecycle.py b/inferencex-e2e/infx/tests/results/agentic/test_power_lifecycle.py deleted file mode 100644 index 35f16c7bc5..0000000000 --- a/inferencex-e2e/infx/tests/results/agentic/test_power_lifecycle.py +++ /dev/null @@ -1,347 +0,0 @@ -"""Shell-contract tests for the shared AgentX power lifecycle.""" - -from __future__ import annotations - -import json -import os -import re -import signal -import subprocess -import sys -import time -from pathlib import Path - -import pytest - -REPO_ROOT = Path(__file__).resolve().parents[4] -BENCHMARK_LIB = REPO_ROOT / "benchmarks" / "benchmark_lib.sh" - - -def _run_lifecycle( - tmp_path: Path, - *, - replay_rc: int = 0, - is_multinode: bool = False, - enable_power: bool = True, - require_power: bool = False, - formal_multinode_power: bool = False, - real_power_adapter: bool = False, - missing_power_env: str | None = None, -) -> subprocess.CompletedProcess[str]: - result_dir = tmp_path / "results" - result_dir.mkdir() - event_log = tmp_path / "events.log" - formal_window_dir = str(tmp_path / "power/windows") if formal_multinode_power else "" - script = f""" -source {str(BENCHMARK_LIB)!r} -start_gpu_monitor() {{ - printf 'monitor-start:%s\n' "$*" >> {str(event_log)!r} - printf 'timestamp,index,power.draw [W]\n' > "$2" -}} -stop_gpu_monitor() {{ printf 'monitor-stop\n' >> {str(event_log)!r}; }} -fake_replay() {{ - printf 'replay\n' >> {str(event_log)!r} - return {replay_rc} -}} -write_agentic_result_json() {{ - printf 'aggregate\n' >> {str(event_log)!r} - printf '{{}}\n' > "$AGENTIC_OUTPUT_DIR/$RESULT_FILENAME.json" -}} -fake_python() {{ - case "$*" in - *infx.results.agentic.power_adapter*) - printf 'adapter:%s\n' "$*" >> {str(event_log)!r} - if [ {'1' if real_power_adapter else '0'} = 1 ]; then - PYTHONPATH={str(REPO_ROOT)!r} {sys.executable!r} "$@" - return $? - fi - ;; - *validate_agentic_result*) - printf 'validate\n' >> {str(event_log)!r} - ;; - *) - printf 'analyze\n' >> {str(event_log)!r} - ;; - esac - return 0 -}} -validate_required_agentic_server_metrics() {{ - printf 'server-metrics\n' >> {str(event_log)!r} -}} -trap 'printf "parent-exit\\n" >> {str(event_log)!r}' EXIT -REPLAY_CMD=fake_replay -AIPERF_PYTHON=fake_python -INFMAX_CONTAINER_WORKSPACE={str(tmp_path)!r} -AGENTIC_OUTPUT_DIR={str(tmp_path)!r} -RESULT_FILENAME=agg_agentx -AIPERF_FAILED_REQUEST_THRESHOLD=0 -TP=3 -PP_SIZE=2 -PCP_SIZE=2 -IS_MULTINODE={'true' if is_multinode else 'false'} -ENABLE_AGENTX_POWER={'1' if enable_power else '0'} -REQUIRE_POWER={'1' if require_power else '0'} -CONC=8 -SRT_MEASUREMENT_WINDOW_DIR={formal_window_dir!r} -{f'unset {missing_power_env}' if missing_power_env else ''} -set +e -run_agentic_replay_and_write_outputs {str(result_dir)!r} -rc=$? -exit "$rc" -""" - return subprocess.run( - ["bash", "-c", script], - env={ - **os.environ, - "PATH": "/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - }, - capture_output=True, - text=True, - check=False, - ) - - -def _events(tmp_path: Path) -> list[str]: - return (tmp_path / "events.log").read_text().splitlines() - - -@pytest.mark.parametrize( - ("replay_rc", "expected_rc"), - [(0, 0), (7, 7)], -) -def test_single_node_monitor_wraps_replay_and_stops_once( - tmp_path: Path, replay_rc: int, expected_rc: int -): - result = _run_lifecycle(tmp_path, replay_rc=replay_rc) - - assert result.returncode == expected_rc, result.stderr - events = _events(tmp_path) - assert events.count("monitor-stop") == 1 - assert events.index("monitor-start:--output " + str(tmp_path / "results/gpu_metrics.csv")) < events.index( - "replay" - ) - assert events.index("replay") < events.index("monitor-stop") - assert events.index("monitor-stop") < events.index("aggregate") - assert (tmp_path / "results/gpu_metrics.csv").is_file() - captured_offset = (tmp_path / "results/agentic_power_timezone_offset.txt").read_text().strip() - assert re.fullmatch(r"[+-]\d{4}", captured_offset) - assert events[-1] == "parent-exit" - - -def test_single_node_invokes_adapter_with_gpu_shape_and_strict_mode(tmp_path: Path): - result = _run_lifecycle(tmp_path, require_power=True) - - assert result.returncode == 0, result.stderr - adapter_event = next(event for event in _events(tmp_path) if event.startswith("adapter:")) - assert "--result-dir " + str(tmp_path / "results") in adapter_event - assert "--agg-result " + str(tmp_path / "agg_agentx.json") in adapter_event - assert "--expected-num-gpus 12" in adapter_event - assert "--require-power" in adapter_event - - -@pytest.mark.parametrize("missing_power_env", ["TP", "PP_SIZE", "PCP_SIZE"]) -@pytest.mark.parametrize("require_power", [False, True]) -def test_missing_power_shape_fails_before_monitor_or_replay( - tmp_path: Path, missing_power_env: str, require_power: bool -): - result = _run_lifecycle( - tmp_path, missing_power_env=missing_power_env, require_power=require_power - ) - - assert result.returncode == 1 - assert f" - {missing_power_env}" in result.stdout - assert _events(tmp_path) == ["parent-exit"] - - -def test_explicit_opt_out_skips_power(tmp_path: Path): - result = _run_lifecycle( - tmp_path, - enable_power=False, - missing_power_env="PP_SIZE", - ) - - assert result.returncode == 0, result.stderr - events = _events(tmp_path) - assert not any(event.startswith("monitor-") for event in events) - assert not any(event.startswith("adapter:") for event in events) - - -@pytest.mark.parametrize("require_power", [False, True]) -def test_multinode_missing_contract_records_invalid_power_and_enforces_strict_mode( - tmp_path: Path, require_power: bool -): - result = _run_lifecycle( - tmp_path, - is_multinode=True, - require_power=require_power, - real_power_adapter=True, - ) - - assert result.returncode == int(require_power), result.stderr - events = _events(tmp_path) - assert not any(event.startswith("monitor-") for event in events) - adapter_event = next(event for event in events if event.startswith("adapter:")) - assert events.index("aggregate") < events.index(adapter_event) - aggregate = json.loads((tmp_path / "agg_agentx.json").read_text()) - validation = json.loads((tmp_path / "results/power_validation.json").read_text()) - assert aggregate["power_valid"] == 0 - assert "total_gpu_energy_j" not in aggregate - assert validation["power_valid"] is False - assert validation["reasons"] == ["multinode_power_contract_missing"] - - -@pytest.mark.parametrize("identity_fails", [False, True]) -def test_nvidia_monitor_preserves_identity_without_requiring_it( - tmp_path: Path, identity_fails: bool -): - metrics_path = tmp_path / "gpu_metrics.csv" - script = f""" -source {str(BENCHMARK_LIB)!r} -nvidia-smi() {{ - case "$*" in - --query-gpu=index,uuid,pci.bus_id,name,driver_version*) - printf 'index, uuid, pci.bus_id, name, driver_version\\n' - if [ {'1' if identity_fails else '0'} = 1 ]; then return 1; fi - printf '0, GPU-device-a, 00000000:01:00.0, NVIDIA Test GPU, 590.00\\n' - ;; - *) - printf '2026/09/09 00:00:00.000, 0, 200 W, 40, 1500, 1200, 80, 30\\n' - if [[ "$*" == *" -l "* ]]; then exec sleep 30; fi - ;; - esac -}} -set -e -start_gpu_monitor --output {str(metrics_path)!r} -stop_gpu_monitor -""" - result = subprocess.run( - ["bash", "-c", script], - env={**os.environ, "PATH": "/usr/bin:/bin"}, - capture_output=True, - text=True, - check=False, - timeout=5, - ) - - assert result.returncode == 0, result.stderr - identity_path = tmp_path / "gpu_metrics_identity.csv" - if identity_fails: - assert not identity_path.exists() - assert "NVIDIA identity sidecar failed" in result.stderr - else: - assert "0, GPU-device-a, 00000000:01:00.0, NVIDIA Test GPU, 590.00" in identity_path.read_text() - assert "Started NVIDIA" in result.stdout - assert "Stopped" in result.stdout - - -def test_multinode_formal_window_wraps_replay_without_local_monitor(tmp_path: Path): - result = _run_lifecycle( - tmp_path, - is_multinode=True, - formal_multinode_power=True, - require_power=True, - ) - - assert result.returncode == 0, result.stderr - events = _events(tmp_path) - assert not any(event.startswith("monitor-") for event in events) - adapters = [event for event in events if event.startswith("adapter:")] - assert len(adapters) == 2 - assert "--write-multinode-window running" in adapters[0] - assert "--write-multinode-window completed" in adapters[1] - assert "--concurrency 8" in adapters[0] - assert "--require-power" in adapters[0] - assert events.index(adapters[0]) < events.index("replay") - assert events.index("aggregate") < events.index(adapters[1]) - captured_offset = (tmp_path / "results/agentic_power_timezone_offset.txt").read_text().strip() - assert re.fullmatch(r"[+-]\d{4}", captured_offset) - - -def test_multinode_formal_window_is_left_running_when_replay_is_interrupted(tmp_path: Path): - result = _run_lifecycle( - tmp_path, - replay_rc=143, - is_multinode=True, - formal_multinode_power=True, - require_power=True, - ) - - assert result.returncode == 143, result.stderr - adapters = [event for event in _events(tmp_path) if event.startswith("adapter:")] - assert len(adapters) == 1 - assert "--write-multinode-window running" in adapters[0] - - -@pytest.mark.parametrize( - ("sent_signal", "expected_rc"), - [(signal.SIGINT, 130), (signal.SIGTERM, 143)], -) -def test_signal_stops_monitor_once_without_replacing_parent_trap( - tmp_path: Path, sent_signal: signal.Signals, expected_rc: int -): - result_dir = tmp_path / "results" - result_dir.mkdir() - event_log = tmp_path / "events.log" - script = f""" -source {str(BENCHMARK_LIB)!r} -start_gpu_monitor() {{ - printf 'monitor-pid:%s\n' "${{BASHPID:-$$}}" >> {str(event_log)!r} -}} -stop_gpu_monitor() {{ printf 'monitor-stop\n' >> {str(event_log)!r}; }} -fake_replay() {{ - exec {sys.executable!r} -c ' -import signal, sys, time -signal.signal(signal.SIGINT, signal.SIG_DFL) -signal.signal(signal.SIGTERM, signal.SIG_DFL) -print("replay-ready", file=open(sys.argv[1], "a"), flush=True) -time.sleep(30) -' {str(event_log)!r} -}} -trap 'printf "parent-exit\\n" >> {str(event_log)!r}' EXIT -trap 'printf "parent-int\\n" >> {str(event_log)!r}; exit 130' INT -trap 'printf "parent-term\\n" >> {str(event_log)!r}; exit 143' TERM -REPLAY_CMD=fake_replay -ENABLE_AGENTX_POWER=1 -REQUIRE_POWER=0 -IS_MULTINODE=false -TP=1 -PP_SIZE=1 -PCP_SIZE=1 -run_agentic_replay_and_write_outputs {str(result_dir)!r} -""" - proc = subprocess.Popen( - ["bash", "-c", script], - env={**os.environ, "PATH": "/usr/bin:/bin"}, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - start_new_session=True, - ) - try: - # The monitor starts before the production signal traps are installed. - # Publish readiness from the execed process after restoring signal handling; - # a shell marker before exec races with the group SIGINT. - deadline = time.monotonic() + 5 - while time.monotonic() < deadline: - if event_log.exists() and "replay-ready" in event_log.read_text().splitlines(): - break - time.sleep(0.01) - else: - pytest.fail("replay did not start") - - os.killpg(proc.pid, sent_signal) - _, stderr = proc.communicate(timeout=5) - finally: - try: - os.killpg(proc.pid, signal.SIGKILL) - except ProcessLookupError: - pass - proc.communicate() - - assert proc.returncode == expected_rc, stderr - events = _events(tmp_path) - assert events.count("monitor-stop") == 1 - expected_parent_event = "parent-int" if sent_signal == signal.SIGINT else "parent-term" - assert expected_parent_event in events - assert events[-1] == "parent-exit" diff --git a/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py b/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py index 788dab7cc5..659f11676a 100644 --- a/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py +++ b/inferencex-e2e/infx/tests/results/power/test_aggregate_power.py @@ -1062,7 +1062,7 @@ def _write_snapshot_pair( start: dict[str, float], end: dict[str, float], ) -> None: - """Sidecars named exactly as start_gpu_monitor/stop_gpu_monitor write them.""" + """Sidecars named exactly as infx.bench.gpu_monitor writes them.""" _write_energy_snapshot(csv.parent / "gpu_metrics_energy_start.csv", start) _write_energy_snapshot(csv.parent / "gpu_metrics_energy_end.csv", end) diff --git a/inferencex-e2e/infx/tests/results/power/test_process_result.py b/inferencex-e2e/infx/tests/results/power/test_process_result.py index 601bbcc04e..99dd2afc64 100644 --- a/inferencex-e2e/infx/tests/results/power/test_process_result.py +++ b/inferencex-e2e/infx/tests/results/power/test_process_result.py @@ -1,11 +1,8 @@ """Exercise the fixed-sequence module CLI with controlled environment and artifacts.""" import json import os -import select -import signal import subprocess import sys -import time from pathlib import Path import pytest @@ -1023,323 +1020,6 @@ def test_multinode_internal_error_preserves_validation( "message": "forced import failure" if fail_import else "forced aggregation failure", } - def test_amd_csv_filter_streams_complete_rows_before_eof(self): - """A live producer must not leave telemetry buffered until shutdown.""" - benchmark_lib = REPO_ROOT / "benchmarks/benchmark_lib.sh" - expected = b"timestamp,gpu,socket_power\n123,0,400\n124,0,410\n" - with subprocess.Popen( - ["bash", "-c", f"source {str(benchmark_lib)!r}; _filter_amd_smi_metrics"], - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - env={"PATH": os.environ["PATH"], "PYTHONDONTWRITEBYTECODE": "1"}, - ) as process: - try: - process.stdin.write( - b"diagnostic before header\ntimestamp,gpu,socket_power\n123,0,400\n" - b"timestamp,gpu,socket_power\n124,0,410\n125,0,4" - ) - process.stdin.flush() - received = b"" - deadline = time.monotonic() + 5 - while len(received) < len(expected): - ready, _, _ = select.select( - [process.stdout], [], [], max(0, deadline - time.monotonic()) - ) - assert ready, "CSV rows remained buffered while the producer was open" - chunk = os.read(process.stdout.fileno(), 4096) - assert chunk, "filter exited before consuming the live stream" - received += chunk - assert received == expected - process.stdin.close() - assert process.wait(timeout=5) == 0 - assert process.stdout.read() == b"" # Discard the incomplete final row. - finally: - if process.poll() is None: - process.kill() - process.wait(timeout=5) - - def test_stop_gpu_monitor_appends_final_nvidia_sample(self, tmp_path): - """Stopping between 1 Hz ticks still records one post-benchmark sample.""" - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - args_log = tmp_path / "nvidia_args.txt" - fake_nvidia_smi = fake_bin / "nvidia-smi" - fake_nvidia_smi.write_text( - "#!/usr/bin/env bash\n" - f"printf '%s\\n' \"$*\" > {str(args_log)!r}\n" - "printf '%s\\n' " - "'2026/07/23 12:00:11.000, 0, 500.00 W, 65, 1000, 1000, 90 %, 10 %'\n" - ) - fake_nvidia_smi.chmod(0o755) - metrics = tmp_path / "gpu_metrics.csv" - metrics.write_text( - "timestamp, index, power.draw [W], temperature.gpu, " - "clocks.current.sm [MHz], clocks.current.memory [MHz], " - "utilization.gpu [%], utilization.memory [%]\n" - ) - benchmark_lib = Path(__file__).parents[4] / "benchmarks/benchmark_lib.sh" - script = f""" -source {str(benchmark_lib)!r} -kill() {{ return 0; }} -wait() {{ return 0; }} -GPU_MONITOR_PID=999 -GPU_MONITOR_VENDOR=nvidia -GPU_METRICS_CSV={str(metrics)!r} -stop_gpu_monitor -""" - env = { - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - } - - result = subprocess.run( - ["bash", "-c", script], - env=env, - capture_output=True, - text=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert "--format=csv,noheader" in args_log.read_text() - assert "2026/07/23 12:00:11.000, 0, 500.00 W" in metrics.read_text() - - def test_stop_gpu_monitor_drops_truncated_row_before_final_sample(self, tmp_path): - """A killed monitor cannot concatenate its partial row with the final sample.""" - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - fake_nvidia_smi = fake_bin / "nvidia-smi" - final_sample = ( - "2026/07/23 12:00:11.000, 0, 500.00 W, " - "65, 1000, 1000, 90 %, 10 %" - ) - fake_nvidia_smi.write_text( - "#!/usr/bin/env bash\n" - f"printf '%s\\n' {final_sample!r}\n" - ) - fake_nvidia_smi.chmod(0o755) - - header = ( - "timestamp, index, power.draw [W], temperature.gpu, " - "clocks.current.sm [MHz], clocks.current.memory [MHz], " - "utilization.gpu [%], utilization.memory [%]" - ) - complete_sample = ( - "2026/07/23 12:00:09.000, 0, 490.00 W, " - "64, 990, 990, 89 %, 9 %" - ) - truncated_sample = "2026/07/23 12:00:10.000, 0, 52" - metrics = tmp_path / "gpu_metrics.csv" - metrics.write_text( - f"{header}\n{complete_sample}\n{truncated_sample}" - ) - - benchmark_lib = Path(__file__).parents[4] / "benchmarks/benchmark_lib.sh" - script = f""" -source {str(benchmark_lib)!r} -kill() {{ return 0; }} -wait() {{ return 0; }} -GPU_MONITOR_PID=999 -GPU_MONITOR_VENDOR=nvidia -GPU_METRICS_CSV={str(metrics)!r} -stop_gpu_monitor -""" - env = { - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - } - - result = subprocess.run( - ["bash", "-c", script], - env=env, - capture_output=True, - text=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert metrics.read_text().splitlines() == [ - header, - complete_sample, - final_sample, - ] - - def test_stop_gpu_monitor_amd_waits_one_tick_and_snapshots_energy(self, tmp_path): - """AMD stop lets the watch stream bracket the window, then snapshots energy.""" - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - args_log = tmp_path / "amd_args.txt" - fake_amd_smi = fake_bin / "amd-smi" - fake_amd_smi.write_text( - "#!/usr/bin/env bash\n" - f"printf '%s\\n' \"$*\" >> {str(args_log)!r}\n" - "printf 'gpu,total_energy_consumption\\n0,178319501.7\\n'\n" - ) - fake_amd_smi.chmod(0o755) - sleep_log = tmp_path / "sleep_args.txt" - contents = "timestamp,gpu,socket_power\n1785881113,0,238\n" - metrics = tmp_path / "gpu_metrics.csv" - metrics.write_text(contents) - benchmark_lib = Path(__file__).parents[4] / "benchmarks/benchmark_lib.sh" - script = f""" -source {str(benchmark_lib)!r} -kill() {{ return 0; }} -wait() {{ return 0; }} -sleep() {{ printf '%s\\n' "$1" > {str(sleep_log)!r}; }} -GPU_MONITOR_PID=999 -GPU_MONITOR_VENDOR=amd -GPU_MONITOR_INTERVAL=3 -GPU_METRICS_CSV={str(metrics)!r} -stop_gpu_monitor -""" - env = { - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - } - - result = subprocess.run( - ["bash", "-c", script], - env=env, - capture_output=True, - text=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert sleep_log.read_text().strip() == "5" - assert metrics.read_text() == contents - assert "metric -E --csv" in args_log.read_text() - energy_end = tmp_path / "gpu_metrics_energy_end.csv" - assert energy_end.read_text().startswith("gpu,total_energy_consumption") - - def test_stop_gpu_monitor_amd_drops_truncated_row_without_append(self, tmp_path): - """The AMD path repairs a partial trailing row but appends no sample.""" - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - fake_amd_smi = fake_bin / "amd-smi" - fake_amd_smi.write_text( - "#!/usr/bin/env bash\n" - "printf 'gpu,total_energy_consumption\\n0,178319501.7\\n'\n" - ) - fake_amd_smi.chmod(0o755) - header = "timestamp,gpu,socket_power" - complete_sample = "1785881113,0,238" - metrics = tmp_path / "gpu_metrics.csv" - metrics.write_text(f"{header}\n{complete_sample}\n1785881114,0,2") - benchmark_lib = Path(__file__).parents[4] / "benchmarks/benchmark_lib.sh" - script = f""" -source {str(benchmark_lib)!r} -kill() {{ return 0; }} -wait() {{ return 0; }} -sleep() {{ return 0; }} -GPU_MONITOR_PID=999 -GPU_MONITOR_VENDOR=amd -GPU_METRICS_CSV={str(metrics)!r} -stop_gpu_monitor -""" - env = { - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - } - - result = subprocess.run( - ["bash", "-c", script], - env=env, - capture_output=True, - text=True, - check=False, - ) - - assert result.returncode == 0, result.stderr - assert metrics.read_text().splitlines() == [header, complete_sample] - assert (tmp_path / "gpu_metrics_energy_end.csv").exists() - - def test_start_stop_gpu_monitor_amd_lifecycle(self, tmp_path): - """Watch rows survive the kill and both boundary snapshots are written.""" - fake_bin = tmp_path / "bin" - fake_bin.mkdir() - fake_amd_smi = fake_bin / "amd-smi" - fake_amd_smi.write_text( - "#!/usr/bin/env bash\n" - "set -e\n" - 'if [[ "$*" == *" -w "* ]]; then\n' - " echo \"'CTRL' + 'C' to stop watching output:\"\n" - " echo 'timestamp,gpu,socket_power,power_management'\n" - " while :; do\n" - " echo \"$(date +%s),0,238,ENABLED\"\n" - " echo 'timestamp,gpu,socket_power,power_management'\n" - " sleep 0.01\n" - " done\n" - 'elif [[ "$*" == *"metric -E --csv"* ]]; then\n' - " printf 'gpu,total_energy_consumption\\n0,178319501.7\\n'\n" - 'elif [[ "$1" == "static" ]]; then\n' - " printf '{\"gpu_data\": []}\\n'\n" - "fi\n" - ) - fake_amd_smi.chmod(0o755) - metrics = tmp_path / "gpu_metrics.csv" - benchmark_lib = Path(__file__).parents[4] / "benchmarks/benchmark_lib.sh" - script = f""" -source {str(benchmark_lib)!r} -# Wait for observable pipeline output instead of a fixed sampling delay. -sleep() {{ - for _ in $(seq 1 500); do - if [[ -f "$GPU_METRICS_CSV" ]] && [[ $(wc -l < "$GPU_METRICS_CSV") -ge 3 ]]; then - return 0 - fi - command sleep 0.01 - done - echo "monitor did not emit two samples" >&2 - exit 1 -}} -start_gpu_monitor --output {str(metrics)!r} --interval 1 -monitor_pid=$GPU_MONITOR_PID -stop_gpu_monitor -if kill -0 "$monitor_pid" 2>/dev/null; then - echo "monitor survived stop_gpu_monitor" >&2 - exit 1 -fi -""" - env = { - "PATH": f"{fake_bin}:/usr/bin:/bin", - "PYTHONDONTWRITEBYTECODE": "1", - } - - proc = subprocess.Popen( - ["bash", "-c", script], - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - start_new_session=True, - ) - try: - _, stderr = proc.communicate(timeout=10) - finally: - # Reap the fake producer even if the pipeline or stop logic regresses. - try: - os.killpg(proc.pid, signal.SIGKILL) - except ProcessLookupError: - pass - proc.communicate() - - assert proc.returncode == 0, stderr - lines = metrics.read_text().splitlines() - assert lines[0] == "timestamp,gpu,socket_power,power_management" - assert sum(1 for line in lines if line.startswith("timestamp,")) == 1 - assert "CTRL" not in metrics.read_text() - data_rows = [line for line in lines[1:] if line] - assert len(data_rows) >= 2 - assert all(row.split(",")[2] == "238" for row in data_rows) - energy_start = tmp_path / "gpu_metrics_energy_start.csv" - assert energy_start.read_text().startswith("gpu,total_energy_consumption") - assert (tmp_path / "gpu_metrics_energy_end.csv").exists() - identity = json.loads((tmp_path / "gpu_metrics_identity.json").read_text()) - assert identity == {"gpu_data": []} - - - class TestMultinodePower: """End-to-end wiring: infx.results.fixed_sequence invokes aggregate_power_multinode.py diff --git a/inferencex-e2e/infx/tests/srt_slurm/test_fixed_sequence_client.py b/inferencex-e2e/infx/tests/srt_slurm/test_fixed_sequence_client.py deleted file mode 100644 index 2704393c5a..0000000000 --- a/inferencex-e2e/infx/tests/srt_slurm/test_fixed_sequence_client.py +++ /dev/null @@ -1,66 +0,0 @@ -"""Exercise the custom benchmark shell entrypoint without launching a client.""" - -import json -import os -import subprocess -import sys -from pathlib import Path - -import pytest - - -@pytest.mark.parametrize("explicit", [True, False]) -def test_client_model_discovery_and_explicit_request_count(tmp_path, explicit): - binaries = tmp_path / "bin" - binaries.mkdir() - scripts = { - "python3": ( - f"#!{sys.executable}\n" - "import json, os, sys\n" - "from pathlib import Path\n" - "if sys.argv[1:2] == ['-c']:\n" - f" os.execv({sys.executable!r}, [{sys.executable!r}, *sys.argv[1:]])\n" - "Path(os.environ['CLIENT_ARGS_FILE']).write_text(json.dumps(sys.argv[1:]))\n" - ), - "curl": ( - "#!/bin/sh\n" - 'printf called > "$CURL_CALLED_FILE"\n' - "printf '%s\\n' '{\"data\":[{\"id\":\"discovered-model\"}]}'\n" - ), - # No real client writes results; avoid creating its container-only /logs directory. - "mkdir": "#!/bin/sh\nexit 0\n", - } - for name, source in scripts.items(): - binary = binaries / name - binary.write_text(source) - binary.chmod(0o755) - args_file = tmp_path / "client-args.json" - curl_called = tmp_path / "curl-called" - env = { - "PATH": f"{binaries}{os.pathsep}{os.environ['PATH']}", - "CLIENT_ARGS_FILE": str(args_file), - "CURL_CALLED_FILE": str(curl_called), - "ISL": "1024", - "OSL": "1024", - "SRT_FRONTEND_HOST": "router", - "SRT_FRONTEND_PORT": "8123", - "CONC_LIST": "1", - "PREFILL_NUM_WORKERS": "1", - "PREFILL_TP": "8", - "DECODE_NUM_WORKERS": "1", - "DECODE_TP": "8", - "CLIENT_BACKEND": "openai-chat", - "USE_CHAT_TEMPLATE": "true", - "SERVED_MODEL_NAME": "launcher-model-not-client-override", - } - if explicit: - env.update(BENCHMARK_SERVED_MODEL_NAME="glm5", NUM_PROMPTS="16") - script = Path(__file__).resolve().parents[3] / "benchmarks/multi_node/srt_fixed_sequence.sh" - subprocess.run(["bash", str(script)], env=env, cwd=tmp_path, check=True, capture_output=True) - args = json.loads(args_file.read_text()) - assert args[:2] == ["-m", "infx.bench_serving.benchmark_serving"] - assert args[args.index("--model") + 1] == ("glm5" if explicit else "discovered-model") - assert args[args.index("--num-prompts") + 1] == ("16" if explicit else "10") - assert args[args.index("--endpoint") + 1] == "/v1/chat/completions" - assert "--use-chat-template" in args - assert curl_called.exists() is not explicit diff --git a/inferencex-e2e/infx/tests/workflows/test_launch_layout.py b/inferencex-e2e/infx/tests/workflows/test_launch_layout.py index cfebb96bc0..1e3ab04092 100644 --- a/inferencex-e2e/infx/tests/workflows/test_launch_layout.py +++ b/inferencex-e2e/infx/tests/workflows/test_launch_layout.py @@ -136,53 +136,3 @@ def test_launch_runs_the_python_entrypoint_from_the_project_root(tmp_path, workf if workflow == "profile.yml": output = dict(line.split("=", 1) for line in github_output.read_text().splitlines()) assert Path(output["trace"]).read_bytes() == b"fixture trace" - - -@pytest.mark.parametrize( - "agentic,eval_only,conc,conc_list,launches", - [ - (True, False, "4", "4", True), - (True, False, "4", "4 8", False), - (True, False, "4", "8", False), - (True, True, "4", "4 8", True), - (False, False, "", "4 8", True), - ], -) -def test_multinode_launch_isolates_agentx_throughput( - tmp_path, agentic, eval_only, conc, conc_list, launches -): - """Reject invalid throughput jobs before launching, without changing eval batching.""" - marker = tmp_path / "launched" - launcher = "import os, pathlib, sys\npathlib.Path(os.environ['LAUNCH_MARKER']).touch()\nsys.exit(77)\n" - checkout, _ = measured_checkout(tmp_path, launcher) - step = workflow_step("benchmark-multinode-tmpl.yml", "Launch multi-node job script") - result = subprocess.run( - ["bash", "--noprofile", "--norc", "-e", "-o", "pipefail", "-c", step["run"]], - cwd=checkout, - env={ - **os.environ, - "LAUNCH_MARKER": str(marker), - "INFERENCEX_LAUNCH_PYTHON": sys.executable, - "GITHUB_WORKSPACE": str(checkout), - "GITHUB_ENV": str(tmp_path / "github-env"), - "RESULT_FILENAME_BASE": "agentx-isolation", - "RECIPE_FINGERPRINT": "", - "PREFILL_ADDITIONAL_SETTINGS": "[]", - "DECODE_ADDITIONAL_SETTINGS": "[]", - "IS_AGENTIC": "1" if agentic else "0", - "SCENARIO_TYPE": "agentic-coding" if agentic else "fixed-seq-len", - "EVAL_ONLY": "true" if eval_only else "false", - "CONC": conc, - "CONC_LIST": conc_list, - "EVAL_CONC": "4", - "RUNNER_NAME": "fixture_01", - "VALIDATION_BENCHMARK_LIB": str(ROOT / "inferencex-e2e/benchmarks/benchmark_lib.sh"), - }, - capture_output=True, - text=True, - timeout=30, - ) - assert result.returncode == (77 if launches else 1), result.stdout + result.stderr - assert marker.exists() is launches - if not launches: - assert "AgentX" in result.stderr diff --git a/inferencex-e2e/runners/srt-slurm/hooks/mi355x-amds/check-rdma.sh b/inferencex-e2e/runners/srt-slurm/hooks/mi355x-amds/check-rdma.sh index 2405301f00..8fb4cffe89 100755 --- a/inferencex-e2e/runners/srt-slurm/hooks/mi355x-amds/check-rdma.sh +++ b/inferencex-e2e/runners/srt-slurm/hooks/mi355x-amds/check-rdma.sh @@ -6,7 +6,7 @@ set -eo pipefail log() { printf '[%s] %s\n' "$(hostname -s)" "$*"; } fail() { log "RDMA preflight failed: $*" >&2; exit 1; } -source "$(dirname "${BASH_SOURCE[0]}")/../../../../benchmarks/benchmark_lib.sh" --validation-only +source "$(dirname "${BASH_SOURCE[0]}")/../../../../benchmarks/check_env.sh" check_env_vars IBDEVICES expected_devices="$IBDEVICES" IFS=',' read -r -a devices <<< "$expected_devices" From ebfbb8b9b92ea7f74698aa8a157a319ad0400b5f Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Fri, 2 Oct 2026 12:43:47 -0500 Subject: [PATCH 2/2] chore(agentx): drop the post-run plot step from infx.bench.agentic Follows the plot removal in the base PR: no generate_aiperf_plots call and no matplotlib in the AIPerf client venv. --- inferencex-e2e/infx/bench/agentic/run.py | 4 +--- inferencex-e2e/infx/bench/agentic/venv.py | 1 - .../infx/tests/bench/test_agentic_command.py | 17 +++++++---------- 3 files changed, 8 insertions(+), 14 deletions(-) diff --git a/inferencex-e2e/infx/bench/agentic/run.py b/inferencex-e2e/infx/bench/agentic/run.py index 12d50c5f68..673b464837 100644 --- a/inferencex-e2e/infx/bench/agentic/run.py +++ b/inferencex-e2e/infx/bench/agentic/run.py @@ -198,11 +198,9 @@ def _open_power_window(plan: Plan, python: str, env: Mapping[str, str]) -> int: def _score(plan: Plan, python: str, env: Mapping[str, str], replay_rc: int) -> int: - """Aggregate, plot, audit power, and validate; return the first failure by precedence.""" + """Aggregate, audit power, and validate; return the first failure by precedence.""" result_dir, artifacts = str(plan.result_dir), str(plan.replay.artifact_dir) aggregate_rc = _results(python, env, "agentic.process_agentic_result") - # Best effort: the aggregate JSON is the success gate. - _results(python, env, "generate_aiperf_plots", result_dir) audit = _power_audit(plan, replay_rc) power_rc = _power_adapter(plan, python, env, *audit) if audit else 0 _results(python, env, "agentic.analyze_benchmark_distributions", artifacts, "-o", result_dir) diff --git a/inferencex-e2e/infx/bench/agentic/venv.py b/inferencex-e2e/infx/bench/agentic/venv.py index cc7126260c..bf0edd3460 100644 --- a/inferencex-e2e/infx/bench/agentic/venv.py +++ b/inferencex-e2e/infx/bench/agentic/venv.py @@ -24,7 +24,6 @@ "tqdm>=4.66", "datasets>=4.7.0", "tiktoken", - "matplotlib", "huggingface_hub[cli]>=0.25.0", "urllib3", "requests", diff --git a/inferencex-e2e/infx/tests/bench/test_agentic_command.py b/inferencex-e2e/infx/tests/bench/test_agentic_command.py index e40547b4a1..3b4d77bc33 100644 --- a/inferencex-e2e/infx/tests/bench/test_agentic_command.py +++ b/inferencex-e2e/infx/tests/bench/test_agentic_command.py @@ -62,7 +62,6 @@ infx.results.agentic.validate_agentic_result) echo validate >> "$EVENTS"; exit "${VALIDATE_RC:-0}" ;; infx.results.agentic.analyze_benchmark_distributions) echo analyze >> "$EVENTS" ;; - infx.results.generate_aiperf_plots) echo plots >> "$EVENTS" ;; *) echo "unexpected $*" >> "$EVENTS"; exit 99 ;; esac """ @@ -178,7 +177,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m pytest.param( {}, 0, - ["replay", "aggregate agentx", "plots", "analyze", "validate"], + ["replay", "aggregate agentx", "analyze", "validate"], REPLAYED, id="power-off", ), @@ -188,7 +187,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m {"CONC_LIST": "8", "KV_OFFLOADING": "dram", "KV_OFFLOAD_BACKEND": "lmcache", "TOTAL_CPU_DRAM_GB": "2400"}, 0, - ["replay", "aggregate agentx_conc8", "plots", "analyze", "validate"], + ["replay", "aggregate agentx_conc8", "analyze", "validate"], {f"conc_8/{name}" for name in REPLAYED}, id="conc-list-point", ), @@ -196,8 +195,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m {"ENABLE_AGENTX_POWER": "1", "REQUIRE_POWER": "1"}, 0, [ - "gpu-identity", "replay", "gpu-final-sample", "aggregate agentx", "plots", - "adapter --result-dir {results} --agg-result {out}/agentx.json" + "gpu-identity", "replay", "gpu-final-sample", "aggregate agentx", "adapter --result-dir {results} --agg-result {out}/agentx.json" " --expected-num-gpus 12 --require-power", "analyze", "validate", ], @@ -207,7 +205,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m pytest.param( WINDOW, 0, - [f"{MARK} running", "replay", "aggregate agentx_conc8", "plots", f"{MARK} completed", + [f"{MARK} running", "replay", "aggregate agentx_conc8", f"{MARK} completed", "analyze", "validate"], {f"conc_8/{name}" for name in (OFFSET, *REPLAYED)}, id="multi-node-window", @@ -215,7 +213,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m pytest.param( {**WINDOW, "REPLAY_RC": "143"}, 143, - [f"{MARK} running", "replay", "aggregate agentx_conc8", "plots", "analyze", "validate"], + [f"{MARK} running", "replay", "aggregate agentx_conc8", "analyze", "validate"], {f"conc_8/{name}" for name in (OFFSET, *REPLAYED)}, id="failed-replay-leaves-the-window-running", ), @@ -230,8 +228,7 @@ def test_points_that_cannot_be_measured_fail_before_setup(tmp_path, overrides, m {"IS_MULTINODE": "true", "ENABLE_AGENTX_POWER": "1"}, 0, [ - "replay", "aggregate agentx_conc8", "plots", - "adapter --result-dir {results}/conc_8 --agg-result {out}/agentx_conc8.json" + "replay", "aggregate agentx_conc8", "adapter --result-dir {results}/conc_8 --agg-result {out}/agentx_conc8.json" " --multinode-contract-missing", "analyze", "validate", ], @@ -275,7 +272,7 @@ def test_every_step_runs_and_the_first_failure_in_precedence_wins( assert rc == expected assert [event.split()[0] for event in _events(tmp_path)] == [ - "gpu-identity", "replay", "gpu-final-sample", "aggregate", "plots", "adapter", "analyze", + "gpu-identity", "replay", "gpu-final-sample", "aggregate", "adapter", "analyze", "validate", ] # fmt: skip