Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
86 changes: 86 additions & 0 deletions benchmarks/single_node/agentic/kimik3_fp4_mi355x_mtp.sh
Original file line number Diff line number Diff line change
Expand Up @@ -109,12 +109,14 @@ SERVER_LOG="$RESULT_DIR/server.log"
mkdir -p "$RESULT_DIR"

SERVER_PID=""
LMCACHE_PID=""

cleanup_agentic_services() {
local exit_code=$?
trap - EXIT INT TERM
set +e
stop_background_process_tree "$SERVER_PID" "vLLM server" 60
stop_background_process_tree "$LMCACHE_PID" "LMCache server"
exit "$exit_code"
}
trap cleanup_agentic_services EXIT
Expand Down Expand Up @@ -143,6 +145,90 @@ case "${KV_OFFLOAD_BACKEND:-}" in
)
echo "SimpleCPUOffloadConnector: ${CPU_BYTES_PER_RANK} B/rank x ${TP} ranks, lazy_offload=$SIMPLE_LAZY_OFFLOAD"
;;
lmcache)
require_agentic_kv_offload_backend "$KV_OFFLOAD_BACKEND"

# Keep the image's tested torch/ROCm stack and install only LMCache's
# missing runtime dependencies, same as the MiniMax-M3 lmcache arm.
LMCACHE_VERSION="0.5.4rc1"
LMCACHE_ROCM_INDEX="https://github.com/LMCache/LMCache/releases/expanded_assets/v${LMCACHE_VERSION}-rocm"
agentic_pip_install --quiet --no-cache-dir --no-deps \
"sortedcontainers==2.4.0" \
"opentelemetry-exporter-prometheus==0.61b0" \
"cupy-rocm-7-0==14.1.1" \
"lmcache==${LMCACHE_VERSION}" --find-links "$LMCACHE_ROCM_INDEX"
python3 -c \
"import cupy; import lmcache.integration.vllm.lmcache_mp_connector; import opentelemetry.exporter.prometheus" \
>/dev/null

# One MP server for the node, per the Kimi-K3 recipe
# (docs.lmcache.ai/recipes/kimi_k3.html), with --chunk-size sized for
# THIS stack rather than the recipe's CUDA-path 768: the connector
# requires the chunk to be a multiple of every engine KV group's
# tokens_per_block, and the hybrid KDA/MLA layout here registers
# attention groups at 1536 ("Setting attention block size to 1536",
# run 31644990546) plus a KDA state group at 3072 (run 31645828378),
# so 3072 is the minimum valid chunk. The multi-group layout also
# requires one object group per sliding-window size:
# --separate-object-groups.
LMCACHE_PORT=6555
LMCACHE_HTTP_PORT=8090
LMCACHE_LOG="$RESULT_DIR/lmcache_server.log"

# The whole generated node-DRAM budget backs the single server's L1,
# which lives in /dev/shm; fail early if it cannot fit.
LMCACHE_L1_SIZE_GB="$TOTAL_CPU_DRAM_GB"
SHM_FREE_GB=$(df -BG --output=avail /dev/shm 2>/dev/null | tail -1 | tr -dc '0-9')
if [ -n "$SHM_FREE_GB" ] && [ "$SHM_FREE_GB" -gt 0 ]; then
SHM_CAP_GB=$((SHM_FREE_GB * 90 / 100))
if [ "$LMCACHE_L1_SIZE_GB" -gt "$SHM_CAP_GB" ]; then
echo "Error: LMCache L1 ${LMCACHE_L1_SIZE_GB} GB exceeds 90% of free /dev/shm (${SHM_CAP_GB} GB)." >&2
exit 1
fi
fi

LMCACHE_CMD=(
lmcache server
--host 127.0.0.1
--port "$LMCACHE_PORT"
--http-host 127.0.0.1
--http-port "$LMCACHE_HTTP_PORT"
--l1-size-gb "$LMCACHE_L1_SIZE_GB"
--l1-init-size-gb 10
--chunk-size 3072
--separate-object-groups
--enable-extra-logging
--max-cpu-workers 8
--max-gpu-workers 1
--eviction-policy LRU
# Pin the server-driven STORE/RETRIEVE path (same as the MiniMax-M3
# arm) so the benchmark measures one deterministic transfer path
# instead of the auto-mode pair. The L1 stays /dev/shm-backed either
# way (shm_name defaults on), which is why the capacity check above
# applies in this mode too.
--supported-transfer-mode lmcache_driven
)
append_command "$RESULT_DIR/lmcache_command.txt" "${LMCACHE_CMD[@]}"
"${LMCACHE_CMD[@]}" > "$LMCACHE_LOG" 2>&1 &
LMCACHE_PID=$!
wait_for_ready \
--endpoint "http://127.0.0.1:${LMCACHE_HTTP_PORT}/healthcheck" \
--log "$LMCACHE_LOG" \
--pid "$LMCACHE_PID" \
--sleep-interval 1 \
--timeout 600

# 100k-330k-token agentic prefixes make single retrieves large; use the
# same MQ timeout headroom as the MiniMax-M3 arm.
OFFLOAD_ARGS=(
--kv-transfer-config
"{\"kv_connector\":\"LMCacheMPConnector\",\"kv_connector_module_path\":\"lmcache.integration.vllm.lmcache_mp_connector\",\"kv_role\":\"kv_both\",\"kv_connector_extra_config\":{\"lmcache.mp.port\":$LMCACHE_PORT,\"lmcache.mp.mq_timeout\":6000.0}}"
)
;;
*)
echo "Error: unsupported KV_OFFLOAD_BACKEND='$KV_OFFLOAD_BACKEND' (expected vllm-simple or lmcache)" >&2
exit 1
;;
esac
fi

Expand Down
22 changes: 22 additions & 0 deletions configs/amd-master.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -651,6 +651,28 @@ kimik3-fp4-mi355x-vllm-agentic-mtp:
- { tp: 8, kv-offloading: none, conc-list: [1, 4, 8] , spec-decoding: mtp}
- { tp: 8, ep: 1, kv-offloading: dram, kv-offload-backend: { name: vllm-simple }, conc-list: [10], spec-decoding: mtp }

# LMCache MP-server DRAM offload on top of the same DSpark MTP serving stack as
# kimik3-fp4-mi355x-vllm-agentic-mtp (same image, script, and topology). A
# dedicated key so LMCache points can be selected and swept without re-running
# the resident and vllm-simple arms of the base key.
kimik3-fp4-mi355x-vllm-agentic-mtp-lmcache:
image: vllm/vllm-openai-rocm:nightly-cb8104839c141609d99f1254459ef3a4f1bd4263
model: moonshotai/Kimi-K3
model-prefix: kimik3
runner: cluster:mi355x-amds
precision: fp4
framework: vllm
multinode: false
scenarios:
agentic-coding:
# 0.40, not the base key's 0.50: the LMCache L1 is /dev/shm-backed and the
# script refuses budgets above 90% of free shm. mi355x-amds nodes mount
# ~1.5 TB of shm (cap ~1360 GB), so 0.50's 1499 GB budget fails the check
# (run 31644286169); 0.40 -> ~1199 GB fits with margin.
- dram-utilization: 0.40
search-space:
- { tp: 8, ep: 1, kv-offloading: dram, kv-offload-backend: { name: lmcache, version: "0.5.4rc2" }, conc-list: [4, 8, 16], spec-decoding: mtp }

dsr1-fp4-mi355x-sglang-disagg:
image: lmsysorg/sglang-rocm:v0.5.12-rocm720-mi35x-20260519
model: amd/DeepSeek-R1-0528-MXFP4-v2
Expand Down
9 changes: 9 additions & 0 deletions perf-changelog.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5918,3 +5918,12 @@
- "Add a TEP2 arm (tp 2, ep 2) to the qwen3.5-fp4-b200-sglang-mtp 8k/1k sweep at concurrency 16, 32, and 64"
- "Rides on the NVFP4-V2 checkpoint switch from #2205"
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2550

- config-keys:
- kimik3-fp4-mi355x-vllm-agentic-mtp-lmcache
scenario-type:
- agentic-coding
description:
- "Add a dedicated LMCache 0.5.4rc1 DRAM KV-offload key at TP8 conc 4/8/16 on top of the unchanged kimik3-fp4-mi355x-vllm-agentic-mtp DSpark MTP stack."
- "Run one LMCache MP server per node with chunk size 768 (the K3 unified block size at 8 GPUs) and --separate-object-groups for the hybrid KDA/MLA two-group KV layout."
pr-link: https://github.com/SemiAnalysisAI/InferenceX/pull/2583
Loading