From 3e23b763bc050028a77aa7d05be227a78a0deace Mon Sep 17 00:00:00 2001 From: Shimi Bandiel Date: Tue, 6 Oct 2026 11:04:07 -0700 Subject: [PATCH] docs(e2e-deploy): fix the guidellm load test for the v0.7+ CLI The guidellm Job used `guidellm benchmark run` with flat flags, the CLI of guidellm v0.6. guidellm v0.7.0 (2026-06-29) replaced it with `guidellm run` and registry-backed `kind=...` options, so with the rolling `latest` tag the job exits at once with "No such command 'benchmark'" and the saturation test runs with no load. Pin the image to v0.8.0 and express the same benchmark in the new syntax: openai_http backend against the gateway with /v1/completions, constant 50 req/s, a 3600 s duration constraint, and synthetic 256/512-token data. The tokenizer now resolves from the backend model, so `--processor` and the USER variable go away. The guide waits for the job pod to be Ready before sampling, since the first run pulls the image and the tokenizer, and notes why the manifest must not go back to `latest`. Verified against `guidellm mock-server` from the same image: backend validated, constant rate held, 2243 requests served, JSON report written. Signed-off-by: Shimi Bandiel --- docs/guides/e2e-deploy.md | 11 ++++-- docs/guides/e2e-deploy/guidellm-loadtest.yaml | 34 ++++++------------- 2 files changed, 19 insertions(+), 26 deletions(-) diff --git a/docs/guides/e2e-deploy.md b/docs/guides/e2e-deploy.md index 45b16c34..e9dd704a 100644 --- a/docs/guides/e2e-deploy.md +++ b/docs/guides/e2e-deploy.md @@ -393,7 +393,10 @@ so that shell can resolve the Redis service configured on the host. Two load generators are available: - **hey** (`docs/guides/e2e-deploy/hey-loadtest.yaml`) — simple HTTP load generator, 200 concurrent workers - **guidellm** (`docs/guides/e2e-deploy/guidellm-loadtest.yaml`) — LLM-specific load testing with synthetic - prompts (256 prompt tokens, 512 output tokens), constant 50 req/s + prompts (256 prompt tokens, 512 output tokens), constant 50 req/s. The manifest pins guidellm v0.8.0 and + uses the `guidellm run` syntax introduced in v0.7.0 (`--backend`, `--profile`, `--constraint`, `--data` + with `kind=...` values); the older `guidellm benchmark run` flags no longer exist in any current image, so + do not switch it back to the `latest` tag #### Option A: hey @@ -449,8 +452,10 @@ kubectl run --rm -i check-drained --image=redis --restart=Never -n ${NAMESPACE} # 1. Start the load test (constant 50 req/s, synthetic data, runs until killed) kubectl apply -n ${NAMESPACE} -f ${ASYNC_REPO}/docs/guides/e2e-deploy/guidellm-loadtest.yaml -# 2. Wait ~40s for startup + Prometheus scrape, then verify saturation -sleep 40 +# 2. Wait for the job to start generating load (the first run pulls the image and the +# tokenizer), then one Prometheus scrape, then verify saturation +kubectl wait --for=condition=Ready pod -l job-name=guidellm-loadtest -n ${NAMESPACE} --timeout=300s +sleep 30 kubectl run --rm -i prom-running --image=curlimages/curl --restart=Never -n ${NAMESPACE} -- \ curl -s --data-urlencode \ 'query=vllm:num_requests_running{inference_pool="optimized-baseline"}' \ diff --git a/docs/guides/e2e-deploy/guidellm-loadtest.yaml b/docs/guides/e2e-deploy/guidellm-loadtest.yaml index 4d3ac83d..41d8ddde 100644 --- a/docs/guides/e2e-deploy/guidellm-loadtest.yaml +++ b/docs/guides/e2e-deploy/guidellm-loadtest.yaml @@ -9,33 +9,21 @@ spec: restartPolicy: Never containers: - name: guidellm - image: ghcr.io/vllm-project/guidellm:latest + image: ghcr.io/vllm-project/guidellm:v0.8.0 env: - - name: USER - value: "guidellm" - name: HF_HUB_CACHE - value: "/tmp/hf_cache" + value: /tmp/hf_cache command: - guidellm - - benchmark - run - - --target - - "http://llm-d-inference-gateway-istio" - - --request-format - - text_completions - - --model - - "Qwen/Qwen3-0.6B" + - --backend + - kind=openai_http,target=http://llm-d-inference-gateway-istio,model=Qwen/Qwen3-0.6B,request_format=/v1/completions - --profile - - constant - - --rate - - "50" - - --max-seconds - - "3600" + - kind=constant,rate=50 + - --constraint + - kind=max_duration,seconds=3600 - --data - - "prompt_tokens=256,output_tokens=512" - - --output-dir - - /tmp - - --max-requests - - "100000" - - --processor - - "Qwen/Qwen3-0.6B" + - kind=synthetic_text,prompt_tokens=256,output_tokens=512 + - --output + - kind=json,path=/tmp/results.json + - --disable-console-interactive