Run Sweep - Add B300 TP1 DSV4.1 Flash 8k1k serving sweep / 新增 B300 TP1 DSV4.1 Flash 8k1k serving 扫描 #14711
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: "Run Sweep" | |
| run-name: Run Sweep - ${{ github.event.pull_request.title || github.event.head_commit.message }} | |
| concurrency: | |
| group: >- | |
| sweep-${{ github.event.pull_request.number || github.sha }}-${{ | |
| github.event_name == 'pull_request' && | |
| (github.event.action == 'labeled' || github.event.action == 'unlabeled') && | |
| github.event.label.name != 'sweep-enabled' && | |
| github.event.label.name != 'full-sweep-enabled' && | |
| github.event.label.name != 'non-canary-full-sweep-enabled' && | |
| github.event.label.name != 'full-sweep-fail-fast' && | |
| github.event.label.name != 'full-sweep-fail-fast-no-canary' && | |
| github.event.label.name != 'all-evals' && | |
| github.event.label.name != 'evals-only' && | |
| github.event.label.name != 'agentx-fast' && | |
| github.event.label.name != 'skip_queue' && | |
| github.event.label.name != 'ci-patchwork' && | |
| github.event.label.name != 'engine-patch' && | |
| github.event.label.name != 'ci-patchwork-waived' && | |
| github.event.label.name != 'ci-checklist-complete' && | |
| github.run_id || | |
| 'active' | |
| }} | |
| cancel-in-progress: true | |
| on: | |
| push: | |
| branches: | |
| - main | |
| paths: | |
| - "perf-changelog.yaml" | |
| pull_request: | |
| branches: | |
| - main | |
| types: | |
| - synchronize | |
| - labeled | |
| - unlabeled | |
| paths: | |
| - "perf-changelog.yaml" | |
| permissions: | |
| contents: read | |
| jobs: | |
| check-changelog: | |
| name: check-changelog | |
| runs-on: ubuntu-latest | |
| permissions: | |
| contents: read | |
| # Labels opt into sweeps regardless of draft status; readiness only starts reviews. | |
| if: >- | |
| github.event_name == 'pull_request' && | |
| github.event.pull_request.head.repo.full_name == github.repository && | |
| ( | |
| (github.event.action != 'labeled' && github.event.action != 'unlabeled') || | |
| github.event.label.name == 'sweep-enabled' || | |
| github.event.label.name == 'full-sweep-enabled' || | |
| github.event.label.name == 'non-canary-full-sweep-enabled' || | |
| github.event.label.name == 'full-sweep-fail-fast' || | |
| github.event.label.name == 'full-sweep-fail-fast-no-canary' || | |
| github.event.label.name == 'all-evals' || | |
| github.event.label.name == 'evals-only' || | |
| github.event.label.name == 'agentx-fast' || | |
| github.event.label.name == 'skip_queue' || | |
| github.event.label.name == 'ci-patchwork' || | |
| github.event.label.name == 'engine-patch' || | |
| github.event.label.name == 'ci-patchwork-waived' || | |
| github.event.label.name == 'ci-checklist-complete' | |
| ) | |
| outputs: | |
| skip-pr-sweep: ${{ steps.sweep_policy.outputs.skip-pr-sweep }} | |
| steps: | |
| - name: Reject conflicting sweep labels | |
| env: | |
| SWEEP_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }} | |
| run: | | |
| count=$(jq ' | |
| [ | |
| .[] | | |
| select( | |
| . == "sweep-enabled" or | |
| . == "full-sweep-enabled" or | |
| . == "non-canary-full-sweep-enabled" or | |
| . == "full-sweep-fail-fast" or | |
| . == "full-sweep-fail-fast-no-canary" | |
| ) | |
| ] | | |
| length | |
| ' <<<"$SWEEP_LABELS") | |
| if [ "$count" -gt 1 ]; then | |
| echo "::error::PR has multiple conflicting sweep labels. Pick exactly one." | |
| exit 1 | |
| fi | |
| - name: Checkout code | |
| uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| fetch-depth: 0 | |
| persist-credentials: false | |
| - name: Read PR sweep policy | |
| id: sweep_policy | |
| env: | |
| HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| run: | | |
| if git log -1 --format=%B "$HEAD_SHA" | grep -Fq '[skip-sweep]'; then | |
| echo "skip-pr-sweep=true" >> "$GITHUB_OUTPUT" | |
| else | |
| echo "skip-pr-sweep=false" >> "$GITHUB_OUTPUT" | |
| fi | |
| - name: Set up uv | |
| uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 | |
| - name: Validate perf-changelog matrix | |
| env: | |
| PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| ALL_EVALS: ${{ contains(github.event.pull_request.labels.*.name, 'all-evals') }} | |
| EVALS_ONLY: ${{ contains(github.event.pull_request.labels.*.name, 'evals-only') }} | |
| run: | | |
| CMD=( | |
| uv run --no-project --exclude-newer PT12H --python 3.12 | |
| --with "pydantic>=2" --with pyyaml python -m infx.workflows.validate_perf_changelog | |
| --changelog-file perf-changelog.yaml | |
| --base-ref "origin/${GITHUB_BASE_REF}" | |
| --head-ref "${PR_HEAD_SHA}" | |
| ) | |
| if [ "$ALL_EVALS" = "true" ]; then | |
| CMD+=(--all-evals) | |
| fi | |
| if [ "$EVALS_ONLY" = "true" ]; then | |
| CMD+=(--evals-only) | |
| fi | |
| "${CMD[@]}" | |
| reuse-sweep-gate: | |
| name: reuse-sweep-gate | |
| needs: check-changelog | |
| runs-on: ubuntu-latest | |
| permissions: | |
| actions: read # Read workflow runs and artifacts. | |
| contents: read | |
| issues: read # Read issue comments. | |
| pull-requests: read # Read PR metadata. | |
| if: >- | |
| always() && | |
| needs.check-changelog.result == 'success' && | |
| github.event_name == 'pull_request' && | |
| github.event.action == 'synchronize' && | |
| !contains(github.event.pull_request.labels.*.name, 'evals-only') && | |
| !contains(github.event.pull_request.labels.*.name, 'agentx-fast') | |
| outputs: | |
| skip-pr-sweep: ${{ steps.gate.outputs.skip-pr-sweep }} | |
| steps: | |
| - name: Checkout code | |
| uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| persist-credentials: false | |
| - name: Check for reusable sweep authorization | |
| id: gate | |
| env: | |
| PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| GH_TOKEN: ${{ github.token }} | |
| EVENT_ACTION: ${{ github.event.action }} | |
| run: | | |
| python3 -m infx.workflows.reuse \ | |
| --repo "${GITHUB_REPOSITORY}" \ | |
| --commit-sha "${PR_HEAD_SHA}" \ | |
| --event-name "${GITHUB_EVENT_NAME}" \ | |
| --event-action "${EVENT_ACTION}" \ | |
| --pr-number "${PR_NUMBER}" \ | |
| --ref "${GITHUB_REF}" \ | |
| --workflow-id "run-sweep.yml" | |
| setup: | |
| name: setup | |
| needs: [check-changelog, reuse-sweep-gate] | |
| runs-on: ubuntu-latest | |
| permissions: | |
| actions: read # Read workflow runs and artifacts. | |
| contents: read | |
| issues: read # Read issue comments. | |
| pull-requests: read # Read PR metadata. | |
| if: >- | |
| always() && | |
| ( | |
| needs.check-changelog.result == 'success' || | |
| needs.check-changelog.result == 'skipped' | |
| ) && | |
| ( | |
| needs.reuse-sweep-gate.result == 'skipped' || | |
| ( | |
| needs.reuse-sweep-gate.result == 'success' && | |
| needs.reuse-sweep-gate.outputs.skip-pr-sweep != 'true' | |
| ) | |
| ) && | |
| ( | |
| ( | |
| github.event_name == 'pull_request' && | |
| needs.check-changelog.result == 'success' && | |
| github.event.pull_request.head.repo.full_name == github.repository && | |
| needs.check-changelog.outputs.skip-pr-sweep != 'true' && | |
| ( | |
| contains(github.event.pull_request.labels.*.name, 'sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'non-canary-full-sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') | |
| ) && | |
| ( | |
| (github.event.action != 'labeled' && github.event.action != 'unlabeled') || | |
| github.event.label.name == 'sweep-enabled' || | |
| github.event.label.name == 'full-sweep-enabled' || | |
| github.event.label.name == 'non-canary-full-sweep-enabled' || | |
| github.event.label.name == 'full-sweep-fail-fast' || | |
| github.event.label.name == 'full-sweep-fail-fast-no-canary' || | |
| github.event.label.name == 'all-evals' || | |
| github.event.label.name == 'evals-only' || | |
| github.event.label.name == 'agentx-fast' || | |
| github.event.label.name == 'skip_queue' || | |
| github.event.label.name == 'ci-patchwork' || | |
| github.event.label.name == 'engine-patch' || | |
| github.event.label.name == 'ci-patchwork-waived' || | |
| github.event.label.name == 'ci-checklist-complete' | |
| ) | |
| ) || | |
| ( | |
| github.event_name == 'push' | |
| ) | |
| ) | |
| outputs: | |
| tooling-ref: ${{ steps.tooling.outputs.result }} | |
| search-space-config: ${{ steps.setup.outputs.search-space-config }} | |
| reuse-enabled: ${{ steps.setup.outputs.reuse-enabled }} | |
| reuse-source-run-id: ${{ steps.setup.outputs.reuse-source-run-id }} | |
| reuse-source-run-attempt: ${{ steps.setup.outputs.reuse-source-run-attempt }} | |
| reuse-source-run-url: ${{ steps.setup.outputs.reuse-source-run-url }} | |
| reuse-source-pr-number: ${{ steps.setup.outputs.reuse-source-pr-number }} | |
| reuse-source-head-sha: ${{ steps.setup.outputs.reuse-source-head-sha }} | |
| steps: | |
| - name: Checkout code | |
| id: checkout | |
| uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| fetch-depth: 0 | |
| persist-credentials: false | |
| - name: Resolve workflow tooling | |
| id: tooling | |
| uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 | |
| env: | |
| CHECKOUT_SHA: ${{ steps.checkout.outputs.commit }} | |
| with: | |
| result-encoding: string | |
| script: | | |
| const { data } = await github.rest.repos.getCommit({ | |
| ...context.repo, ref: context.payload.repository.default_branch, | |
| }); | |
| await core.summary.addHeading('Revisions', 2).addTable([ | |
| ['Setup checkout', process.env.CHECKOUT_SHA], | |
| ['infx tooling', data.sha], | |
| ]).write(); | |
| return data.sha; | |
| - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 | |
| - name: Install Claude Code 2.1.282 | |
| id: claude_cli | |
| if: vars.PRIORITY_SCHEDULER_ENABLED == 'true' && github.event_name == 'pull_request' | |
| run: | # zizmor: ignore[adhoc-packages] Claude CLI and its native packages are pinned to 2.1.282 | |
| npm install --prefix "$RUNNER_TEMP/claude-code" --no-audit --no-fund @anthropic-ai/claude-code@2.1.282 | |
| claude_cli="$RUNNER_TEMP/claude-code/node_modules/.bin/claude" | |
| test "$("$claude_cli" --version)" = "2.1.282 (Claude Code)" | |
| echo "path=$claude_cli" >> "$GITHUB_OUTPUT" | |
| - name: Classify priority criteria | |
| id: classify | |
| if: >- | |
| vars.PRIORITY_SCHEDULER_ENABLED == 'true' && | |
| github.event_name == 'pull_request' | |
| continue-on-error: true | |
| uses: anthropics/claude-code-action@9171db3e57d6a3140a37ddc2ba92788584e0ead6 # v1.0.234 | |
| with: | |
| path_to_claude_code_executable: ${{ steps.claude_cli.outputs.path }} | |
| github_token: ${{ github.token }} | |
| # Repository integration credential for the scoped sweep/ingest job. | |
| anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env] | |
| track_progress: false | |
| settings: | | |
| {"fastMode": true} | |
| claude_args: | | |
| --model 'claude-opus-5-5' | |
| --effort low | |
| --max-turns 8 | |
| --allowedTools "Read,Glob,Grep,Bash(git diff:*)" | |
| --json-schema '{"type":"object","properties":{"criteria":{"type":"array","items":{"type":"string","enum":["multi-node","agentic","eval-only","fp4","mtp","eagle","eagle3","sglang","vllm","dynamo-sglang","dynamo-vllm","glm5","glm5.1","kimik3","dsv4","minimaxm3","qwen3.5","dsr1","checklist-complete","patchwork"]},"uniqueItems":true},"reason":{"type":"string"}},"required":["criteria","reason"]}' | |
| prompt: | | |
| Inspect this pull request's diff against its base branch. | |
| Return every matching priority criterion: | |
| - multi-node: separate prefill and decode workers | |
| - agentic: agentic-coding or AgentX workload | |
| - eval-only: evaluation without throughput measurement | |
| - fp4: FP4 precision | |
| - mtp, eagle, eagle3: speculative decoding method | |
| - sglang, vllm, dynamo-vllm: runtime framework | |
| - model criterion: matching configured model family; legacy criteria remain in the shared priority taxonomy for retained SPEED-Bench collectors and do not authorize deprecated master coverage | |
| - checklist-complete: PR checklist is satisfied | |
| - patchwork: modified upstream engine or runtime source | |
| Patchwork includes patch files, git apply, source mutation, | |
| monkey-patching, local engine builds, forks, or images | |
| containing unmerged engine changes. | |
| Return a concise reason grounded in the diff. | |
| - name: Normalize priority classification | |
| id: priority-criteria | |
| if: vars.PRIORITY_SCHEDULER_ENABLED == 'true' && always() | |
| env: | |
| OUTPUT: ${{ steps.classify.outputs.structured_output || '{}' }} | |
| run: | | |
| if jq -e 'type == "object" and (.criteria | type == "array")' \ | |
| >/dev/null <<<"$OUTPUT"; then | |
| criteria=$(jq -c '.criteria' <<<"$OUTPUT") | |
| else | |
| criteria='["patchwork"]' | |
| fi | |
| echo "criteria=$criteria" >> "$GITHUB_OUTPUT" | |
| echo "Priority criteria: $criteria" | |
| - id: setup | |
| env: | |
| PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }} | |
| PUSH_BEFORE: ${{ github.event.before }} | |
| PUSH_AFTER: ${{ github.event.after }} | |
| PR_NUMBER: ${{ github.event.pull_request.number || 0 }} | |
| GH_TOKEN: ${{ github.token }} | |
| PR_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }} | |
| PRIORITY_CRITERIA: ${{ steps.priority-criteria.outputs.criteria || '' }} | |
| TRIM_CONC: >- | |
| ${{ | |
| github.event_name == 'pull_request' && | |
| contains(github.event.pull_request.labels.*.name, 'sweep-enabled') | |
| }} | |
| ALL_EVALS: >- | |
| ${{ | |
| github.event_name == 'pull_request' && | |
| contains(github.event.pull_request.labels.*.name, 'all-evals') | |
| }} | |
| EVALS_ONLY: >- | |
| ${{ | |
| github.event_name == 'pull_request' && | |
| contains(github.event.pull_request.labels.*.name, 'evals-only') | |
| }} | |
| run: | | |
| if [ "${GITHUB_EVENT_NAME}" == "pull_request" ]; then | |
| BASE_REF="origin/${GITHUB_BASE_REF}" | |
| HEAD_REF="${PR_HEAD_SHA}" | |
| else | |
| BASE_REF="${PUSH_BEFORE}" | |
| HEAD_REF="${PUSH_AFTER}" | |
| fi | |
| CMD=( | |
| uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ | |
| python -m infx.matrix.plan | |
| --changelog-file "${GITHUB_WORKSPACE}/perf-changelog.yaml" | |
| --base-ref "$BASE_REF" | |
| --head-ref "$HEAD_REF" | |
| ) | |
| if [ "$TRIM_CONC" = "true" ]; then | |
| CMD+=(--trim-conc) | |
| fi | |
| if [ "$ALL_EVALS" = "true" ]; then | |
| CMD+=(--all-evals) | |
| fi | |
| if [ "$EVALS_ONLY" = "true" ]; then | |
| CMD+=(--evals-only) | |
| fi | |
| CONFIG_JSON=$("${CMD[@]}") | |
| CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \ | |
| python -m infx.workflows.benchmark_schema --plan) | |
| CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \ | |
| python -m infx.workflows.ci_priority \ | |
| --event-name "${GITHUB_EVENT_NAME}" \ | |
| --queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}" \ | |
| --labels-json "$PR_LABELS" \ | |
| --pr-number "${PR_NUMBER}" \ | |
| --criteria-json "$PRIORITY_CRITERIA") | |
| echo "search-space-config=$CONFIG_JSON" >> "$GITHUB_OUTPUT" | |
| python3 -m infx.workflows.reuse \ | |
| --repo "${GITHUB_REPOSITORY}" \ | |
| --commit-sha "${GITHUB_SHA}" \ | |
| --event-name "${GITHUB_EVENT_NAME}" \ | |
| --ref "${GITHUB_REF}" \ | |
| --workflow-id "run-sweep.yml" | |
| - name: Summarize reused source | |
| if: steps.setup.outputs.reuse-source-head-sha != '' | |
| uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 | |
| env: | |
| REUSE_SHA: ${{ steps.setup.outputs.reuse-source-head-sha }} | |
| with: | |
| script: | | |
| await core.summary.addTable([ | |
| ['Reused benchmark source', process.env.REUSE_SHA], | |
| ]).write(); | |
| canary-select: | |
| name: canary-select | |
| needs: setup | |
| if: >- | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| github.event_name == 'pull_request' && | |
| ( | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') | |
| ) | |
| runs-on: ubuntu-latest | |
| outputs: | |
| canary-config: ${{ steps.pick.outputs.canary-config }} | |
| multi-node-canary-config: ${{ steps.pick.outputs.multi-node-canary-config }} | |
| remaining-search-space-config: ${{ steps.pick.outputs.remaining-search-space-config }} | |
| steps: | |
| - id: pick | |
| env: | |
| SEARCH_SPACE: ${{ needs.setup.outputs.search-space-config }} | |
| run: | | |
| selection=$(jq -c ' | |
| def remove_one($needle): | |
| if $needle == null then . | |
| else | |
| (index($needle)) as $idx | |
| | if $idx == null then . else del(.[$idx]) end | |
| end; | |
| # Canary is a benchmark-only smoke test — exclude entries | |
| # whose primary purpose is eval (run-eval == true) so the | |
| # picked canary never runs an eval pass. | |
| (((.single_node["1k1k"] // []) + (.single_node["8k1k"] // []) + (.single_node.agentic // [])) | |
| | map(select(.["run-eval"] != true))) as $single_candidates | |
| | ((.multi_node.agentic // []) | map(select(.["run-eval"] != true))) as $multi_candidates | |
| | (if ($single_candidates | length) == 0 then null else ($single_candidates | min_by(.conc)) end) as $canary | |
| | (if $canary != null or ($multi_candidates | length) == 0 then null | |
| else ($multi_candidates | min_by(.conc[0])) end) as $multi_canary | |
| | { | |
| canary: (if $canary == null then [] else [$canary] end), | |
| multi_node_canary: (if $multi_canary == null then [] else [$multi_canary] end), | |
| remaining: ( | |
| . | |
| | .single_node = (.single_node // {}) | |
| | .single_node["1k1k"] = ((.single_node["1k1k"] // []) | remove_one($canary)) | |
| | .single_node["8k1k"] = ((.single_node["8k1k"] // []) | remove_one($canary)) | |
| | .single_node.agentic = ((.single_node.agentic // []) | remove_one($canary)) | |
| | .multi_node = (.multi_node // {}) | |
| | .multi_node.agentic = ((.multi_node.agentic // []) | remove_one($multi_canary)) | |
| ) | |
| } | |
| ' <<<"$SEARCH_SPACE") | |
| { | |
| echo "canary-config=$(jq -c '.canary' <<<"$selection")" | |
| echo "multi-node-canary-config=$(jq -c '.multi_node_canary' <<<"$selection")" | |
| echo "remaining-search-space-config=$(jq -c '.remaining' <<<"$selection")" | |
| } >> "$GITHUB_OUTPUT" | |
| canary-sweep: | |
| needs: canary-select | |
| if: ${{ needs.canary-select.outputs.canary-config != '' && needs.canary-select.outputs.canary-config != '[]' }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: canary / | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| config: ${{ fromJson(needs.canary-select.outputs.canary-config) }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| # Only same-repository PRs owned by Klaud receive background priority. | |
| klaud-run: &klaud-run ${{ github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && github.event.pull_request.user.login == 'Klaud-Cold' && (startsWith(github.head_ref, 'klaud/auto-') || startsWith(github.head_ref, 'klaude/auto-')) }} | |
| require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} | |
| runner: ${{ matrix.config.runner }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| dp-attn: ${{ matrix.config.dp-attn }} | |
| run-eval: false | |
| scenario-type: ${{ matrix.config['scenario-type'] || 'fixed-seq-len' }} | |
| duration: ${{ matrix.config.duration || '3600' }} | |
| agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }} | |
| canary-multi-node-sweep: | |
| needs: canary-select | |
| if: ${{ needs.canary-select.outputs.multi-node-canary-config != '' && needs.canary-select.outputs.multi-node-canary-config != '[]' }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node canary / | |
| strategy: | |
| fail-fast: false | |
| matrix: | |
| config: ${{ fromJson(needs.canary-select.outputs.multi-node-canary-config) }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| node-count: ${{ matrix.config.node-count }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| conc-list: ${{ toJson(matrix.config.conc) }} | |
| conc: ${{ matrix.config.conc[0] }} | |
| total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }} | |
| duration: ${{ matrix.config.duration }} | |
| run-eval: false | |
| scenario-type: agentic-coding | |
| agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }} | |
| sweep-multi-node-1k1k: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k']) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node 1k1k / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: &multi-node-inputs | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| node-count: ${{ matrix.config.node-count }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} | |
| conc-list: ${{ toJson(matrix.config.conc) }} | |
| run-eval: false | |
| sweep-multi-node-8k1k: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k']) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node 8k1k / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: *multi-node-inputs | |
| sweep-single-node-1k1k: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != 'null' && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != '[]' | |
| }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: single-node 1k1k / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: &single-node-inputs | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }} | |
| runner: ${{ matrix.config.runner }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| dp-attn: ${{ matrix.config.dp-attn }} | |
| run-eval: ${{ matrix.config.run-eval }} | |
| sweep-single-node-8k1k: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != 'null' && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != '[]' | |
| }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: single-node 8k1k / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: *single-node-inputs | |
| sweep-agentic: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic']) != 'null' && | |
| toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic']) != '[]' | |
| }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: agentic / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| dp-attn: ${{ matrix.config.dp-attn }} | |
| duration: ${{ matrix.config.duration }} | |
| run-eval: false | |
| scenario-type: agentic-coding | |
| agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }} | |
| sweep-multi-node-agentic: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic']) != 'null' && | |
| toJson(fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic']) != '[]' | |
| }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node agentic / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic'] }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| node-count: ${{ matrix.config.node-count }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| conc-list: ${{ toJson(matrix.config.conc) }} | |
| conc: ${{ matrix.config.conc[0] }} | |
| total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }} | |
| duration: ${{ matrix.config.duration }} | |
| run-eval: false | |
| scenario-type: agentic-coding | |
| agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }} | |
| sweep-evals: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).evals) != '[]' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).evals) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: eval / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).evals }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| dp-attn: ${{ matrix.config.dp-attn }} | |
| eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }} | |
| eval-suite: ${{ matrix.config['eval-suite'] || '' }} | |
| run-eval: true | |
| eval-only: true | |
| # Agentic GSM8K eval rows carry the agentic input shape, so they are | |
| # dispatched with sweep-agentic's inputs rather than sweep-evals' fixed-seq-len | |
| # inputs (isl/osl/max-model-len, which agentic rows don't have). | |
| sweep-agentic-evals: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != '[]' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-tmpl.yml | |
| name: agentic eval / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).agentic_evals }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| dp-attn: ${{ matrix.config.dp-attn }} | |
| duration: ${{ matrix.config.duration }} | |
| eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }} | |
| eval-suite: ${{ matrix.config['eval-suite'] || '' }} | |
| run-eval: true | |
| eval-only: true | |
| scenario-type: agentic-coding | |
| sweep-multi-node-evals: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != '[]' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node eval / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_evals }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| node-count: ${{ matrix.config.node-count }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| conc-list: ${{ toJson(matrix.config.conc) }} | |
| eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }} | |
| eval-suite: ${{ matrix.config['eval-suite'] || '' }} | |
| run-eval: true | |
| eval-only: true | |
| eval-conc: ${{ (matrix.config['eval-framework'] || 'lm-eval') == 'lm-eval' && matrix.config['eval-all-concs'] && join(matrix.config.conc, ' ') || matrix.config['eval-conc'] }} | |
| # Multi-node agentic GSM8K eval rows carry the agentic input shape, so | |
| # they are dispatched with sweep-multi-node-agentic's inputs rather than | |
| # sweep-multi-node-evals' fixed-seq-len inputs (isl/osl/max-model-len, | |
| # which agentic rows don't have). Agentic selection uses one highest | |
| # eval-conc per topology; fixed-sequence --all-evals rows may instead pass | |
| # a joined concurrency list to lm-eval. | |
| sweep-multi-node-agentic-evals: | |
| needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep] | |
| if: >- | |
| ${{ | |
| !cancelled() && | |
| needs.setup.result == 'success' && | |
| needs.setup.outputs.reuse-enabled != 'true' && | |
| (needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') && | |
| (needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != '[]' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != 'null' | |
| }} | |
| uses: $/.github/workflows/benchmark-multinode-tmpl.yml | |
| name: multi-node agentic eval / | |
| strategy: | |
| fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }} | |
| matrix: | |
| config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals }} | |
| secrets: | |
| INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }} | |
| MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }} | |
| MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }} | |
| with: | |
| config: ${{ toJSON(matrix.config) }} | |
| klaud-run: *klaud-run | |
| runner: ${{ matrix.config.runner }} | |
| node-count: ${{ matrix.config.node-count }} | |
| priority: ${{ matrix.config.priority }} | |
| queue-token: ${{ matrix.config['queue-token'] }} | |
| skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }} | |
| conc-list: ${{ format('[{0}]', matrix.config['eval-conc']) }} | |
| conc: ${{ matrix.config['eval-conc'] }} | |
| total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }} | |
| duration: ${{ matrix.config.duration }} | |
| eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }} | |
| eval-suite: ${{ matrix.config['eval-suite'] || '' }} | |
| run-eval: true | |
| eval-only: true | |
| eval-conc: ${{ matrix.config['eval-conc'] }} | |
| scenario-type: agentic-coding | |
| collect-results: | |
| name: collect-results | |
| needs: | |
| [ | |
| canary-sweep, | |
| canary-multi-node-sweep, | |
| sweep-single-node-1k1k, | |
| sweep-single-node-8k1k, | |
| sweep-agentic, | |
| sweep-multi-node-1k1k, | |
| sweep-multi-node-8k1k, | |
| sweep-multi-node-agentic, | |
| setup, | |
| ] | |
| if: >- | |
| ${{ | |
| always() && | |
| needs.setup.result == 'success' && | |
| ( | |
| needs.canary-sweep.result == 'success' || | |
| needs.canary-multi-node-sweep.result == 'success' || | |
| needs.sweep-single-node-1k1k.result != 'skipped' || | |
| needs.sweep-single-node-8k1k.result != 'skipped' || | |
| needs.sweep-multi-node-1k1k.result != 'skipped' || | |
| needs.sweep-multi-node-8k1k.result != 'skipped' | |
| ) | |
| }} | |
| uses: $/.github/workflows/collect-results.yml | |
| with: | |
| result-prefix: "bmk" | |
| tooling-ref: ${{ needs.setup.outputs.tooling-ref }} | |
| collect-evals: | |
| name: collect-evals | |
| needs: [sweep-evals, sweep-agentic-evals, sweep-multi-node-evals, sweep-multi-node-agentic-evals, setup] | |
| if: ${{ always() && needs.setup.result != 'skipped' && (needs.sweep-evals.result != 'skipped' || needs.sweep-agentic-evals.result != 'skipped' || needs.sweep-multi-node-evals.result != 'skipped' || needs.sweep-multi-node-agentic-evals.result != 'skipped') }} | |
| uses: $/.github/workflows/collect-evals.yml | |
| with: | |
| tooling-ref: ${{ needs.setup.outputs.tooling-ref }} | |
| upload-changelog-metadata: | |
| name: upload-changelog-metadata | |
| needs: [setup, collect-results] | |
| if: ${{ always() && needs.setup.result == 'success' }} | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Extract and save changelog metadata | |
| env: | |
| SWEEP_MATRIX: ${{ needs.setup.outputs.search-space-config }} | |
| SWEEP_HEAD: ${{ github.event.pull_request.head.sha || github.sha }} | |
| SWEEP_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }} | |
| FULL_SWEEP: ${{ github.event_name == 'pull_request' && needs.setup.outputs.reuse-enabled != 'true' }} | |
| run: | | |
| python3 - <<'PY' | |
| import json, os | |
| from pathlib import Path | |
| matrix = json.loads(os.environ['SWEEP_MATRIX']) | |
| labels = set(json.loads(os.environ['SWEEP_LABELS']) or []) | |
| sweep_labels = labels & { | |
| 'sweep-enabled', 'full-sweep-enabled', 'non-canary-full-sweep-enabled', | |
| 'full-sweep-fail-fast', 'full-sweep-fail-fast-no-canary', | |
| 'all-evals', 'evals-only', 'agentx-fast', | |
| } | |
| Path('changelog_metadata.json').write_text(json.dumps(matrix['changelog_metadata'])) | |
| Path('sweep_manifest.json').write_text(json.dumps({ | |
| 'head': os.environ['SWEEP_HEAD'], 'run-id': int(os.environ['GITHUB_RUN_ID']), | |
| 'run-attempt': int(os.environ['GITHUB_RUN_ATTEMPT']), | |
| 'full-sweep': os.environ['FULL_SWEEP'] == 'true' and sweep_labels in ( | |
| {'full-sweep-fail-fast'}, {'full-sweep-enabled'}, | |
| ), | |
| 'matrix': matrix, | |
| })) | |
| PY | |
| - name: Upload changelog artifact | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: changelog-metadata | |
| path: changelog_metadata.json | |
| - name: Upload Klaud validation manifest | |
| if: *klaud-run | |
| uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: klaud-sweep-manifest | |
| path: sweep_manifest.json | |
| if-no-files-found: error | |
| calc-success-rate: | |
| name: calc-success-rate | |
| needs: [setup, collect-results] | |
| if: ${{ always() && needs.collect-results.result != 'skipped'}} | |
| runs-on: ubuntu-latest | |
| permissions: | |
| actions: read # Read job timings for run statistics. | |
| contents: read | |
| env: | |
| RESULTS_DIR: "results/" | |
| STATS_FILENAME: "run_stats" | |
| GITHUB_TOKEN: ${{ github.token }} | |
| steps: | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| ref: ${{ needs.setup.outputs.tooling-ref }} | |
| fetch-depth: 0 | |
| persist-credentials: false | |
| - name: Download results artifacts | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 | |
| with: | |
| path: ${{ env.RESULTS_DIR }} | |
| pattern: results_* | |
| - name: Set up uv | |
| uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 | |
| - name: Calculate success rate | |
| run: >- | |
| uv run --locked --extra workflows | |
| python -m infx.workflows.calc_success_rate "$STATS_FILENAME" | |
| - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 | |
| with: | |
| name: "run-stats" | |
| path: ${{ env.STATS_FILENAME }}.json | |
| compare-results: | |
| name: compare-results | |
| needs: | |
| [ | |
| collect-results, | |
| setup, | |
| ] | |
| if: >- | |
| always() && | |
| github.event_name == 'pull_request' && | |
| needs.collect-results.result == 'success' | |
| runs-on: ubuntu-latest | |
| env: | |
| # Repository integration credential for the scoped sweep/ingest job. | |
| DATABASE_URL: ${{ secrets.DB_RO_URL }} # zizmor: ignore[secrets-outside-env] | |
| steps: | |
| - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| ref: ${{ needs.setup.outputs.tooling-ref }} | |
| persist-credentials: false | |
| - name: Download results artifacts | |
| uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 | |
| with: | |
| path: results/ | |
| pattern: results_bmk | |
| - name: Set up uv | |
| uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 | |
| - name: Compare results against main | |
| run: >- | |
| uv run --locked --extra results | |
| python -m infx.results.compare_results results/ >> "$GITHUB_STEP_SUMMARY" | |
| trigger-ingest: | |
| name: trigger-ingest | |
| needs: | |
| [ | |
| collect-results, | |
| collect-evals, | |
| calc-success-rate, | |
| setup, | |
| upload-changelog-metadata, | |
| ] | |
| if: >- | |
| always() && | |
| github.event_name == 'push' && | |
| github.ref == 'refs/heads/main' && | |
| needs.setup.result == 'success' && | |
| ( | |
| ( | |
| toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) == 'null' || | |
| toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) == '[]' | |
| ) && | |
| ( | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) == 'null' || | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) == '[]' | |
| ) | |
| ) && | |
| ( | |
| needs.collect-results.result != 'skipped' || | |
| needs.collect-evals.result != 'skipped' || | |
| needs.setup.outputs.reuse-enabled == 'true' | |
| ) | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Trigger database ingest | |
| uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 | |
| env: | |
| SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} | |
| MERGE_RUN_ID: ${{ github.run_id }} | |
| with: | |
| # Repository integration credential for the scoped sweep/ingest job. | |
| github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] | |
| script: | | |
| await github.rest.repos.createDispatchEvent({ | |
| owner: "SemiAnalysisAI", | |
| repo: "InferenceX-app", | |
| event_type: "ingest-results", | |
| client_payload: { | |
| "source-run-id": process.env.SOURCE_RUN_ID, | |
| "merge-run-id": process.env.MERGE_RUN_ID | |
| } | |
| }); | |
| trigger-agentic-ingest: | |
| name: trigger-agentic-ingest | |
| needs: | |
| [ | |
| setup, | |
| sweep-agentic, | |
| sweep-multi-node-agentic, | |
| upload-changelog-metadata, | |
| ] | |
| if: >- | |
| always() && | |
| github.event_name == 'push' && | |
| github.ref == 'refs/heads/main' && | |
| needs.setup.result == 'success' && | |
| needs.upload-changelog-metadata.result == 'success' && | |
| ( | |
| ( | |
| needs.setup.outputs.reuse-enabled == 'true' && | |
| ( | |
| ( | |
| toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) != 'null' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) != '[]' | |
| ) || | |
| ( | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) != 'null' && | |
| toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) != '[]' | |
| ) | |
| ) | |
| ) || | |
| ( | |
| ( | |
| needs.sweep-agentic.result == 'success' || | |
| needs.sweep-multi-node-agentic.result == 'success' | |
| ) && | |
| ( | |
| needs.sweep-agentic.result == 'success' || | |
| needs.sweep-agentic.result == 'skipped' | |
| ) && | |
| ( | |
| needs.sweep-multi-node-agentic.result == 'success' || | |
| needs.sweep-multi-node-agentic.result == 'skipped' | |
| ) | |
| ) | |
| ) | |
| runs-on: ubuntu-latest | |
| steps: | |
| - name: Trigger agentic database ingest | |
| uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 | |
| env: | |
| SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }} | |
| MERGE_RUN_ID: ${{ github.run_id }} | |
| with: | |
| # Repository integration credential for the scoped sweep/ingest job. | |
| github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env] | |
| script: | | |
| await github.rest.repos.createDispatchEvent({ | |
| owner: "SemiAnalysisAI", | |
| repo: "InferenceX-app", | |
| event_type: "ingest-agentic-results", | |
| client_payload: { | |
| "source-run-id": process.env.SOURCE_RUN_ID, | |
| "merge-run-id": process.env.MERGE_RUN_ID, | |
| "database-target": "production" | |
| } | |
| }); | |
| comment-unofficial-run-visualizer: | |
| name: comment-unofficial-run-visualizer | |
| needs: | |
| [ | |
| collect-results, | |
| collect-evals, | |
| calc-success-rate, | |
| upload-changelog-metadata, | |
| setup, | |
| ] | |
| if: >- | |
| always() && | |
| needs.setup.result == 'success' && | |
| github.event_name == 'pull_request' && | |
| ( | |
| contains(github.event.pull_request.labels.*.name, 'sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'non-canary-full-sweep-enabled') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || | |
| contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') | |
| ) && | |
| ( | |
| (github.event.action != 'labeled' && github.event.action != 'unlabeled') || | |
| github.event.label.name == 'sweep-enabled' || | |
| github.event.label.name == 'full-sweep-enabled' || | |
| github.event.label.name == 'non-canary-full-sweep-enabled' || | |
| github.event.label.name == 'full-sweep-fail-fast' || | |
| github.event.label.name == 'full-sweep-fail-fast-no-canary' || | |
| github.event.label.name == 'all-evals' || | |
| github.event.label.name == 'evals-only' || | |
| github.event.label.name == 'agentx-fast' | |
| ) | |
| runs-on: ubuntu-latest | |
| concurrency: | |
| group: unofficial-run-visualizer-${{ github.event.pull_request.number }} | |
| cancel-in-progress: false | |
| permissions: | |
| pull-requests: write # Publish PR feedback. | |
| steps: | |
| - name: Update unofficial run visualizer links on PR | |
| uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 | |
| with: | |
| github-token: ${{ github.token }} | |
| script: | | |
| const marker = '<!-- inferencex-unofficial-run-visualizer -->'; | |
| const inferenceUrl = `https://inferencex.semianalysis.com/inference?unofficialRun=${context.runId}`; | |
| const evaluationUrl = `https://inferencex.semianalysis.com/evaluation?unofficialRun=${context.runId}`; | |
| const body = [ | |
| marker, | |
| `View unofficial run (performance): ${inferenceUrl}`, | |
| `View unofficial run (accuracy): ${evaluationUrl}`, | |
| ].join('\n\n'); | |
| const comments = await github.paginate(github.rest.issues.listComments, { | |
| ...context.repo, | |
| issue_number: context.issue.number, | |
| per_page: 100, | |
| }); | |
| const botComments = comments.filter(comment => comment.user?.login === 'github-actions[bot]'); | |
| const existing = botComments.find(comment => comment.body?.includes(marker)) | |
| ?? botComments.reverse().find(comment => comment.body?.startsWith( | |
| 'see unofficial run visualizer at https://inferencex.semianalysis.com/inference?unofficialRun=' | |
| )); | |
| if (existing) { | |
| const previousRunId = existing.body.match(/unofficialRun=(\d+)/)?.[1]; | |
| if (previousRunId && Number(previousRunId) > context.runId) { | |
| core.info('Keeping visualizer links for the newer run.'); | |
| return; | |
| } | |
| await github.rest.issues.updateComment({ | |
| ...context.repo, | |
| comment_id: existing.id, | |
| body, | |
| }); | |
| } else { | |
| await github.rest.issues.createComment({ | |
| ...context.repo, | |
| issue_number: context.issue.number, | |
| body, | |
| }); | |
| } |