Skip to content

Run Sweep - Add B300 TP1 DSV4.1 Flash 8k1k serving sweep / 新增 B300 TP1 DSV4.1 Flash 8k1k serving 扫描 #14711

Run Sweep - Add B300 TP1 DSV4.1 Flash 8k1k serving sweep / 新增 B300 TP1 DSV4.1 Flash 8k1k serving 扫描

Run Sweep - Add B300 TP1 DSV4.1 Flash 8k1k serving sweep / 新增 B300 TP1 DSV4.1 Flash 8k1k serving 扫描 #14711

Workflow file for this run

name: "Run Sweep"
run-name: Run Sweep - ${{ github.event.pull_request.title || github.event.head_commit.message }}
concurrency:
group: >-
sweep-${{ github.event.pull_request.number || github.sha }}-${{
github.event_name == 'pull_request' &&
(github.event.action == 'labeled' || github.event.action == 'unlabeled') &&
github.event.label.name != 'sweep-enabled' &&
github.event.label.name != 'full-sweep-enabled' &&
github.event.label.name != 'non-canary-full-sweep-enabled' &&
github.event.label.name != 'full-sweep-fail-fast' &&
github.event.label.name != 'full-sweep-fail-fast-no-canary' &&
github.event.label.name != 'all-evals' &&
github.event.label.name != 'evals-only' &&
github.event.label.name != 'agentx-fast' &&
github.event.label.name != 'skip_queue' &&
github.event.label.name != 'ci-patchwork' &&
github.event.label.name != 'engine-patch' &&
github.event.label.name != 'ci-patchwork-waived' &&
github.event.label.name != 'ci-checklist-complete' &&
github.run_id ||
'active'
}}
cancel-in-progress: true
on:
push:
branches:
- main
paths:
- "perf-changelog.yaml"
pull_request:
branches:
- main
types:
- synchronize
- labeled
- unlabeled
paths:
- "perf-changelog.yaml"
permissions:
contents: read
jobs:
check-changelog:
name: check-changelog
runs-on: ubuntu-latest
permissions:
contents: read
# Labels opt into sweeps regardless of draft status; readiness only starts reviews.
if: >-
github.event_name == 'pull_request' &&
github.event.pull_request.head.repo.full_name == github.repository &&
(
(github.event.action != 'labeled' && github.event.action != 'unlabeled') ||
github.event.label.name == 'sweep-enabled' ||
github.event.label.name == 'full-sweep-enabled' ||
github.event.label.name == 'non-canary-full-sweep-enabled' ||
github.event.label.name == 'full-sweep-fail-fast' ||
github.event.label.name == 'full-sweep-fail-fast-no-canary' ||
github.event.label.name == 'all-evals' ||
github.event.label.name == 'evals-only' ||
github.event.label.name == 'agentx-fast' ||
github.event.label.name == 'skip_queue' ||
github.event.label.name == 'ci-patchwork' ||
github.event.label.name == 'engine-patch' ||
github.event.label.name == 'ci-patchwork-waived' ||
github.event.label.name == 'ci-checklist-complete'
)
outputs:
skip-pr-sweep: ${{ steps.sweep_policy.outputs.skip-pr-sweep }}
steps:
- name: Reject conflicting sweep labels
env:
SWEEP_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }}
run: |
count=$(jq '
[
.[] |
select(
. == "sweep-enabled" or
. == "full-sweep-enabled" or
. == "non-canary-full-sweep-enabled" or
. == "full-sweep-fail-fast" or
. == "full-sweep-fail-fast-no-canary"
)
] |
length
' <<<"$SWEEP_LABELS")
if [ "$count" -gt 1 ]; then
echo "::error::PR has multiple conflicting sweep labels. Pick exactly one."
exit 1
fi
- name: Checkout code
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
persist-credentials: false
- name: Read PR sweep policy
id: sweep_policy
env:
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
run: |
if git log -1 --format=%B "$HEAD_SHA" | grep -Fq '[skip-sweep]'; then
echo "skip-pr-sweep=true" >> "$GITHUB_OUTPUT"
else
echo "skip-pr-sweep=false" >> "$GITHUB_OUTPUT"
fi
- name: Set up uv
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
- name: Validate perf-changelog matrix
env:
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
ALL_EVALS: ${{ contains(github.event.pull_request.labels.*.name, 'all-evals') }}
EVALS_ONLY: ${{ contains(github.event.pull_request.labels.*.name, 'evals-only') }}
run: |
CMD=(
uv run --no-project --exclude-newer PT12H --python 3.12
--with "pydantic>=2" --with pyyaml python -m infx.workflows.validate_perf_changelog
--changelog-file perf-changelog.yaml
--base-ref "origin/${GITHUB_BASE_REF}"
--head-ref "${PR_HEAD_SHA}"
)
if [ "$ALL_EVALS" = "true" ]; then
CMD+=(--all-evals)
fi
if [ "$EVALS_ONLY" = "true" ]; then
CMD+=(--evals-only)
fi
"${CMD[@]}"
reuse-sweep-gate:
name: reuse-sweep-gate
needs: check-changelog
runs-on: ubuntu-latest
permissions:
actions: read # Read workflow runs and artifacts.
contents: read
issues: read # Read issue comments.
pull-requests: read # Read PR metadata.
if: >-
always() &&
needs.check-changelog.result == 'success' &&
github.event_name == 'pull_request' &&
github.event.action == 'synchronize' &&
!contains(github.event.pull_request.labels.*.name, 'evals-only') &&
!contains(github.event.pull_request.labels.*.name, 'agentx-fast')
outputs:
skip-pr-sweep: ${{ steps.gate.outputs.skip-pr-sweep }}
steps:
- name: Checkout code
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
persist-credentials: false
- name: Check for reusable sweep authorization
id: gate
env:
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
PR_NUMBER: ${{ github.event.pull_request.number }}
GH_TOKEN: ${{ github.token }}
EVENT_ACTION: ${{ github.event.action }}
run: |
python3 -m infx.workflows.reuse \
--repo "${GITHUB_REPOSITORY}" \
--commit-sha "${PR_HEAD_SHA}" \
--event-name "${GITHUB_EVENT_NAME}" \
--event-action "${EVENT_ACTION}" \
--pr-number "${PR_NUMBER}" \
--ref "${GITHUB_REF}" \
--workflow-id "run-sweep.yml"
setup:
name: setup
needs: [check-changelog, reuse-sweep-gate]
runs-on: ubuntu-latest
permissions:
actions: read # Read workflow runs and artifacts.
contents: read
issues: read # Read issue comments.
pull-requests: read # Read PR metadata.
if: >-
always() &&
(
needs.check-changelog.result == 'success' ||
needs.check-changelog.result == 'skipped'
) &&
(
needs.reuse-sweep-gate.result == 'skipped' ||
(
needs.reuse-sweep-gate.result == 'success' &&
needs.reuse-sweep-gate.outputs.skip-pr-sweep != 'true'
)
) &&
(
(
github.event_name == 'pull_request' &&
needs.check-changelog.result == 'success' &&
github.event.pull_request.head.repo.full_name == github.repository &&
needs.check-changelog.outputs.skip-pr-sweep != 'true' &&
(
contains(github.event.pull_request.labels.*.name, 'sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'non-canary-full-sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary')
) &&
(
(github.event.action != 'labeled' && github.event.action != 'unlabeled') ||
github.event.label.name == 'sweep-enabled' ||
github.event.label.name == 'full-sweep-enabled' ||
github.event.label.name == 'non-canary-full-sweep-enabled' ||
github.event.label.name == 'full-sweep-fail-fast' ||
github.event.label.name == 'full-sweep-fail-fast-no-canary' ||
github.event.label.name == 'all-evals' ||
github.event.label.name == 'evals-only' ||
github.event.label.name == 'agentx-fast' ||
github.event.label.name == 'skip_queue' ||
github.event.label.name == 'ci-patchwork' ||
github.event.label.name == 'engine-patch' ||
github.event.label.name == 'ci-patchwork-waived' ||
github.event.label.name == 'ci-checklist-complete'
)
) ||
(
github.event_name == 'push'
)
)
outputs:
tooling-ref: ${{ steps.tooling.outputs.result }}
search-space-config: ${{ steps.setup.outputs.search-space-config }}
reuse-enabled: ${{ steps.setup.outputs.reuse-enabled }}
reuse-source-run-id: ${{ steps.setup.outputs.reuse-source-run-id }}
reuse-source-run-attempt: ${{ steps.setup.outputs.reuse-source-run-attempt }}
reuse-source-run-url: ${{ steps.setup.outputs.reuse-source-run-url }}
reuse-source-pr-number: ${{ steps.setup.outputs.reuse-source-pr-number }}
reuse-source-head-sha: ${{ steps.setup.outputs.reuse-source-head-sha }}
steps:
- name: Checkout code
id: checkout
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
persist-credentials: false
- name: Resolve workflow tooling
id: tooling
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
CHECKOUT_SHA: ${{ steps.checkout.outputs.commit }}
with:
result-encoding: string
script: |
const { data } = await github.rest.repos.getCommit({
...context.repo, ref: context.payload.repository.default_branch,
});
await core.summary.addHeading('Revisions', 2).addTable([
['Setup checkout', process.env.CHECKOUT_SHA],
['infx tooling', data.sha],
]).write();
return data.sha;
- uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
- name: Install Claude Code 2.1.282
id: claude_cli
if: vars.PRIORITY_SCHEDULER_ENABLED == 'true' && github.event_name == 'pull_request'
run: | # zizmor: ignore[adhoc-packages] Claude CLI and its native packages are pinned to 2.1.282
npm install --prefix "$RUNNER_TEMP/claude-code" --no-audit --no-fund @anthropic-ai/claude-code@2.1.282
claude_cli="$RUNNER_TEMP/claude-code/node_modules/.bin/claude"
test "$("$claude_cli" --version)" = "2.1.282 (Claude Code)"
echo "path=$claude_cli" >> "$GITHUB_OUTPUT"
- name: Classify priority criteria
id: classify
if: >-
vars.PRIORITY_SCHEDULER_ENABLED == 'true' &&
github.event_name == 'pull_request'
continue-on-error: true
uses: anthropics/claude-code-action@9171db3e57d6a3140a37ddc2ba92788584e0ead6 # v1.0.234
with:
path_to_claude_code_executable: ${{ steps.claude_cli.outputs.path }}
github_token: ${{ github.token }}
# Repository integration credential for the scoped sweep/ingest job.
anthropic_api_key: ${{ secrets.ANTHROPIC_API_KEY }} # zizmor: ignore[secrets-outside-env]
track_progress: false
settings: |
{"fastMode": true}
claude_args: |
--model 'claude-opus-5-5'
--effort low
--max-turns 8
--allowedTools "Read,Glob,Grep,Bash(git diff:*)"
--json-schema '{"type":"object","properties":{"criteria":{"type":"array","items":{"type":"string","enum":["multi-node","agentic","eval-only","fp4","mtp","eagle","eagle3","sglang","vllm","dynamo-sglang","dynamo-vllm","glm5","glm5.1","kimik3","dsv4","minimaxm3","qwen3.5","dsr1","checklist-complete","patchwork"]},"uniqueItems":true},"reason":{"type":"string"}},"required":["criteria","reason"]}'
prompt: |
Inspect this pull request's diff against its base branch.
Return every matching priority criterion:
- multi-node: separate prefill and decode workers
- agentic: agentic-coding or AgentX workload
- eval-only: evaluation without throughput measurement
- fp4: FP4 precision
- mtp, eagle, eagle3: speculative decoding method
- sglang, vllm, dynamo-vllm: runtime framework
- model criterion: matching configured model family; legacy criteria remain in the shared priority taxonomy for retained SPEED-Bench collectors and do not authorize deprecated master coverage
- checklist-complete: PR checklist is satisfied
- patchwork: modified upstream engine or runtime source
Patchwork includes patch files, git apply, source mutation,
monkey-patching, local engine builds, forks, or images
containing unmerged engine changes.
Return a concise reason grounded in the diff.
- name: Normalize priority classification
id: priority-criteria
if: vars.PRIORITY_SCHEDULER_ENABLED == 'true' && always()
env:
OUTPUT: ${{ steps.classify.outputs.structured_output || '{}' }}
run: |
if jq -e 'type == "object" and (.criteria | type == "array")' \
>/dev/null <<<"$OUTPUT"; then
criteria=$(jq -c '.criteria' <<<"$OUTPUT")
else
criteria='["patchwork"]'
fi
echo "criteria=$criteria" >> "$GITHUB_OUTPUT"
echo "Priority criteria: $criteria"
- id: setup
env:
PR_HEAD_SHA: ${{ github.event.pull_request.head.sha }}
PUSH_BEFORE: ${{ github.event.before }}
PUSH_AFTER: ${{ github.event.after }}
PR_NUMBER: ${{ github.event.pull_request.number || 0 }}
GH_TOKEN: ${{ github.token }}
PR_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }}
PRIORITY_CRITERIA: ${{ steps.priority-criteria.outputs.criteria || '' }}
TRIM_CONC: >-
${{
github.event_name == 'pull_request' &&
contains(github.event.pull_request.labels.*.name, 'sweep-enabled')
}}
ALL_EVALS: >-
${{
github.event_name == 'pull_request' &&
contains(github.event.pull_request.labels.*.name, 'all-evals')
}}
EVALS_ONLY: >-
${{
github.event_name == 'pull_request' &&
contains(github.event.pull_request.labels.*.name, 'evals-only')
}}
run: |
if [ "${GITHUB_EVENT_NAME}" == "pull_request" ]; then
BASE_REF="origin/${GITHUB_BASE_REF}"
HEAD_REF="${PR_HEAD_SHA}"
else
BASE_REF="${PUSH_BEFORE}"
HEAD_REF="${PUSH_AFTER}"
fi
CMD=(
uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \
python -m infx.matrix.plan
--changelog-file "${GITHUB_WORKSPACE}/perf-changelog.yaml"
--base-ref "$BASE_REF"
--head-ref "$HEAD_REF"
)
if [ "$TRIM_CONC" = "true" ]; then
CMD+=(--trim-conc)
fi
if [ "$ALL_EVALS" = "true" ]; then
CMD+=(--all-evals)
fi
if [ "$EVALS_ONLY" = "true" ]; then
CMD+=(--evals-only)
fi
CONFIG_JSON=$("${CMD[@]}")
CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pydantic --with pyyaml \
python -m infx.workflows.benchmark_schema --plan)
CONFIG_JSON=$(printf '%s' "$CONFIG_JSON" | uv run --no-project --exclude-newer PT12H --python 3.12 --with pyyaml \
python -m infx.workflows.ci_priority \
--event-name "${GITHUB_EVENT_NAME}" \
--queue-namespace "${GITHUB_RUN_ID}:${GITHUB_RUN_ATTEMPT}" \
--labels-json "$PR_LABELS" \
--pr-number "${PR_NUMBER}" \
--criteria-json "$PRIORITY_CRITERIA")
echo "search-space-config=$CONFIG_JSON" >> "$GITHUB_OUTPUT"
python3 -m infx.workflows.reuse \
--repo "${GITHUB_REPOSITORY}" \
--commit-sha "${GITHUB_SHA}" \
--event-name "${GITHUB_EVENT_NAME}" \
--ref "${GITHUB_REF}" \
--workflow-id "run-sweep.yml"
- name: Summarize reused source
if: steps.setup.outputs.reuse-source-head-sha != ''
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
REUSE_SHA: ${{ steps.setup.outputs.reuse-source-head-sha }}
with:
script: |
await core.summary.addTable([
['Reused benchmark source', process.env.REUSE_SHA],
]).write();
canary-select:
name: canary-select
needs: setup
if: >-
needs.setup.outputs.reuse-enabled != 'true' &&
github.event_name == 'pull_request' &&
(
contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast')
)
runs-on: ubuntu-latest
outputs:
canary-config: ${{ steps.pick.outputs.canary-config }}
multi-node-canary-config: ${{ steps.pick.outputs.multi-node-canary-config }}
remaining-search-space-config: ${{ steps.pick.outputs.remaining-search-space-config }}
steps:
- id: pick
env:
SEARCH_SPACE: ${{ needs.setup.outputs.search-space-config }}
run: |
selection=$(jq -c '
def remove_one($needle):
if $needle == null then .
else
(index($needle)) as $idx
| if $idx == null then . else del(.[$idx]) end
end;
# Canary is a benchmark-only smoke test — exclude entries
# whose primary purpose is eval (run-eval == true) so the
# picked canary never runs an eval pass.
(((.single_node["1k1k"] // []) + (.single_node["8k1k"] // []) + (.single_node.agentic // []))
| map(select(.["run-eval"] != true))) as $single_candidates
| ((.multi_node.agentic // []) | map(select(.["run-eval"] != true))) as $multi_candidates
| (if ($single_candidates | length) == 0 then null else ($single_candidates | min_by(.conc)) end) as $canary
| (if $canary != null or ($multi_candidates | length) == 0 then null
else ($multi_candidates | min_by(.conc[0])) end) as $multi_canary
| {
canary: (if $canary == null then [] else [$canary] end),
multi_node_canary: (if $multi_canary == null then [] else [$multi_canary] end),
remaining: (
.
| .single_node = (.single_node // {})
| .single_node["1k1k"] = ((.single_node["1k1k"] // []) | remove_one($canary))
| .single_node["8k1k"] = ((.single_node["8k1k"] // []) | remove_one($canary))
| .single_node.agentic = ((.single_node.agentic // []) | remove_one($canary))
| .multi_node = (.multi_node // {})
| .multi_node.agentic = ((.multi_node.agentic // []) | remove_one($multi_canary))
)
}
' <<<"$SEARCH_SPACE")
{
echo "canary-config=$(jq -c '.canary' <<<"$selection")"
echo "multi-node-canary-config=$(jq -c '.multi_node_canary' <<<"$selection")"
echo "remaining-search-space-config=$(jq -c '.remaining' <<<"$selection")"
} >> "$GITHUB_OUTPUT"
canary-sweep:
needs: canary-select
if: ${{ needs.canary-select.outputs.canary-config != '' && needs.canary-select.outputs.canary-config != '[]' }}
uses: $/.github/workflows/benchmark-tmpl.yml
name: canary /
strategy:
fail-fast: false
matrix:
config: ${{ fromJson(needs.canary-select.outputs.canary-config) }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
# Only same-repository PRs owned by Klaud receive background priority.
klaud-run: &klaud-run ${{ github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository && github.event.pull_request.user.login == 'Klaud-Cold' && (startsWith(github.head_ref, 'klaud/auto-') || startsWith(github.head_ref, 'klaude/auto-')) }}
require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
dp-attn: ${{ matrix.config.dp-attn }}
run-eval: false
scenario-type: ${{ matrix.config['scenario-type'] || 'fixed-seq-len' }}
duration: ${{ matrix.config.duration || '3600' }}
agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }}
canary-multi-node-sweep:
needs: canary-select
if: ${{ needs.canary-select.outputs.multi-node-canary-config != '' && needs.canary-select.outputs.multi-node-canary-config != '[]' }}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node canary /
strategy:
fail-fast: false
matrix:
config: ${{ fromJson(needs.canary-select.outputs.multi-node-canary-config) }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
conc-list: ${{ toJson(matrix.config.conc) }}
conc: ${{ matrix.config.conc[0] }}
total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }}
duration: ${{ matrix.config.duration }}
run-eval: false
scenario-type: agentic-coding
agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }}
sweep-multi-node-1k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k']) != 'null'
}}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node 1k1k /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['1k1k'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with: &multi-node-inputs
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }}
conc-list: ${{ toJson(matrix.config.conc) }}
run-eval: false
sweep-multi-node-8k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k']) != 'null'
}}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node 8k1k /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).multi_node['8k1k'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with: *multi-node-inputs
sweep-single-node-1k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != 'null' &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k']) != '[]'
}}
uses: $/.github/workflows/benchmark-tmpl.yml
name: single-node 1k1k /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['1k1k'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with: &single-node-inputs
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
require-power: ${{ matrix.config['require-power'] == true && matrix.config['eval-only'] != true }}
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
dp-attn: ${{ matrix.config.dp-attn }}
run-eval: ${{ matrix.config.run-eval }}
sweep-single-node-8k1k:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != 'null' &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k']) != '[]'
}}
uses: $/.github/workflows/benchmark-tmpl.yml
name: single-node 8k1k /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['8k1k'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with: *single-node-inputs
sweep-agentic:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic']) != 'null' &&
toJson(fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic']) != '[]'
}}
uses: $/.github/workflows/benchmark-tmpl.yml
name: agentic /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson((needs.canary-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).single_node['agentic'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
dp-attn: ${{ matrix.config.dp-attn }}
duration: ${{ matrix.config.duration }}
run-eval: false
scenario-type: agentic-coding
agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }}
sweep-multi-node-agentic:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic']) != 'null' &&
toJson(fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic']) != '[]'
}}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node agentic /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson((needs.canary-multi-node-sweep.result == 'success' && needs.canary-select.outputs.remaining-search-space-config) || needs.setup.outputs.search-space-config).multi_node['agentic'] }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
conc-list: ${{ toJson(matrix.config.conc) }}
conc: ${{ matrix.config.conc[0] }}
total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }}
duration: ${{ matrix.config.duration }}
run-eval: false
scenario-type: agentic-coding
agentx-fast: ${{ contains(github.event.pull_request.labels.*.name, 'agentx-fast') }}
sweep-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).evals) != '[]' &&
toJson(fromJson(needs.setup.outputs.search-space-config).evals) != 'null'
}}
uses: $/.github/workflows/benchmark-tmpl.yml
name: eval /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).evals }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
dp-attn: ${{ matrix.config.dp-attn }}
eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }}
eval-suite: ${{ matrix.config['eval-suite'] || '' }}
run-eval: true
eval-only: true
# Agentic GSM8K eval rows carry the agentic input shape, so they are
# dispatched with sweep-agentic's inputs rather than sweep-evals' fixed-seq-len
# inputs (isl/osl/max-model-len, which agentic rows don't have).
sweep-agentic-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != '[]' &&
toJson(fromJson(needs.setup.outputs.search-space-config).agentic_evals) != 'null'
}}
uses: $/.github/workflows/benchmark-tmpl.yml
name: agentic eval /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).agentic_evals }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
dp-attn: ${{ matrix.config.dp-attn }}
duration: ${{ matrix.config.duration }}
eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }}
eval-suite: ${{ matrix.config['eval-suite'] || '' }}
run-eval: true
eval-only: true
scenario-type: agentic-coding
sweep-multi-node-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != '[]' &&
toJson(fromJson(needs.setup.outputs.search-space-config).multinode_evals) != 'null'
}}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node eval /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_evals }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
conc-list: ${{ toJson(matrix.config.conc) }}
eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }}
eval-suite: ${{ matrix.config['eval-suite'] || '' }}
run-eval: true
eval-only: true
eval-conc: ${{ (matrix.config['eval-framework'] || 'lm-eval') == 'lm-eval' && matrix.config['eval-all-concs'] && join(matrix.config.conc, ' ') || matrix.config['eval-conc'] }}
# Multi-node agentic GSM8K eval rows carry the agentic input shape, so
# they are dispatched with sweep-multi-node-agentic's inputs rather than
# sweep-multi-node-evals' fixed-seq-len inputs (isl/osl/max-model-len,
# which agentic rows don't have). Agentic selection uses one highest
# eval-conc per topology; fixed-sequence --all-evals rows may instead pass
# a joined concurrency list to lm-eval.
sweep-multi-node-agentic-evals:
needs: [setup, canary-select, canary-sweep, canary-multi-node-sweep]
if: >-
${{
!cancelled() &&
needs.setup.result == 'success' &&
needs.setup.outputs.reuse-enabled != 'true' &&
(needs.canary-sweep.result == 'success' || needs.canary-sweep.result == 'skipped') &&
(needs.canary-multi-node-sweep.result == 'success' || needs.canary-multi-node-sweep.result == 'skipped') &&
toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != '[]' &&
toJson(fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals) != 'null'
}}
uses: $/.github/workflows/benchmark-multinode-tmpl.yml
name: multi-node agentic eval /
strategy:
fail-fast: ${{ contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') || contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary') }}
matrix:
config: ${{ fromJson(needs.setup.outputs.search-space-config).multinode_agentic_evals }}
secrets:
INFERENCEX_OFFICIAL_RO_HF_TOKEN: ${{ secrets.INFERENCEX_OFFICIAL_RO_HF_TOKEN }}
MODAL_TOKEN_ID: ${{ secrets.MODAL_TOKEN_ID }}
MODAL_TOKEN_SECRET: ${{ secrets.MODAL_TOKEN_SECRET }}
with:
config: ${{ toJSON(matrix.config) }}
klaud-run: *klaud-run
runner: ${{ matrix.config.runner }}
node-count: ${{ matrix.config.node-count }}
priority: ${{ matrix.config.priority }}
queue-token: ${{ matrix.config['queue-token'] }}
skip-queue-pr: ${{ matrix.config['skip-queue-pr'] || '' }}
conc-list: ${{ format('[{0}]', matrix.config['eval-conc']) }}
conc: ${{ matrix.config['eval-conc'] }}
total-cpu-dram-gb: ${{ matrix.config['total-cpu-dram-gb'] }}
duration: ${{ matrix.config.duration }}
eval-framework: ${{ matrix.config['eval-framework'] || 'lm-eval' }}
eval-suite: ${{ matrix.config['eval-suite'] || '' }}
run-eval: true
eval-only: true
eval-conc: ${{ matrix.config['eval-conc'] }}
scenario-type: agentic-coding
collect-results:
name: collect-results
needs:
[
canary-sweep,
canary-multi-node-sweep,
sweep-single-node-1k1k,
sweep-single-node-8k1k,
sweep-agentic,
sweep-multi-node-1k1k,
sweep-multi-node-8k1k,
sweep-multi-node-agentic,
setup,
]
if: >-
${{
always() &&
needs.setup.result == 'success' &&
(
needs.canary-sweep.result == 'success' ||
needs.canary-multi-node-sweep.result == 'success' ||
needs.sweep-single-node-1k1k.result != 'skipped' ||
needs.sweep-single-node-8k1k.result != 'skipped' ||
needs.sweep-multi-node-1k1k.result != 'skipped' ||
needs.sweep-multi-node-8k1k.result != 'skipped'
)
}}
uses: $/.github/workflows/collect-results.yml
with:
result-prefix: "bmk"
tooling-ref: ${{ needs.setup.outputs.tooling-ref }}
collect-evals:
name: collect-evals
needs: [sweep-evals, sweep-agentic-evals, sweep-multi-node-evals, sweep-multi-node-agentic-evals, setup]
if: ${{ always() && needs.setup.result != 'skipped' && (needs.sweep-evals.result != 'skipped' || needs.sweep-agentic-evals.result != 'skipped' || needs.sweep-multi-node-evals.result != 'skipped' || needs.sweep-multi-node-agentic-evals.result != 'skipped') }}
uses: $/.github/workflows/collect-evals.yml
with:
tooling-ref: ${{ needs.setup.outputs.tooling-ref }}
upload-changelog-metadata:
name: upload-changelog-metadata
needs: [setup, collect-results]
if: ${{ always() && needs.setup.result == 'success' }}
runs-on: ubuntu-latest
steps:
- name: Extract and save changelog metadata
env:
SWEEP_MATRIX: ${{ needs.setup.outputs.search-space-config }}
SWEEP_HEAD: ${{ github.event.pull_request.head.sha || github.sha }}
SWEEP_LABELS: ${{ toJson(github.event.pull_request.labels.*.name) }}
FULL_SWEEP: ${{ github.event_name == 'pull_request' && needs.setup.outputs.reuse-enabled != 'true' }}
run: |
python3 - <<'PY'
import json, os
from pathlib import Path
matrix = json.loads(os.environ['SWEEP_MATRIX'])
labels = set(json.loads(os.environ['SWEEP_LABELS']) or [])
sweep_labels = labels & {
'sweep-enabled', 'full-sweep-enabled', 'non-canary-full-sweep-enabled',
'full-sweep-fail-fast', 'full-sweep-fail-fast-no-canary',
'all-evals', 'evals-only', 'agentx-fast',
}
Path('changelog_metadata.json').write_text(json.dumps(matrix['changelog_metadata']))
Path('sweep_manifest.json').write_text(json.dumps({
'head': os.environ['SWEEP_HEAD'], 'run-id': int(os.environ['GITHUB_RUN_ID']),
'run-attempt': int(os.environ['GITHUB_RUN_ATTEMPT']),
'full-sweep': os.environ['FULL_SWEEP'] == 'true' and sweep_labels in (
{'full-sweep-fail-fast'}, {'full-sweep-enabled'},
),
'matrix': matrix,
}))
PY
- name: Upload changelog artifact
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: changelog-metadata
path: changelog_metadata.json
- name: Upload Klaud validation manifest
if: *klaud-run
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: klaud-sweep-manifest
path: sweep_manifest.json
if-no-files-found: error
calc-success-rate:
name: calc-success-rate
needs: [setup, collect-results]
if: ${{ always() && needs.collect-results.result != 'skipped'}}
runs-on: ubuntu-latest
permissions:
actions: read # Read job timings for run statistics.
contents: read
env:
RESULTS_DIR: "results/"
STATS_FILENAME: "run_stats"
GITHUB_TOKEN: ${{ github.token }}
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ needs.setup.outputs.tooling-ref }}
fetch-depth: 0
persist-credentials: false
- name: Download results artifacts
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
path: ${{ env.RESULTS_DIR }}
pattern: results_*
- name: Set up uv
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
- name: Calculate success rate
run: >-
uv run --locked --extra workflows
python -m infx.workflows.calc_success_rate "$STATS_FILENAME"
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: "run-stats"
path: ${{ env.STATS_FILENAME }}.json
compare-results:
name: compare-results
needs:
[
collect-results,
setup,
]
if: >-
always() &&
github.event_name == 'pull_request' &&
needs.collect-results.result == 'success'
runs-on: ubuntu-latest
env:
# Repository integration credential for the scoped sweep/ingest job.
DATABASE_URL: ${{ secrets.DB_RO_URL }} # zizmor: ignore[secrets-outside-env]
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
ref: ${{ needs.setup.outputs.tooling-ref }}
persist-credentials: false
- name: Download results artifacts
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
path: results/
pattern: results_bmk
- name: Set up uv
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
- name: Compare results against main
run: >-
uv run --locked --extra results
python -m infx.results.compare_results results/ >> "$GITHUB_STEP_SUMMARY"
trigger-ingest:
name: trigger-ingest
needs:
[
collect-results,
collect-evals,
calc-success-rate,
setup,
upload-changelog-metadata,
]
if: >-
always() &&
github.event_name == 'push' &&
github.ref == 'refs/heads/main' &&
needs.setup.result == 'success' &&
(
(
toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) == 'null' ||
toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) == '[]'
) &&
(
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) == 'null' ||
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) == '[]'
)
) &&
(
needs.collect-results.result != 'skipped' ||
needs.collect-evals.result != 'skipped' ||
needs.setup.outputs.reuse-enabled == 'true'
)
runs-on: ubuntu-latest
steps:
- name: Trigger database ingest
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }}
MERGE_RUN_ID: ${{ github.run_id }}
with:
# Repository integration credential for the scoped sweep/ingest job.
github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env]
script: |
await github.rest.repos.createDispatchEvent({
owner: "SemiAnalysisAI",
repo: "InferenceX-app",
event_type: "ingest-results",
client_payload: {
"source-run-id": process.env.SOURCE_RUN_ID,
"merge-run-id": process.env.MERGE_RUN_ID
}
});
trigger-agentic-ingest:
name: trigger-agentic-ingest
needs:
[
setup,
sweep-agentic,
sweep-multi-node-agentic,
upload-changelog-metadata,
]
if: >-
always() &&
github.event_name == 'push' &&
github.ref == 'refs/heads/main' &&
needs.setup.result == 'success' &&
needs.upload-changelog-metadata.result == 'success' &&
(
(
needs.setup.outputs.reuse-enabled == 'true' &&
(
(
toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) != 'null' &&
toJson(fromJson(needs.setup.outputs.search-space-config).single_node['agentic']) != '[]'
) ||
(
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) != 'null' &&
toJson(fromJson(needs.setup.outputs.search-space-config).multi_node['agentic']) != '[]'
)
)
) ||
(
(
needs.sweep-agentic.result == 'success' ||
needs.sweep-multi-node-agentic.result == 'success'
) &&
(
needs.sweep-agentic.result == 'success' ||
needs.sweep-agentic.result == 'skipped'
) &&
(
needs.sweep-multi-node-agentic.result == 'success' ||
needs.sweep-multi-node-agentic.result == 'skipped'
)
)
)
runs-on: ubuntu-latest
steps:
- name: Trigger agentic database ingest
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
SOURCE_RUN_ID: ${{ needs.setup.outputs.reuse-enabled == 'true' && needs.setup.outputs.reuse-source-run-id || github.run_id }}
MERGE_RUN_ID: ${{ github.run_id }}
with:
# Repository integration credential for the scoped sweep/ingest job.
github-token: ${{ secrets.FRONTEND_PAT }} # zizmor: ignore[secrets-outside-env]
script: |
await github.rest.repos.createDispatchEvent({
owner: "SemiAnalysisAI",
repo: "InferenceX-app",
event_type: "ingest-agentic-results",
client_payload: {
"source-run-id": process.env.SOURCE_RUN_ID,
"merge-run-id": process.env.MERGE_RUN_ID,
"database-target": "production"
}
});
comment-unofficial-run-visualizer:
name: comment-unofficial-run-visualizer
needs:
[
collect-results,
collect-evals,
calc-success-rate,
upload-changelog-metadata,
setup,
]
if: >-
always() &&
needs.setup.result == 'success' &&
github.event_name == 'pull_request' &&
(
contains(github.event.pull_request.labels.*.name, 'sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'non-canary-full-sweep-enabled') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast') ||
contains(github.event.pull_request.labels.*.name, 'full-sweep-fail-fast-no-canary')
) &&
(
(github.event.action != 'labeled' && github.event.action != 'unlabeled') ||
github.event.label.name == 'sweep-enabled' ||
github.event.label.name == 'full-sweep-enabled' ||
github.event.label.name == 'non-canary-full-sweep-enabled' ||
github.event.label.name == 'full-sweep-fail-fast' ||
github.event.label.name == 'full-sweep-fail-fast-no-canary' ||
github.event.label.name == 'all-evals' ||
github.event.label.name == 'evals-only' ||
github.event.label.name == 'agentx-fast'
)
runs-on: ubuntu-latest
concurrency:
group: unofficial-run-visualizer-${{ github.event.pull_request.number }}
cancel-in-progress: false
permissions:
pull-requests: write # Publish PR feedback.
steps:
- name: Update unofficial run visualizer links on PR
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
with:
github-token: ${{ github.token }}
script: |
const marker = '<!-- inferencex-unofficial-run-visualizer -->';
const inferenceUrl = `https://inferencex.semianalysis.com/inference?unofficialRun=${context.runId}`;
const evaluationUrl = `https://inferencex.semianalysis.com/evaluation?unofficialRun=${context.runId}`;
const body = [
marker,
`View unofficial run (performance): ${inferenceUrl}`,
`View unofficial run (accuracy): ${evaluationUrl}`,
].join('\n\n');
const comments = await github.paginate(github.rest.issues.listComments, {
...context.repo,
issue_number: context.issue.number,
per_page: 100,
});
const botComments = comments.filter(comment => comment.user?.login === 'github-actions[bot]');
const existing = botComments.find(comment => comment.body?.includes(marker))
?? botComments.reverse().find(comment => comment.body?.startsWith(
'see unofficial run visualizer at https://inferencex.semianalysis.com/inference?unofficialRun='
));
if (existing) {
const previousRunId = existing.body.match(/unofficialRun=(\d+)/)?.[1];
if (previousRunId && Number(previousRunId) > context.runId) {
core.info('Keeping visualizer links for the newer run.');
return;
}
await github.rest.issues.updateComment({
...context.repo,
comment_id: existing.id,
body,
});
} else {
await github.rest.issues.createComment({
...context.repo,
issue_number: context.issue.number,
body,
});
}