Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 vs H200 FP8 on GLM-5: Up to 3.65x Better Performance per Dollar with SGLang MTP'
subtitle: 'Both SKUs run SGLang EAGLE MTP; the Blackwell generation lifts perf/$ by ~1.2x at the peak and the NVIDIA GLM-5-NVFP4 checkpoint on FlashInfer TRT-LLM sparse MLA stacks another ~2.4–3.0x on 8K/1K'
seoDescription: 'B200 NVFP4 delivers up to 3.65x better perf/$ than H200 FP8 on GLM-5 with SGLang MTP, stacking Blackwell gains and a sparse-MLA checkpoint.'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 vs H100 FP8 on MiniMax-M2.5: Up to 8.2x Better Performance per Dollar with vLLM'
subtitle: 'vLLM PR #36307 unlocks the trtllm-gen FP8 MoE kernel for MiniMax on B200; combined with NVFP4, perf/$ scales from 4.0x at 22 tok/s/user to 8.2x at 110 on 8K/1K'
seoDescription: 'B200 NVFP4 delivers up to 8.2x better perf/$ than H100 FP8 on MiniMax-M2.5 with vLLM — trtllm-gen FP8 MoE plus NVFP4 on 8K/1K.'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 vs H200 INT4 on Kimi K2.5/K2.6: Up to 2.95x Better Performance per Dollar'
subtitle: "On vLLM 8K/1K the NVFP4 path on B200 is 2.71x–2.95x cheaper per million tokens than H200 INT4 across the entire 30–90 tok/s/user serving band, and 2.45x–2.74x cheaper than B200 INT4 on the same silicon. Both factors decompose cleanly into B200's HBM bandwidth, HBM capacity, and NVFP4 tensor cores"
seoDescription: 'B200 NVFP4 runs Kimi K2.5/K2.6 up to 2.95x cheaper per token than H200 INT4 on vLLM 8K/1K, across the full 30–90 tok/s/user serving band.'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'GB200 NVL72 vs B200 on DeepSeek R1 670B: Up to 4.4x Throughput per GPU at 125 tok/s/user'
subtitle: "DeepSeek R1 FP4 1k/1k. NVL72's 72-GPU NVLink scale-up fabric lets decode run wide EP up to EP=32, where B200's 8-GPU NVLink island caps out at EP=8 over RoCEv2"
seoDescription: 'GB200 NVL72 hits up to 4.4x more per-GPU throughput than B200 on DeepSeek R1 670B at 125 tok/s/user — 72-GPU NVLink enables wide EP=32.'
date: '2026-05-23'
publishDate: '2026-05-23'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'GB300 NVL72 vs GB200 NVL72 Inference Performance & Perf per Dollar - on DeepSeek-V4-Pro 1.6T: Up to 2.83x Throughput'
subtitle: "DSv4-Pro FP4 8K/1K, Dynamo+vLLM, disaggregated on both racks. GB300's 50% extra HBM (288 vs 192 GB/GPU) unlocks a wider prefill+decode recipe GB200 can't fit — lifting middle-of-curve perf/$ by 2.31x despite a 20% per-GPU TCO premium."
seoDescription: 'GB300 NVL72 adds 50% more HBM, lifting DeepSeek-V4-Pro perf/$ up to 2.31x over GB200 NVL72 on vLLM FP4 8K/1K — up to 2.83x throughput per GPU.'
date: '2026-05-27'
publishDate: '2026-05-27'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'InferenceMAX: Open Source Inference Benchmarking'
subtitle: 'NVIDIA GB200 NVL72, AMD MI355X, Throughput Token per GPU, Latency Tok/s/user, Perf per Dollar, Cost per Million Tokens, Tokens per Provisioned Megawatt, DeepSeek R1 670B, GPTOSS 120B, Llama3 70B'
seoDescription: 'InferenceMAX: open-source inference benchmarks across NVIDIA GB200 NVL72 and AMD MI355X — throughput, latency, cost per token and perf per watt.'
date: '2025-10-09'
publishDate: '2025-10-09'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'MI355X DeepSeek-V4-Pro on SGLang: 110.5x Throughput per GPU in 26 Days'
subtitle: 'The amd/deepseek_v4 side branch shipped TileLang attention indexer, Triton sparse MLA, fused RoPE/Hadamard, FlyDSL MoE, and FP4 weights across 31 performance optimizations PRs — lifting first-light 20 tok/s/GPU at 2.4 tok/s/user into 2,256 tok/s/GPU at 9.4 tok/s/user on 8K/1K, with both throughput and interactivity climbing together'
seoDescription: 'MI355X ran DeepSeek-V4-Pro 110.5x faster per GPU in 26 days on SGLang — from 20 to 2,256 tok/s/GPU via TileLang, sparse MLA and FP4 on 8K/1K.'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'AMD MI355X GLM-5 Inference: Up to 40% Cheaper per Million Tokens than B200 on SGLang FP8'
subtitle: '14 weeks after GLM-5 launched, AMD landed both MTP and non-MTP SGLang FP8 recipes on MI355X — fused MLA + FP8 KV cache via TileLang flips the single-node FP8 cost curve in AMD favor across most of the performance Pareto'
seoDescription: 'AMD MI355X runs GLM-5 up to 40% cheaper per million tokens than B200 on SGLang FP8, with fused MLA and FP8 KV cache via TileLang.'
date: '2026-05-25'
publishDate: '2026-05-25'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'AMD MI355X Qwen3.5 397B-A17B Inference: Up to 19x Throughput per GPU in 3 Months on SGLang FP8'
subtitle: 'From v0.5.8 (Feb) → v0.5.10rc0 (Apr) → v0.5.12 (May), three AITER kernel landings on MI355X plus a TP=8 → TP=2/TP=4 retune push Qwen3.5 8k/1k peak from 1.3k to 6.4k tok/s/GPU and extend the curve out to 75 tok/s/user'
seoDescription: 'AMD MI355X lifted Qwen3.5 397B-A17B up to 19x per-GPU throughput in 3 months on SGLang FP8 — 1.3k to 6.4k tok/s/GPU via AITER kernels.'
date: '2026-05-25'
publishDate: '2026-05-25'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 对比 H200 FP8 运行 GLM-5:SGLang MTP 下性价比提升高达 3.65 倍'
subtitle: '两款 GPU 均运行 SGLang EAGLE MTP;Blackwell 世代在峰值处带来约 1.2 倍的性价比提升,NVIDIA GLM-5-NVFP4 检查点搭配 FlashInfer TRT-LLM 稀疏 MLA 在 8K/1K 场景下再叠加约 2.4–3.0 倍优势'
seoDescription: 'B200 NVFP4 在 SGLang MTP 下运行 GLM-5,性价比比 H200 FP8 最高提升 3.65 倍——叠加 Blackwell 世代增益与稀疏 MLA 检查点。'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 vs H100 FP8 运行 MiniMax-M2.5:vLLM 下每美元性能最高提升 8.2 倍'
subtitle: 'vLLM PR #36307 为 MiniMax 在 B200 上解锁了 trtllm-gen FP8 MoE 模块化内核;结合 NVFP4,在 8K/1K 负载下性能/成本从 22 tok/s/user 时的 4.0 倍扩大到 110 tok/s/user 时的 8.2 倍'
seoDescription: 'B200 NVFP4 在 vLLM 下运行 MiniMax-M2.5,每美元性能比 H100 FP8 最高提升 8.2 倍——trtllm-gen FP8 MoE 结合 NVFP4(8K/1K)。'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'B200 NVFP4 对比 H200 INT4 运行 Kimi K2.5/K2.6:性价比提升高达 2.95 倍'
subtitle: '在 vLLM 8K/1K 工作负载下,B200 NVFP4 路径在 30–90 tok/s/user 推理区间内每百万 tokens 成本比 H200 INT4 低 2.71x–2.95x,比同一 B200 硬件上的 INT4 低 2.45x–2.74x。三个因素——B200 的 HBM 带宽、HBM 容量和 NVFP4 张量核心——可清晰分解该优势'
seoDescription: '在 vLLM 8K/1K 下,B200 NVFP4 运行 Kimi K2.5/K2.6 每 token 成本比 H200 INT4 最多低 2.95 倍,覆盖 30–90 tok/s/user 全区间。'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'GB200 NVL72 对比 B200 运行 DeepSeek R1 670B:在 125 tok/s/user 下每 GPU 吞吐量最高达 4.4 倍'
subtitle: 'DeepSeek R1 FP4 1k/1k。NVL72 的 72-GPU NVLink 扩展域允许解码使用最高 EP=32 的宽专家并行,而 B200 的 8-GPU NVLink 岛通过 RoCEv2 上限为 EP=8'
seoDescription: 'GB200 NVL72 在 125 tok/s/user 下运行 DeepSeek R1 670B,每 GPU 吞吐量比 B200 最高提升 4.4 倍——72-GPU NVLink 支持 EP=32 宽专家并行。'
date: '2026-05-23'
publishDate: '2026-05-23'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'GB300 NVL72 vs GB200 NVL72 推理性能与性价比对比 — DeepSeek-V4-Pro 1.6T:吞吐量最高提升 2.83 倍'
subtitle: 'DSv4-Pro FP4 8K/1K,Dynamo+vLLM,两套机架均采用分离式部署。GB300 多出 50% 的 HBM(每 GPU 288 GB vs 192 GB)解锁了 GB200 无法容纳的更宽预填充+解码配方——尽管单 GPU TCO 溢价 20%,曲线中段性价比仍提升 2.31 倍。'
seoDescription: 'GB300 NVL72 多出的 50% HBM 使 DeepSeek-V4-Pro 性价比比 GB200 NVL72 最高提升 2.31 倍(vLLM FP4 8K/1K),每 GPU 吞吐量最高提升 2.83 倍。'
date: '2026-05-27'
publishDate: '2026-05-27'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'InferenceMAX:开源推理基准测试'
subtitle: 'NVIDIA GB200 NVL72、AMD MI355X、每 GPU 吞吐量 Token、延迟 Tok/s/user、性价比、每百万 Token 成本、每配置兆瓦 Token 数、DeepSeek R1 670B、GPTOSS 120B、Llama3 70B'
seoDescription: 'InferenceMAX:面向 NVIDIA GB200 NVL72 与 AMD MI355X 的开源推理基准测试——吞吐量、延迟、每 token 成本与每瓦性能。'
date: '2025-10-09'
publishDate: '2025-10-09'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'MI355X 上 DeepSeek-V4-Pro 搭配 SGLang:26 天内每 GPU 吞吐量提升 110.5 倍'
subtitle: 'amd/deepseek_v4 分支合入了 TileLang 注意力索引器、Triton 稀疏 MLA、融合 RoPE/Hadamard、FlyDSL MoE 以及 FP4 权重,历经 31 个性能优化 PR——将首次点亮时 20 tok/s/GPU、2.4 tok/s/user 的水平提升至 8K/1K 负载下 2,256 tok/s/GPU、9.4 tok/s/user,吞吐量与交互性同步攀升'
seoDescription: 'MI355X 在 SGLang 上 26 天内将 DeepSeek-V4-Pro 的每 GPU 吞吐量提升 110.5 倍——8K/1K 下经 TileLang、稀疏 MLA 与 FP4 从 20 提升至 2,256 tok/s/GPU。'
date: '2026-05-26'
publishDate: '2026-05-26'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'AMD MI355X GLM-5 推理:SGLang FP8 单节点每百万 token 成本比 B200 最高低 40%'
subtitle: 'GLM-5 发布 14 周后,AMD 在 MI355X 上同时实现了 SGLang FP8 的 MTP 和非 MTP 方案 — 通过 TileLang 实现的融合 MLA + FP8 KV 缓存在大部分性能 Pareto 前沿上将单节点 FP8 成本曲线翻转为 AMD 占优'
seoDescription: 'AMD MI355X 在 SGLang FP8 上运行 GLM-5,每百万 token 成本比 B200 最高低 40%——TileLang 融合 MLA 与 FP8 KV 缓存。'
date: '2026-05-25'
publishDate: '2026-05-25'
tags:
Expand Down
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
---
title: 'AMD MI355X Qwen3.5 397B-A17B 推理:SGLang FP8 三个月内每 GPU 吞吐量提升最高 19 倍'
subtitle: '从 v0.5.8(2 月)→ v0.5.10rc0(4 月)→ v0.5.12(5 月),三次 AITER 内核合入 MI355X 加上从 TP=8 到 TP=2/TP=4 的重新调优,将 Qwen3.5 8k/1k 峰值从 1.3k 推高至 6.4k tok/s/GPU,并将曲线延伸至 75 tok/s/user'
seoDescription: 'AMD MI355X 在 SGLang FP8 上 3 个月内将 Qwen3.5 397B-A17B 每 GPU 吞吐量最高提升 19 倍——经 AITER 内核从 1.3k 提升至 6.4k tok/s/GPU。'
date: '2026-05-25'
publishDate: '2026-05-25'
tags:
Expand Down
18 changes: 13 additions & 5 deletions packages/app/src/app/blog/[slug]/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ import { ShareTwitterButton, ShareLinkedInButton } from '@/components/share-butt
import { Card } from '@/components/ui/card';
import { JsonLd } from '@/components/json-ld';
import {
blogDescription,
getAllPosts,
getAdjacentPosts,
extractHeadings,
Expand Down Expand Up @@ -43,10 +44,14 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
const result = getPostBySlug(slug);
if (!result) return {};
const { meta } = result;
const description = blogDescription(meta);

return {
title: meta.title,
description: meta.subtitle,
// `absolute` keeps a short " | InferenceX" suffix instead of the long
// "%s | InferenceX by SemiAnalysis" root template, leaving more room for
// the headline before Google truncates the SERP title.
title: { absolute: `${meta.title} | ${SITE_NAME}` },
description,
keywords: meta.tags,
authors: [{ name: AUTHOR_NAME }],
alternates: {
Expand All @@ -56,7 +61,7 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
},
openGraph: {
title: `${meta.title} | ${SITE_NAME}`,
description: meta.subtitle,
description,
url: `${SITE_URL}/blog/${slug}`,
type: 'article',
publishedTime: `${meta.date}T00:00:00Z`,
Expand All @@ -67,7 +72,7 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
twitter: {
card: 'summary_large_image',
title: meta.title,
description: meta.subtitle,
description,
Comment thread
cursor[bot] marked this conversation as resolved.
site: AUTHOR_HANDLE,
creator: AUTHOR_HANDLE,
},
Expand Down Expand Up @@ -139,7 +144,10 @@ export default async function BlogPostPage({ params }: Props) {
publisher: { '@type': 'Organization', name: AUTHOR_NAME },
datePublished: `${meta.date}T00:00:00Z`,
...(meta.modifiedDate && { dateModified: `${meta.modifiedDate}T00:00:00Z` }),
description: meta.subtitle,
// Keep structured-data description in sync with the SERP/OG/Twitter meta
// (both go through blogDescription) so they never diverge for posts with a
// seoDescription or a long subtitle.
description: blogDescription(meta),
url: `${SITE_URL}/blog/${slug}`,
wordCount: raw.trim().split(/\s+/u).length,
timeRequired: `PT${meta.readingTime}M`,
Expand Down
36 changes: 24 additions & 12 deletions packages/app/src/app/compare/[slug]/page.tsx
Original file line number Diff line number Diff line change
@@ -1,12 +1,7 @@
import type { Metadata } from 'next';
import { notFound, permanentRedirect } from 'next/navigation';

import {
HW_REGISTRY,
SITE_NAME,
SITE_URL,
SUPPORTERS_LINE,
} from '@semianalysisai/inferencex-constants';
import { HW_REGISTRY, SITE_NAME, SITE_URL } from '@semianalysisai/inferencex-constants';

import { JsonLd } from '@/components/json-ld';
import { languageAlternates } from '@/lib/i18n';
Expand All @@ -15,11 +10,14 @@ import {
canonicalCompareSlug,
compareDisplayLabel,
compareModelDisplayLabel,
compareModelSeoName,
compareSeoTitle,
parseCompareSlug,
} from '@/lib/compare-slug';
import {
buildBreadcrumbJsonLd,
buildJsonLd,
compareMetaDescription,
compareTableNarrative,
computeCompareTableData,
dateRangeForPair,
Expand All @@ -46,16 +44,30 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
if (!parsed) return {};
const fullLabel = compareModelDisplayLabel(parsed.model, parsed.a, parsed.b);
const gpuLabel = compareDisplayLabel(parsed.a, parsed.b);
const url = `${SITE_URL}/compare/${canonicalCompareSlug(parsed.model.slug, parsed.a, parsed.b)}`;
const description = `${gpuLabel} inference benchmark on ${parsed.model.label}: verified, reproducible head-to-head results from InferenceX, the independent open-source GPU benchmark by SemiAnalysis. ${SUPPORTERS_LINE} Compare latency, throughput & cost.`;
const modelSeoName = compareModelSeoName(parsed.model);
const canonical = canonicalCompareSlug(parsed.model.slug, parsed.a, parsed.b);
const url = `${SITE_URL}/compare/${canonical}`;

// Lead the SEO title with the GPU pair — that's the phrase people search
// ("b200 vs b300") and must survive Google's ~60-char SERP truncation. The
// `absolute` form bypasses the long "%s | InferenceX by SemiAnalysis" root
// template so the query isn't pushed off the end.
const title = `${compareSeoTitle(gpuLabel, modelSeoName)} | ${SITE_NAME}`;

// Stat-led meta description built from the interpolated head-to-head numbers
// at the slug's default operating point (falls back to boilerplate for
// sparse-data pairs). Fetch is blob-cached and shared with the page render.
const rows = await getCachedBenchmarks(parsed.model.dbKeys);
const { sequence, precision } = pickPairDefaults(rows, parsed.a, parsed.b);
const { ssrRows } = computeCompareTableData(rows, parsed.a, parsed.b, sequence, precision);
const description = compareMetaDescription(parsed.model, parsed.a, parsed.b, ssrRows);

return {
title: `${fullLabel} Inference Benchmark`,
title: { absolute: title },
description,
alternates: {
canonical: url,
languages: languageAlternates(
`/compare/${canonicalCompareSlug(parsed.model.slug, parsed.a, parsed.b)}`,
),
languages: languageAlternates(`/compare/${canonical}`),
},
openGraph: {
title: `${fullLabel} | ${SITE_NAME}`,
Expand Down
24 changes: 18 additions & 6 deletions packages/app/src/app/zh/blog/[slug]/page.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,13 @@ import { ReadingProgressBar } from '@/components/blog/reading-progress-bar';
import { ShareTwitterButton, ShareLinkedInButton } from '@/components/share-buttons';
import { Card } from '@/components/ui/card';
import { JsonLd } from '@/components/json-ld';
import { getAllPosts, getAdjacentPosts, extractHeadings, getPostBySlug } from '@/lib/blog';
import {
blogDescription,
getAllPosts,
getAdjacentPosts,
extractHeadings,
getPostBySlug,
} from '@/lib/blog';
import { ZH_LANG_TAG, ZH_OG_LOCALE, zhAlternates } from '@/lib/i18n';
import {
AUTHOR_HANDLE,
Expand All @@ -38,16 +44,19 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
const result = getPostBySlug(slug, 'zh');
if (!result) return {};
const { meta } = result;
const description = blogDescription(meta);

return {
title: meta.title,
description: meta.subtitle,
// Short " | InferenceX" suffix via `absolute` (mirrors the English page)
// so the headline keeps more of the SERP title before truncation.
title: { absolute: `${meta.title} | ${SITE_NAME}` },
description,
keywords: meta.tags,
authors: [{ name: AUTHOR_NAME }],
alternates: zhAlternates(`/blog/${slug}`),
openGraph: {
title: `${meta.title} | ${SITE_NAME}`,
description: meta.subtitle,
description,
url: `${SITE_URL}/zh/blog/${slug}`,
type: 'article',
locale: ZH_OG_LOCALE,
Expand All @@ -59,7 +68,7 @@ export async function generateMetadata({ params }: Props): Promise<Metadata> {
twitter: {
card: 'summary_large_image',
title: meta.title,
description: meta.subtitle,
description,
site: AUTHOR_HANDLE,
creator: AUTHOR_HANDLE,
},
Expand Down Expand Up @@ -131,7 +140,10 @@ export default async function ZhBlogPostPage({ params }: Props) {
publisher: { '@type': 'Organization', name: AUTHOR_NAME },
datePublished: `${meta.date}T00:00:00Z`,
...(meta.modifiedDate && { dateModified: `${meta.modifiedDate}T00:00:00Z` }),
description: meta.subtitle,
// Keep structured-data description in sync with the SERP/OG/Twitter meta
// (both go through blogDescription) so they never diverge for posts with a
// seoDescription or a long subtitle.
description: blogDescription(meta),
url: `${SITE_URL}/zh/blog/${slug}`,
inLanguage: ZH_LANG_TAG,
wordCount: raw.trim().split(/\s+/u).length,
Expand Down
Loading
Loading