{
  "apiVersion": "catalog.swiss/v1",
  "site": "https://modelsphere.github.io/model-catalog/",
  "count": 18,
  "models": [
    {
      "name": "deepseek-v4-flash",
      "displayName": "DeepSeek-V4-Flash",
      "description": "DeepSeek-V4-Flash on 8 H100 GPUs. Configs use EAGLE speculative decoding.",
      "family": "deepseek",
      "tags": [
        "chat",
        "speculative-decoding"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 51,
          "report": "deepseek-v4-flash-h100-report.html"
        }
      ],
      "source": {
        "hf": "deepseek-ai/DeepSeek-V4-Flash"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/deepseek-v4-flash/deepseek-v4-flash-1.0.0.yaml",
          "digest": "sha256:4a9b78a84cdc877cdde458b58e37cf1acdd2997034a3a97c4c4bbd11a139a78e",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, EAGLE speculative decoding. +51% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "deepseek-v4-flash-0731",
      "displayName": "DeepSeek-V4-Flash-0731",
      "description": "DeepSeek-V4-Flash-0731 on 8 H100 GPUs. Configs use DSPARK speculative decoding.",
      "family": "deepseek",
      "tags": [
        "chat",
        "reasoning",
        "tool-use",
        "speculative-decoding"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 58,
          "workloads": [
            {
              "name": "50k + 1.5k",
              "uplift": 58
            },
            {
              "name": "8k + 1k",
              "uplift": 23.4
            }
          ],
          "report": "deepseek-v4-flash-0731-h100-report.html"
        }
      ],
      "source": {
        "hf": "deepseek-ai/DeepSeek-V4-Flash-0731"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/deepseek-v4-flash-0731/deepseek-v4-flash-0731-1.0.0.yaml",
          "digest": "sha256:67c88ae9c47318b8419eb32e89db2e01d615eb219bd86f396da5539dfcdcb648",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, DSPARK speculative decoding. Normalized throughput within the SLO vs the baseline variant: +58.0% (50k + 1.5k), +23.4% (8k + 1k).",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "deepseek-v4.1-flash",
      "displayName": "DeepSeek-V4.1-Flash",
      "description": "DeepSeek-V4.1-Flash on 8 H100 GPUs.",
      "family": "deepseek",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 89,
          "report": "deepseek-v4.1-flash-h100-report.html"
        }
      ],
      "source": {
        "hf": "deepseek-ai/DeepSeek-V4.1-Flash"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/deepseek-v4.1-flash/deepseek-v4.1-flash-1.0.0.yaml",
          "digest": "sha256:b83e3b7467a242a5784bae0f3513fc6ce494e29a9701624c0ee67b8475302875",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. +89% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "glm-5.3-flash",
      "displayName": "GLM-5.3-Flash",
      "description": "GLM-5.3-Flash TP8 on eight H100 GPUs",
      "family": "glm",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 5,
          "report": "glm-5.3-flash-h100-report.html"
        }
      ],
      "source": {
        "hf": "zai-org/GLM-5.3-Flash"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/glm-5.3-flash/glm-5.3-flash-1.0.0.yaml",
          "digest": "sha256:df8e48d2fa630e4d2cc57f8fb43b27088fd43e16c8c3e3410812b1deafb7c54b",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. +5% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "glm5.1",
      "displayName": "GLM 5.1 (Qwen3.6-35B-A3B)",
      "description": "Qwen3.6-35B-A3B-793303-glm-5, served as \"glm-5\". Needs a merged MoE config directory on every node it can land on; SGLANG_MOE_CONFIG_DIR points at it.",
      "family": "qwen",
      "tags": [
        "chat",
        "moe",
        "tool-use"
      ],
      "source": {
        "hf": "modelforge/Qwen3.6-35B-A3B-793303-glm-5"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/glm5.1/glm5.1-1.0.0.yaml",
          "digest": "sha256:241016119677dc257af2412e3b52a26634dc18ffbef0c3082596d5daab5952ea",
          "variants": [
            {
              "id": "sglang-tp2",
              "engine": "sglang",
              "default": true,
              "description": "TP2 on two GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.0"
              },
              "requires": {
                "gpus": 2,
                "topology": "single-node"
              }
            }
          ]
        }
      ]
    },
    {
      "name": "glm5.3",
      "displayName": "GLM 5.3 (NVFP4)",
      "description": "GLM-5.3 quantised to NVFP4, served as \"glm-5.3\" on four B300 GPUs. EAGLE speculative decoding over a 150 GiB hierarchical KV cache. The FP4 kernels are Blackwell-only, so there is no A100 or H100 variant to fall back to.",
      "family": "glm",
      "tags": [
        "chat",
        "reasoning",
        "tool-use",
        "fp4",
        "speculative-decoding"
      ],
      "source": {
        "hf": "zai-org/GLM-5.3-NVFP4"
      },
      "latest": "1.0.1",
      "versions": [
        {
          "version": "1.0.1",
          "path": "models/glm5.3/glm5.3-1.0.1.yaml",
          "digest": "sha256:0f82e50f1f4bdd65b948b409b13722385eb164e59ea1787a5398f806e2d684da",
          "variants": [
            {
              "id": "sglang-tp4-b300",
              "engine": "sglang",
              "default": true,
              "description": "TP4 on four B300 GPUs. NVFP4 weights, EAGLE speculative decoding, hierarchical KV cache.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.8"
              },
              "requires": {
                "gpus": 4,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            }
          ]
        },
        {
          "version": "1.0.0",
          "path": "models/glm5.3/glm5.3-1.0.0.yaml",
          "digest": "sha256:1296677c857314d885333a361f63b2a079798afeae1fb77c38a8f2d1d25b2fb9",
          "variants": [
            {
              "id": "sglang-tp4-b300",
              "engine": "sglang",
              "default": true,
              "description": "TP4 on four B300 GPUs. NVFP4 weights, EAGLE speculative decoding, hierarchical KV cache.",
              "chart": {
                "name": "sglang",
                "version": "0.7.1"
              },
              "requires": {
                "gpus": 4,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "gpt-oss-120b",
      "displayName": "gpt-oss-120b",
      "description": "gpt-oss-120b on 8 H100 GPUs. Configs use fp8 KV cache.",
      "family": "gpt",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp2-h100-baseline",
          "optimized": "sglang-tp2-dp4-h100-optimized",
          "uplift": 44,
          "report": "gpt-oss-120b-h100-report.html"
        }
      ],
      "source": {
        "hf": "openai-mirror/gpt-oss-120b"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/gpt-oss-120b/gpt-oss-120b-1.0.0.yaml",
          "digest": "sha256:f31de789a4b40a96189760ba9729e5ab2246d177e22a24d503d261bb5d02a5eb",
          "variants": [
            {
              "id": "sglang-tp2-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP2 on two H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 2,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp2-dp4-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP2 DP4 on eight H100 GPUs, single node, fp8 KV cache, EAGLE3 speculative decoding (lmsys/EAGLE3-gpt-oss-120b-bf16).  +44% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "hy3",
      "displayName": "Hy3",
      "description": "Hy3 on 8 H100 GPUs. Configs use NEXTN speculative decoding.",
      "family": "hunyuan",
      "tags": [
        "chat",
        "moe",
        "reasoning",
        "tool-use",
        "speculative-decoding"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 64.2,
          "workloads": [
            {
              "name": "50k + 1.5k",
              "uplift": 64.2
            },
            {
              "name": "8k + 1k",
              "uplift": 7.6
            }
          ],
          "report": "hy3-h100-report.html"
        }
      ],
      "source": {
        "hf": "tencent/Hy3-FP8"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/hy3/hy3-1.0.0.yaml",
          "digest": "sha256:8941058ce7a9e536377f9a25694341acd440eee4c8b3d483df52d3f74dda95fe",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, NEXTN speculative decoding. Normalized throughput within the SLO vs the baseline variant: +64.2% (50k + 1.5k), +7.6% (8k + 1k).",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "kimi-k2.5",
      "displayName": "Kimi K2.5",
      "description": "Does not fit on one node. A two-pod LeaderWorkerSet group, TP8 inside each node and PP2 across the InfiniBand fabric between them. Needs the LWS controller, rdma-shared-dev-plugin and rdma-injector.",
      "family": "kimi",
      "tags": [
        "chat",
        "reasoning",
        "long-context",
        "multi-node"
      ],
      "source": {
        "hf": "moonshotai/Kimi-K2.5"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/kimi-k2.5/kimi-k2.5-1.0.0.yaml",
          "digest": "sha256:fdb3f8f7b58ea31c7ae717891b88b96faeb2a03aa7debfd1e119e4f2084537f4",
          "variants": [
            {
              "id": "sglang-pp2-lws",
              "engine": "sglang",
              "default": true,
              "description": "TP8 per node, PP2 across two nodes over IB.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.0"
              },
              "requires": {
                "gpus": 8,
                "nodes": 2,
                "topology": "lws",
                "rdma": true
              }
            }
          ]
        }
      ]
    },
    {
      "name": "kimi-k3",
      "displayName": "Kimi K3",
      "description": "Kimi K3 on a single eight-GPU B300 node -- TP8 with decode context parallel across the same eight cards, and an fp8 KV cache. DSPARK speculative decoding needs the Kimi-K3-DSpark draft weights present on every node it can land on, mounted beside the target model.",
      "family": "kimi",
      "tags": [
        "chat",
        "reasoning",
        "tool-use",
        "long-context",
        "speculative-decoding"
      ],
      "source": {
        "hf": "moonshotai/Kimi-K3"
      },
      "latest": "1.0.2",
      "versions": [
        {
          "version": "1.0.2",
          "path": "models/kimi-k3/kimi-k3-1.0.2.yaml",
          "digest": "sha256:8714a07e22ecb033403cb9dacd31423988bf3ce4496028d8673e7a0e06a6184f",
          "variants": [
            {
              "id": "sglang-tp8-b300",
              "engine": "sglang",
              "default": true,
              "description": "TP8 with DCP8 on one eight-GPU B300 node, fp8 KV cache, DSPARK speculative decoding.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.9"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            },
            {
              "id": "vllm-tp8-b300",
              "engine": "vllm",
              "chart": {
                "name": "vllm",
                "version": ">=0.4.6"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            }
          ]
        },
        {
          "version": "1.0.1",
          "path": "models/kimi-k3/kimi-k3-1.0.1.yaml",
          "digest": "sha256:6ffbd93351deba5a49dbf728318d6cc17fe4149e0510a3c453d386af5b757464",
          "variants": [
            {
              "id": "sglang-tp8-b300",
              "engine": "sglang",
              "default": true,
              "description": "TP8 with DCP8 on one eight-GPU B300 node, fp8 KV cache, DSPARK speculative decoding.",
              "chart": {
                "name": "sglang",
                "version": "0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            },
            {
              "id": "vllm-tp8-b300",
              "engine": "vllm",
              "chart": {
                "name": "vllm",
                "version": "0.4.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            }
          ]
        },
        {
          "version": "1.0.0",
          "path": "models/kimi-k3/kimi-k3-1.0.0.yaml",
          "digest": "sha256:42666567313e6220c73c28d0101a897455d7eb27a77c93209afac3b6ca06becb",
          "variants": [
            {
              "id": "sglang-tp8-b300",
              "engine": "sglang",
              "default": true,
              "description": "TP8 with DCP8 on one eight-GPU B300 node, fp8 KV cache, DSPARK speculative decoding.",
              "chart": {
                "name": "sglang",
                "version": "0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-B300-SXM6-AC"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "laguna-s-2.1",
      "displayName": "Laguna-S-2.1",
      "description": "Laguna-S-2.1 on 4 or 8 H100 GPUs. Configs use fp8 KV cache.",
      "family": "laguna",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp4-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 26,
          "report": "laguna-s-2.1-h100-report.html"
        }
      ],
      "source": {
        "hf": "poolside/Laguna-S-2.1"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/laguna-s-2.1/laguna-s-2.1-1.0.0.yaml",
          "digest": "sha256:38d4620b23d29b1e00a227dedc746079ab6c4260d5c069533b453fc9c40be85c",
          "variants": [
            {
              "id": "sglang-tp4-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP4 on four H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 4,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, fp8 KV cache. +26% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "ling-3.0-flash",
      "displayName": "Ling-3.0-flash",
      "description": "Ling-3.0-flash on 8 H100 GPUs.",
      "family": "ling",
      "tags": [
        "chat",
        "moe",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 15.5,
          "workloads": [
            {
              "name": "50k + 1.5k",
              "uplift": 15.5
            },
            {
              "name": "8k + 1k",
              "uplift": 21.3
            }
          ],
          "report": "ling-3.0-flash-h100-report.html"
        }
      ],
      "source": {
        "hf": "inclusionAI/Ling-3.0-flash"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/ling-3.0-flash/ling-3.0-flash-1.0.0.yaml",
          "digest": "sha256:0830c77008b3bc5a7e1d1024d00ab9314762719ca99ad4a997d9fe281918456e",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. Normalized throughput within the SLO vs the baseline variant: +15.5% (50k + 1.5k), +21.3% (8k + 1k).",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "ling-3.0-flash-vl",
      "displayName": "Ling-3.0-flash-VL",
      "description": "Ling-3.0-flash-VL on 8 H100 GPUs.",
      "family": "ling",
      "tags": [
        "chat",
        "reasoning",
        "tool-use",
        "vision"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 15,
          "report": "ling-3.0-flash-vl-h100-report.html"
        }
      ],
      "source": {
        "hf": "inclusionAI/Ling-3.0-flash-VL"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/ling-3.0-flash-vl/ling-3.0-flash-vl-1.0.0.yaml",
          "digest": "sha256:dd05fd435c6e347c2c1b1b82abdcf88896b3438131bf02e79bf57613baf1d5f1",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. +15% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "mimo-v2.5",
      "displayName": "MiMo-V2.5",
      "description": "MiMo-V2.5 on 8 H100 GPUs. Configs use fp8 KV cache, EAGLE speculative decoding.",
      "family": "mimo",
      "tags": [
        "chat",
        "reasoning",
        "tool-use",
        "speculative-decoding"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 121,
          "report": "mimo-v2.5-h100-report.html"
        }
      ],
      "source": {
        "hf": "XiaomiMiMo/MiMo-V2.5"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/mimo-v2.5/mimo-v2.5-1.0.0.yaml",
          "digest": "sha256:bde5a122e647760bfff7cfbbd2e9cbdc225c7ca997417b7fb4e4334367fc7583",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, fp8 KV cache, EAGLE speculative decoding. +121% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "nemotron-3.5-lightning-30b-a3b",
      "displayName": "Nemotron-3.5-Lightning-30B-A3B",
      "description": "Nemotron-3.5-Lightning-30B-A3B on 1 or 8 H100 GPUs.",
      "family": "nemotron",
      "tags": [
        "chat",
        "moe",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp1-h100-baseline",
          "optimized": "sglang-tp1-dp8-h100-optimized",
          "uplift": 27,
          "report": "nemotron-3.5-lightning-30b-a3b-h100-report.html"
        }
      ],
      "source": {
        "hf": "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/nemotron-3.5-lightning-30b-a3b/nemotron-3.5-lightning-30b-a3b-1.0.0.yaml",
          "digest": "sha256:66118bc88e8fd9c13f36e9d3daf60f47c1619e944a0613a41121cc5dbd8557cc",
          "variants": [
            {
              "id": "sglang-tp1-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP1 on one H100 GPU, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 1,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp1-dp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP1 DP8 on eight H100 GPUs, single node. +27% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "nex-n2.5-pro",
      "displayName": "Nex-N2.5-Pro",
      "description": "Nex-N2.5-Pro on 8 H100 GPUs. Configs use fp8 KV cache.",
      "family": "nex",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 24,
          "report": "nex-n2.5-pro-h100-report.html"
        }
      ],
      "source": {
        "hf": "nex-agi/Nex-N2.5-Pro"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/nex-n2.5-pro/nex-n2.5-pro-1.0.0.yaml",
          "digest": "sha256:a8dac321c540171666e395a9e7204d6c2548b4d47b0f374ef862c983707f875f",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node, fp8 KV cache. +24% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "qwen3.6-35b-a3b",
      "displayName": "Qwen3.6 35B-A3B",
      "description": "Stock Qwen3.6-35B-A3B on two H100 or H800 GPUs or eight H100 GPUs, served as \"qwen\". The upstream weights, not one of the modelforge builds derived from them — those are published separately and served under other names.",
      "family": "qwen",
      "tags": [
        "chat",
        "moe",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.1.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 4.3,
          "workloads": [
            {
              "name": "50k + 1.5k",
              "uplift": 4.3
            },
            {
              "name": "8k + 1k",
              "uplift": 10
            }
          ],
          "report": "qwen3.6-35b-a3b-h100-report.html"
        }
      ],
      "source": {
        "hf": "Qwen/Qwen3.6-35B-A3B"
      },
      "latest": "1.1.0",
      "versions": [
        {
          "version": "1.1.0",
          "path": "models/qwen3.6-35b-a3b/qwen3.6-35b-a3b-1.1.0.yaml",
          "digest": "sha256:e86386a18a52bc1a5790cfcadc8e7601f5a3fbabf2806ee70af646a5eb0aef6f",
          "variants": [
            {
              "id": "sglang-tp2-h100",
              "engine": "sglang",
              "default": true,
              "description": "TP2 on two H100 or H800 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 2,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3",
                  "NVIDIA-H800"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. Normalized throughput within the SLO vs the baseline variant: +4.3% (50k + 1.5k), +10.0% (8k + 1k).",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        },
        {
          "version": "1.0.0",
          "path": "models/qwen3.6-35b-a3b/qwen3.6-35b-a3b-1.0.0.yaml",
          "digest": "sha256:c29c8ca02e33295a436debf518b88de60d358c08734f86b7598fe10f32128ece",
          "variants": [
            {
              "id": "sglang-tp2-h100",
              "engine": "sglang",
              "default": true,
              "description": "TP2 on two H100 or H800 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": "0.7.1"
              },
              "requires": {
                "gpus": 2,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3",
                  "NVIDIA-H800"
                ]
              }
            }
          ]
        }
      ]
    },
    {
      "name": "qwen3.8-flash-next",
      "displayName": "Qwen3.8-Flash-Next",
      "description": "Qwen3.8-Flash-Next on 8 H100 GPUs.",
      "family": "qwen",
      "tags": [
        "chat",
        "reasoning",
        "tool-use"
      ],
      "tuning": [
        {
          "version": "1.0.0",
          "baseline": "sglang-tp8-h100-baseline",
          "optimized": "sglang-tp8-h100-optimized",
          "uplift": 6,
          "report": "qwen3.8-flash-next-h100-report.html"
        }
      ],
      "source": {
        "hf": "Qwen/Qwen3.8-Flash-Next"
      },
      "latest": "1.0.0",
      "versions": [
        {
          "version": "1.0.0",
          "path": "models/qwen3.8-flash-next/qwen3.8-flash-next-1.0.0.yaml",
          "digest": "sha256:5befa6f279c07664df1f5abf70277ae3767314c34387087d778e3aecb3d66944",
          "variants": [
            {
              "id": "sglang-tp8-h100-baseline",
              "engine": "sglang",
              "description": "Baseline: TP8 on eight H100 GPUs, single node.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            },
            {
              "id": "sglang-tp8-h100-optimized",
              "engine": "sglang",
              "default": true,
              "description": "Tuned by LLM AutoTune: TP8 on eight H100 GPUs, single node. +6% on the tuning benchmark vs the baseline variant.",
              "chart": {
                "name": "sglang",
                "version": ">=0.7.1"
              },
              "requires": {
                "gpus": 8,
                "topology": "single-node",
                "gpuProduct": [
                  "NVIDIA-H100-80GB-HBM3"
                ]
              }
            }
          ]
        }
      ]
    }
  ]
}
