> ## Documentation Index
> Fetch the complete documentation index at: https://docs.sambanova.ai/docs/llms.txt
> Use this file to discover all available pages before exploring further.

# Supported models and bundles

export const StackModelCard = ({provider = "", id = "", use = "", modalities = [], capabilities = [], endpoints = [], features = [], bundles = [], hf = "", customCheckpoints, draftModels = [], sequenceLengths = [], specDecoding = false}) => {
  const MARK_PATHS = {
    deepseek: `<path d="M23.748 4.651c-.254-.124-.364.113-.512.233-.051.04-.094.09-.137.137-.372.397-.806.657-1.373.626-.829-.046-1.537.214-2.163.848-.133-.782-.575-1.248-1.247-1.548-.352-.155-.708-.311-.955-.65-.172-.24-.219-.509-.305-.774-.055-.16-.11-.323-.293-.35-.2-.031-.278.136-.356.276-.313.572-.434 1.202-.422 1.84.027 1.436.633 2.58 1.838 3.393.137.094.172.187.129.323-.082.28-.18.553-.266.833-.055.179-.137.218-.328.14a5.5 5.5 0 0 1-1.737-1.179c-.857-.828-1.631-1.743-2.597-2.46a12 12 0 0 0-.689-.47c-.985-.957.13-1.743.387-1.836.27-.098.094-.433-.778-.428-.872.003-1.67.295-2.687.685a3 3 0 0 1-.465.136 9.6 9.6 0 0 0-2.883-.101c-1.885.21-3.39 1.1-4.497 2.622C.082 8.776-.231 10.854.152 13.02c.403 2.284 1.568 4.175 3.36 5.653 1.857 1.533 3.997 2.284 6.438 2.14 1.482-.085 3.132-.284 4.994-1.86.47.234.962.328 1.78.398.629.058 1.235-.031 1.705-.129.735-.155.684-.836.418-.961-2.155-1.004-1.682-.595-2.112-.926 1.095-1.295 2.768-3.598 3.284-6.733.05-.346.115-.834.108-1.114-.004-.171.035-.238.23-.257a4.2 4.2 0 0 0 1.545-.475c1.397-.763 1.96-2.016 2.093-3.517.02-.23-.004-.467-.247-.588M11.58 18.168c-2.088-1.642-3.101-2.183-3.52-2.16-.39.024-.32.472-.234.763.09.288.207.487.371.74.114.167.192.416-.113.603-.673.416-1.842-.14-1.897-.168-1.361-.801-2.5-1.86-3.301-3.306-.775-1.393-1.225-2.888-1.299-4.482-.02-.385.094-.522.477-.592a4.7 4.7 0 0 1 1.53-.038c2.131.311 3.946 1.264 5.467 2.774.868.86 1.525 1.887 2.202 2.89.72 1.066 1.494 2.082 2.48 2.915.348.291.626.513.892.677-.802.09-2.14.109-3.055-.615zm1.001-6.44a.306.306 0 0 1 .415-.287.3.3 0 0 1 .113.074.3.3 0 0 1 .086.214c0 .17-.136.307-.308.307a.303.303 0 0 1-.306-.307m3.11 1.596c-.2.081-.4.151-.591.16a1.25 1.25 0 0 1-.798-.254c-.274-.23-.47-.358-.551-.758a1.7 1.7 0 0 1 .015-.588c.07-.327-.007-.537-.238-.727-.188-.156-.426-.199-.689-.199a.6.6 0 0 1-.254-.078.253.253 0 0 1-.114-.358 1 1 0 0 1 .192-.21c.356-.202.767-.136 1.146.016.352.144.618.408 1.001.782.392.451.462.576.685.915.176.264.336.536.446.848.066.194-.02.353-.25.45"/>`,
    generic: `<path fill-rule="evenodd" clip-rule="evenodd" d="M12 1.5 21.5 7v10L12 22.5 2.5 17V7L12 1.5Zm0 2.31L4.5 8.16v7.68L12 20.19l7.5-4.35V8.16L12 3.81Z"/>`,
    google: `<path d="M12.48 10.92v3.28h7.84c-.24 1.84-.853 3.187-1.787 4.133-1.147 1.147-2.933 2.4-6.053 2.4-4.827 0-8.6-3.893-8.6-8.72s3.773-8.72 8.6-8.72c2.6 0 4.507 1.027 5.907 2.347l2.307-2.307C18.747 1.44 16.133 0 12.48 0 5.867 0 .307 5.387.307 12s5.56 12 12.173 12c3.573 0 6.267-1.173 8.373-3.36 2.16-2.16 2.84-5.213 2.84-7.667 0-.76-.053-1.467-.173-2.053H12.48z"/>`,
    meta: `<path d="M6.915 4.03c-1.968 0-3.683 1.28-4.871 3.113C.704 9.208 0 11.883 0 14.449c0 .706.07 1.369.21 1.973a6.624 6.624 0 0 0 .265.86 5.297 5.297 0 0 0 .371.761c.696 1.159 1.818 1.927 3.593 1.927 1.497 0 2.633-.671 3.965-2.444.76-1.012 1.144-1.626 2.663-4.32l.756-1.339.186-.325c.061.1.121.196.183.3l2.152 3.595c.724 1.21 1.665 2.556 2.47 3.314 1.046.987 1.992 1.22 3.06 1.22 1.075 0 1.876-.355 2.455-.843a3.743 3.743 0 0 0 .81-.973c.542-.939.861-2.127.861-3.745 0-2.72-.681-5.357-2.084-7.45-1.282-1.912-2.957-2.93-4.716-2.93-1.047 0-2.088.467-3.053 1.308-.652.57-1.257 1.29-1.82 2.05-.69-.875-1.335-1.547-1.958-2.056-1.182-.966-2.315-1.303-3.454-1.303zm10.16 2.053c1.147 0 2.188.758 2.992 1.999 1.132 1.748 1.647 4.195 1.647 6.4 0 1.548-.368 2.9-1.839 2.9-.58 0-1.027-.23-1.664-1.004-.496-.601-1.343-1.878-2.832-4.358l-.617-1.028a44.908 44.908 0 0 0-1.255-1.98c.07-.109.141-.224.211-.327 1.12-1.667 2.118-2.602 3.358-2.602zm-10.201.553c1.265 0 2.058.791 2.675 1.446.307.327.737.871 1.234 1.579l-1.02 1.566c-.757 1.163-1.882 3.017-2.837 4.338-1.191 1.649-1.81 1.817-2.486 1.817-.524 0-1.038-.237-1.383-.794-.263-.426-.464-1.13-.464-2.046 0-2.221.63-4.535 1.66-6.088.454-.687.964-1.226 1.533-1.533a2.264 2.264 0 0 1 1.088-.285z"/>`,
    minimax: `<path d="M11.43 3.92a.86.86 0 1 0-1.718 0v14.236a1.999 1.999 0 0 1-3.997 0V9.022a.86.86 0 1 0-1.718 0v3.87a1.999 1.999 0 0 1-3.997 0V11.49a.57.57 0 0 1 1.139 0v1.404a.86.86 0 0 0 1.719 0V9.022a1.999 1.999 0 0 1 3.997 0v9.134a.86.86 0 0 0 1.719 0V3.92a1.998 1.998 0 1 1 3.996 0v11.788a.57.57 0 1 1-1.139 0zm10.572 3.105a2 2 0 0 0-1.999 1.997v7.63a.86.86 0 0 1-1.718 0V3.923a1.999 1.999 0 0 0-3.997 0v16.16a.86.86 0 0 1-1.719 0V18.08a.57.57 0 1 0-1.138 0v2a1.998 1.998 0 0 0 3.996 0V3.92a.86.86 0 0 1 1.719 0v12.73a1.999 1.999 0 0 0 3.996 0V9.023a.86.86 0 1 1 1.72 0v6.686a.57.57 0 0 0 1.138 0V9.022a2 2 0 0 0-1.998-1.997"/>`,
    mistralai: `<path d="M17.143 3.429v3.428h-3.429v3.429h-3.428V6.857H6.857V3.43H3.43v13.714H0v3.428h10.286v-3.428H6.857v-3.429h3.429v3.429h3.429v-3.429h3.428v3.429h-3.428v3.428H24v-3.428h-3.43V3.429z"/>`,
    openai: `<path d="M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.0729zm-9.022 12.6081a4.4755 4.4755 0 0 1-2.8764-1.0408l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.872zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231l-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654l2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z"/>`,
    qwen: `<path d="M23.919 14.545 20.817 9.17l1.47-2.544a.56.56 0 0 0 0-.566l-1.633-2.83a.57.57 0 0 0-.49-.283h-6.207L12.487.402a.57.57 0 0 0-.49-.284H8.732a.56.56 0 0 0-.49.284L5.139 5.775h-2.94a.56.56 0 0 0-.49.284L.077 8.887a.56.56 0 0 0 0 .567L3.18 14.83l-1.47 2.545a.56.56 0 0 0 0 .566l1.634 2.83a.57.57 0 0 0 .49.283h6.205l1.47 2.545a.57.57 0 0 0 .49.284h3.266a.57.57 0 0 0 .49-.284l3.104-5.375h2.94a.57.57 0 0 0 .49-.283l1.634-2.828a.55.55 0 0 0-.004-.568M8.733.686l1.634 2.828-1.634 2.828H21.8L20.164 9.17H7.425L5.63 6.06Zm1.306 19.801-6.205-.002 1.634-2.83h3.265L2.201 6.344h3.267q3.182 5.517 6.367 11.032zm10.124-5.66L18.53 12l-6.532 11.315-1.634-2.83c2.129-3.673 4.25-7.351 6.373-11.028h3.592l3.102 5.374z"/>`
  };
  const PROVIDER_MARKS = {
    Meta: {
      icon: "meta",
      c: "#0467DF",
      d: "#2279E3"
    },
    MiniMax: {
      icon: "minimax",
      c: "#E73562",
      d: "#E73562"
    },
    "Mistral AI": {
      icon: "mistralai",
      c: "#FA520F",
      d: "#FA520F"
    },
    DeepSeek: {
      icon: "deepseek",
      c: "#5786FE",
      d: "#5786FE"
    },
    OpenAI: {
      icon: "openai",
      c: "#0D0D0D",
      d: "#F5F5F5"
    },
    Google: {
      icon: "google",
      c: "#4285F4",
      d: "#4285F4"
    },
    "Alibaba Cloud": {
      icon: "qwen",
      c: "#6950EF",
      d: "#7B65F1"
    }
  };
  const NEUTRAL = {
    icon: "generic",
    c: "#645D70",
    d: "#A9A1B6"
  };
  const mk = PROVIDER_MARKS[provider] || NEUTRAL;
  const dotFor = v => v === "Supported" ? "sst-dot sst-dot-ok" : v === "Limited" ? "sst-dot sst-dot-lim" : "sst-dot sst-dot-no";
  const featClass = v => v === "Supported" ? "sst-feat sst-feat-ok" : v === "Limited" ? "sst-feat sst-feat-lim" : "sst-feat";
  const slug = String(id).toLowerCase().replace(/[^a-z0-9_]+/g, "-").replace(/^-|-$/g, "");
  const [open, setOpen] = useState(false);
  useEffect(() => {
    const sync = () => {
      if (decodeURIComponent(window.location.hash).slice(1) === slug) setOpen(true);
    };
    sync();
    window.addEventListener("hashchange", sync);
    return () => window.removeEventListener("hashchange", sync);
  }, [slug]);
  return <div className="sst">
      <details className="sst-card sst-det" open={open} onToggle={e => setOpen(e.currentTarget.open)}>
        <summary className="sst-head">
          <div className="sst-head-l">
            <span className="sst-mark" style={{
    "--mk": mk.c,
    "--mkd": mk.d
  }} aria-hidden="true">
              <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
    __html: MARK_PATHS[mk.icon]
  }} />
            </span>
            <div className="sst-head-txt">
              <div className="sst-prov">{provider}</div>
              <div className="sst-id">{id}</div>
              {use && <p className="sst-use">{use}</p>}
            </div>
          </div>
          <svg className="sst-chev" viewBox="0 0 24 24" fill="none" stroke="currentColor" strokeWidth="2" strokeLinecap="round" strokeLinejoin="round" aria-hidden="true">
            <path d="m6 9 6 6 6-6" />
          </svg>
        </summary>

        <div className="sst-body">
          <div>
            <p className="sst-lbl">Modalities</p>
            <div className="sst-row">
              {modalities.length ? modalities.map(x => <span className="sst-tag sst-tag-mod" key={x}>
                    {x}
                  </span>) : <span className="sst-tag">—</span>}
            </div>
          </div>

          {capabilities.length > 0 && <div>
              <p className="sst-lbl">Capabilities</p>
              <div className="sst-row">
                {capabilities.map(x => <span className="sst-tag" key={x}>
                    {x}
                  </span>)}
              </div>
            </div>}

          {endpoints.length > 0 && <div>
              <p className="sst-lbl">Endpoints</p>
              <div className="sst-eps">
                {endpoints.map(([name, path, verdict]) => <div className={verdict ? "sst-ep" : "sst-ep sst-ep-off"} key={name}>
                    <span className={dotFor(verdict)} />
                    <div className="sst-ep-t">
                      <div className="sst-ep-n">{name}</div>
                      <div className="sst-ep-p">{path}</div>
                    </div>
                  </div>)}
              </div>
            </div>}

          {features.length > 0 && <div>
              <p className="sst-lbl">Features</p>
              <div className="sst-feats">
                {features.map(([name, verdict]) => <span className={featClass(verdict)} key={name}>
                    <span className={dotFor(verdict)} />
                    {name}
                    {verdict === "Limited" && " (limited)"}
                  </span>)}
              </div>
            </div>}

          {bundles.length > 0 && <div>
              <p className="sst-lbl">Bundle configurations</p>
              <div className="sst-bundles">
                {bundles.map(([name, ctx, isSuggested, status]) => <div className={isSuggested ? "sst-bundle sst-bundle-sug" : "sst-bundle"} key={name}>
                    <span className="sst-bn">{name}</span>
                    <span className="sst-stat">{ctx}</span>
                    {status && <span className={"sst-pef sst-pef-" + status} title={"PEF status: " + status}>
                        {status}
                      </span>}
                    {isSuggested && <span className="sst-sug">Suggested</span>}
                  </div>)}
              </div>
            </div>}

          {sequenceLengths.length > 0 && <div>
              <p className="sst-lbl">Context length (batch size)</p>
              <div className="sst-bundles">
                {sequenceLengths.map(sl => <div className="sst-bundle" key={sl}>
                    <span className="sst-bn">{sl}</span>
                  </div>)}
              </div>
            </div>}

          {(typeof customCheckpoints === "boolean" || draftModels.length > 0 || specDecoding) && <div>
              <p className="sst-lbl">Deployment options</p>
              <div className="sst-opts">
                {typeof customCheckpoints === "boolean" && <div className="sst-opt">
                    <span className="sst-opt-k">Custom checkpoints (BYOC)</span>
                    <span className="sst-opt-v">{customCheckpoints ? "Supported" : "Not supported"}</span>
                  </div>}
                {draftModels.length > 0 ? <div className="sst-opt">
                    <span className="sst-opt-k">Speculative decoding</span>
                    <span className="sst-opt-v">
                      Draft {draftModels.length > 1 ? "models" : "model"}:{" "}
                      {draftModels.map(d => <code key={d}>{d}</code>)}
                    </span>
                  </div> : specDecoding ? <div className="sst-opt">
                    <span className="sst-opt-k">Speculative decoding</span>
                    <span className="sst-opt-v">Supported</span>
                  </div> : null}
              </div>
            </div>}

          {hf && <a className="sst-link" href={hf} target="_blank" rel="noreferrer">
              View model card on Hugging Face ↗
            </a>}
        </div>
      </details>
    </div>;
};

export const StackModelExplorer = ({models = []}) => {
  const BRAND = "#974fc7";
  const MARK_PATHS = {
    deepseek: `<path d="M23.748 4.651c-.254-.124-.364.113-.512.233-.051.04-.094.09-.137.137-.372.397-.806.657-1.373.626-.829-.046-1.537.214-2.163.848-.133-.782-.575-1.248-1.247-1.548-.352-.155-.708-.311-.955-.65-.172-.24-.219-.509-.305-.774-.055-.16-.11-.323-.293-.35-.2-.031-.278.136-.356.276-.313.572-.434 1.202-.422 1.84.027 1.436.633 2.58 1.838 3.393.137.094.172.187.129.323-.082.28-.18.553-.266.833-.055.179-.137.218-.328.14a5.5 5.5 0 0 1-1.737-1.179c-.857-.828-1.631-1.743-2.597-2.46a12 12 0 0 0-.689-.47c-.985-.957.13-1.743.387-1.836.27-.098.094-.433-.778-.428-.872.003-1.67.295-2.687.685a3 3 0 0 1-.465.136 9.6 9.6 0 0 0-2.883-.101c-1.885.21-3.39 1.1-4.497 2.622C.082 8.776-.231 10.854.152 13.02c.403 2.284 1.568 4.175 3.36 5.653 1.857 1.533 3.997 2.284 6.438 2.14 1.482-.085 3.132-.284 4.994-1.86.47.234.962.328 1.78.398.629.058 1.235-.031 1.705-.129.735-.155.684-.836.418-.961-2.155-1.004-1.682-.595-2.112-.926 1.095-1.295 2.768-3.598 3.284-6.733.05-.346.115-.834.108-1.114-.004-.171.035-.238.23-.257a4.2 4.2 0 0 0 1.545-.475c1.397-.763 1.96-2.016 2.093-3.517.02-.23-.004-.467-.247-.588M11.58 18.168c-2.088-1.642-3.101-2.183-3.52-2.16-.39.024-.32.472-.234.763.09.288.207.487.371.74.114.167.192.416-.113.603-.673.416-1.842-.14-1.897-.168-1.361-.801-2.5-1.86-3.301-3.306-.775-1.393-1.225-2.888-1.299-4.482-.02-.385.094-.522.477-.592a4.7 4.7 0 0 1 1.53-.038c2.131.311 3.946 1.264 5.467 2.774.868.86 1.525 1.887 2.202 2.89.72 1.066 1.494 2.082 2.48 2.915.348.291.626.513.892.677-.802.09-2.14.109-3.055-.615zm1.001-6.44a.306.306 0 0 1 .415-.287.3.3 0 0 1 .113.074.3.3 0 0 1 .086.214c0 .17-.136.307-.308.307a.303.303 0 0 1-.306-.307m3.11 1.596c-.2.081-.4.151-.591.16a1.25 1.25 0 0 1-.798-.254c-.274-.23-.47-.358-.551-.758a1.7 1.7 0 0 1 .015-.588c.07-.327-.007-.537-.238-.727-.188-.156-.426-.199-.689-.199a.6.6 0 0 1-.254-.078.253.253 0 0 1-.114-.358 1 1 0 0 1 .192-.21c.356-.202.767-.136 1.146.016.352.144.618.408 1.001.782.392.451.462.576.685.915.176.264.336.536.446.848.066.194-.02.353-.25.45"/>`,
    generic: `<path fill-rule="evenodd" clip-rule="evenodd" d="M12 1.5 21.5 7v10L12 22.5 2.5 17V7L12 1.5Zm0 2.31L4.5 8.16v7.68L12 20.19l7.5-4.35V8.16L12 3.81Z"/>`,
    google: `<path d="M12.48 10.92v3.28h7.84c-.24 1.84-.853 3.187-1.787 4.133-1.147 1.147-2.933 2.4-6.053 2.4-4.827 0-8.6-3.893-8.6-8.72s3.773-8.72 8.6-8.72c2.6 0 4.507 1.027 5.907 2.347l2.307-2.307C18.747 1.44 16.133 0 12.48 0 5.867 0 .307 5.387.307 12s5.56 12 12.173 12c3.573 0 6.267-1.173 8.373-3.36 2.16-2.16 2.84-5.213 2.84-7.667 0-.76-.053-1.467-.173-2.053H12.48z"/>`,
    meta: `<path d="M6.915 4.03c-1.968 0-3.683 1.28-4.871 3.113C.704 9.208 0 11.883 0 14.449c0 .706.07 1.369.21 1.973a6.624 6.624 0 0 0 .265.86 5.297 5.297 0 0 0 .371.761c.696 1.159 1.818 1.927 3.593 1.927 1.497 0 2.633-.671 3.965-2.444.76-1.012 1.144-1.626 2.663-4.32l.756-1.339.186-.325c.061.1.121.196.183.3l2.152 3.595c.724 1.21 1.665 2.556 2.47 3.314 1.046.987 1.992 1.22 3.06 1.22 1.075 0 1.876-.355 2.455-.843a3.743 3.743 0 0 0 .81-.973c.542-.939.861-2.127.861-3.745 0-2.72-.681-5.357-2.084-7.45-1.282-1.912-2.957-2.93-4.716-2.93-1.047 0-2.088.467-3.053 1.308-.652.57-1.257 1.29-1.82 2.05-.69-.875-1.335-1.547-1.958-2.056-1.182-.966-2.315-1.303-3.454-1.303zm10.16 2.053c1.147 0 2.188.758 2.992 1.999 1.132 1.748 1.647 4.195 1.647 6.4 0 1.548-.368 2.9-1.839 2.9-.58 0-1.027-.23-1.664-1.004-.496-.601-1.343-1.878-2.832-4.358l-.617-1.028a44.908 44.908 0 0 0-1.255-1.98c.07-.109.141-.224.211-.327 1.12-1.667 2.118-2.602 3.358-2.602zm-10.201.553c1.265 0 2.058.791 2.675 1.446.307.327.737.871 1.234 1.579l-1.02 1.566c-.757 1.163-1.882 3.017-2.837 4.338-1.191 1.649-1.81 1.817-2.486 1.817-.524 0-1.038-.237-1.383-.794-.263-.426-.464-1.13-.464-2.046 0-2.221.63-4.535 1.66-6.088.454-.687.964-1.226 1.533-1.533a2.264 2.264 0 0 1 1.088-.285z"/>`,
    minimax: `<path d="M11.43 3.92a.86.86 0 1 0-1.718 0v14.236a1.999 1.999 0 0 1-3.997 0V9.022a.86.86 0 1 0-1.718 0v3.87a1.999 1.999 0 0 1-3.997 0V11.49a.57.57 0 0 1 1.139 0v1.404a.86.86 0 0 0 1.719 0V9.022a1.999 1.999 0 0 1 3.997 0v9.134a.86.86 0 0 0 1.719 0V3.92a1.998 1.998 0 1 1 3.996 0v11.788a.57.57 0 1 1-1.139 0zm10.572 3.105a2 2 0 0 0-1.999 1.997v7.63a.86.86 0 0 1-1.718 0V3.923a1.999 1.999 0 0 0-3.997 0v16.16a.86.86 0 0 1-1.719 0V18.08a.57.57 0 1 0-1.138 0v2a1.998 1.998 0 0 0 3.996 0V3.92a.86.86 0 0 1 1.719 0v12.73a1.999 1.999 0 0 0 3.996 0V9.023a.86.86 0 1 1 1.72 0v6.686a.57.57 0 0 0 1.138 0V9.022a2 2 0 0 0-1.998-1.997"/>`,
    mistralai: `<path d="M17.143 3.429v3.428h-3.429v3.429h-3.428V6.857H6.857V3.43H3.43v13.714H0v3.428h10.286v-3.428H6.857v-3.429h3.429v3.429h3.429v-3.429h3.428v3.429h-3.428v3.428H24v-3.428h-3.43V3.429z"/>`,
    openai: `<path d="M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.0729zm-9.022 12.6081a4.4755 4.4755 0 0 1-2.8764-1.0408l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.872zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231l-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654l2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z"/>`,
    qwen: `<path d="M23.919 14.545 20.817 9.17l1.47-2.544a.56.56 0 0 0 0-.566l-1.633-2.83a.57.57 0 0 0-.49-.283h-6.207L12.487.402a.57.57 0 0 0-.49-.284H8.732a.56.56 0 0 0-.49.284L5.139 5.775h-2.94a.56.56 0 0 0-.49.284L.077 8.887a.56.56 0 0 0 0 .567L3.18 14.83l-1.47 2.545a.56.56 0 0 0 0 .566l1.634 2.83a.57.57 0 0 0 .49.283h6.205l1.47 2.545a.57.57 0 0 0 .49.284h3.266a.57.57 0 0 0 .49-.284l3.104-5.375h2.94a.57.57 0 0 0 .49-.283l1.634-2.828a.55.55 0 0 0-.004-.568M8.733.686l1.634 2.828-1.634 2.828H21.8L20.164 9.17H7.425L5.63 6.06Zm1.306 19.801-6.205-.002 1.634-2.83h3.265L2.201 6.344h3.267q3.182 5.517 6.367 11.032zm10.124-5.66L18.53 12l-6.532 11.315-1.634-2.83c2.129-3.673 4.25-7.351 6.373-11.028h3.592l3.102 5.374z"/>`
  };
  const PROVIDER_MARKS = {
    Meta: {
      icon: "meta",
      c: "#0467DF",
      d: "#2279E3"
    },
    MiniMax: {
      icon: "minimax",
      c: "#E73562",
      d: "#E73562"
    },
    "Mistral AI": {
      icon: "mistralai",
      c: "#FA520F",
      d: "#FA520F"
    },
    DeepSeek: {
      icon: "deepseek",
      c: "#5786FE",
      d: "#5786FE"
    },
    OpenAI: {
      icon: "openai",
      c: "#0D0D0D",
      d: "#F5F5F5"
    },
    Google: {
      icon: "google",
      c: "#4285F4",
      d: "#4285F4"
    },
    "Alibaba Cloud": {
      icon: "qwen",
      c: "#6950EF",
      d: "#7B65F1"
    }
  };
  const NEUTRAL = {
    icon: "generic",
    c: "#645D70",
    d: "#A9A1B6"
  };
  const markFor = p => PROVIDER_MARKS[p] || NEUTRAL;
  const CSS = `
  .sst { --sst-bd: rgba(0,0,0,.10); --sst-mu: rgba(0,0,0,.55); --sst-sf: rgba(0,0,0,.025); --sst-sf2: rgba(0,0,0,.05); }
  .dark .sst { --sst-bd: rgba(255,255,255,.14); --sst-mu: rgba(255,255,255,.60); --sst-sf: rgba(255,255,255,.04); --sst-sf2: rgba(255,255,255,.07); }

  /* ---- controls ---- */
  .sst-controls { display:flex; flex-wrap:wrap; gap:10px; align-items:center; margin:4px 0 14px; }
  .sst-search { flex:1 1 300px; min-width:200px; font:inherit; font-size:14px; padding:10px 14px; border:1px solid var(--sst-bd); border-radius:10px; background:transparent; color:inherit; }
  .sst-search:focus { outline:none; border-color:${BRAND}; box-shadow:0 0 0 3px rgba(151,79,199,.14); }
  .sst-chips { display:flex; flex-wrap:wrap; gap:7px; margin-bottom:10px; }
  .sst-chip { display:inline-flex; align-items:center; gap:7px; font:inherit; font-size:12.5px; padding:5px 12px 5px 6px; border:1px solid var(--sst-bd); border-radius:999px; background:transparent; color:inherit; cursor:pointer; transition:.12s; }
  .sst-chip-plain { padding-left:12px; }
  .sst-chip:hover { border-color:${BRAND}; }
  .sst-chip-on { border-color:${BRAND}; background:rgba(151,79,199,.12); color:${BRAND}; font-weight:600; }
  .sst-chip-n { opacity:.55; font-variant-numeric:tabular-nums; }
  .sst-count { font-size:13px; color:var(--sst-mu); margin:0 0 14px; }

  /* ---- overview grid ---- */
  .sst-grid { display:grid; grid-template-columns:repeat(auto-fill,minmax(260px,1fr)); gap:12px; }
  .sst-mini { display:flex; flex-direction:column; gap:9px; padding:14px; border:1px solid var(--sst-bd); border-radius:12px; background:var(--sst-sf); text-decoration:none!important; color:inherit!important; transition:.12s; }
  .sst-mini:hover { border-color:${BRAND}; background:var(--sst-sf2); transform:translateY(-1px); }
  .sst-mini-top { display:flex; justify-content:space-between; align-items:center; gap:8px; }
  .sst-prov { font-size:10.5px; font-weight:700; letter-spacing:.06em; text-transform:uppercase; color:var(--sst-mu); }
  .sst-mini-id { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:13px; font-weight:600; word-break:break-all; line-height:1.35; }
  .sst-mini-foot { display:flex; flex-wrap:wrap; gap:6px; align-items:center; margin-top:auto; }

  /* ---- provider mark ---- */
  .sst-mark {
    width:34px; height:34px; flex-shrink:0;
    display:inline-flex; align-items:center; justify-content:center;
    font-family:ui-monospace,SFMono-Regular,Menlo,monospace;
    font-size:12px; font-weight:700; letter-spacing:-.02em; line-height:1;
    border-radius:9px;
    color:var(--mk);
    background:color-mix(in srgb, var(--mk) 13%, transparent);
    border:1px solid color-mix(in srgb, var(--mk) 24%, transparent);
  }
  .dark .sst-mark {
    color:var(--mkd);
    background:color-mix(in srgb, var(--mkd) 15%, transparent);
    border-color:color-mix(in srgb, var(--mkd) 28%, transparent);
  }
  .sst-mark-sm { width:20px; height:20px; border-radius:6px; font-size:9px; }
  /* Inline brand marks. The SVG is emitted with fill="currentColor", so the
     glyph takes the colour set on the tile — which is the provider hue, with a
     separate value for dark mode. A mask was used before, against an external
     SVG; that broke under the production /docs base path. */
  .sst-mark-img { width:18px; height:18px; display:block; }
  .sst-mark-sm .sst-mark-img { width:11px; height:11px; }

  /* ---- badges ---- */

  .sst-stat { font-size:11.5px; color:var(--sst-mu); font-variant-numeric:tabular-nums; }
  .sst-tag { font-size:11px; padding:2px 8px; border-radius:999px; border:1px solid var(--sst-bd); color:var(--sst-mu); white-space:nowrap; }
  .sst-tag-mod { border-color:transparent; background:var(--sst-sf2); color:inherit; font-weight:600; }

  /* ---- detail card ---- */
  .sst-card { border:1px solid var(--sst-bd); border-radius:14px; overflow:hidden; margin:2px 0 8px; }
  .sst-head { display:flex; flex-wrap:wrap; justify-content:space-between; align-items:flex-start; gap:12px; padding:16px 18px; background:var(--sst-sf); border-bottom:1px solid var(--sst-bd); }
  .sst-head-l { min-width:0; display:flex; gap:12px; align-items:flex-start; }
  .sst-head-txt { min-width:0; }
  .sst-id { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:15px; font-weight:700; word-break:break-all; line-height:1.3; }
  .sst-use { font-size:13px; color:var(--sst-mu); margin:5px 0 0; }

  /* Collapsible card.

     The full catalogue runs to several screens of cards, which pushed the
     filter controls far off the top of the page. Each card is a <details>:
     the header — provider, model ID, and use — always shows, and the detail
     below opens on click. Native <details> keeps this keyboard
     accessible and lets Chrome's find-in-page open a closed card. */
  .sst-det > summary { cursor:pointer; list-style:none; }
  .sst-det > summary::-webkit-details-marker { display:none; }
  .sst-det > summary:hover { background:var(--sst-sf2); }
  .sst-det:not([open]) > summary { border-bottom:none; }
  .sst-det > summary:focus-visible { outline:2px solid ${BRAND}; outline-offset:-2px; }
  .sst-chev { flex-shrink:0; width:18px; height:18px; margin-top:3px; color:var(--sst-mu); transition:transform .15s ease; }
  .sst-det[open] .sst-chev { transform:rotate(180deg); }

  .sst-body { padding:16px 18px; display:flex; flex-direction:column; gap:16px; }
  .sst-row { display:flex; flex-wrap:wrap; gap:7px; align-items:center; }
  .sst-lbl { font-size:10.5px; font-weight:700; letter-spacing:.07em; text-transform:uppercase; color:var(--sst-mu); margin:0 0 7px; }

  /* endpoints */
  .sst-eps { display:grid; grid-template-columns:repeat(auto-fill,minmax(215px,1fr)); gap:8px; }
  .sst-ep { display:flex; align-items:center; gap:9px; padding:9px 11px; border:1px solid var(--sst-bd); border-radius:9px; background:var(--sst-sf); }
  .sst-ep-off { opacity:.5; }
  .sst-dot { width:7px; height:7px; border-radius:50%; flex-shrink:0; }
  .sst-dot-ok { background:#059669; }
  .sst-dot-lim { background:#d97706; }
  .sst-dot-no { background:currentColor; opacity:.3; }
  .sst-ep-t { min-width:0; }
  .sst-ep-n { font-size:12.5px; font-weight:600; line-height:1.3; }
  .sst-ep-p { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:10.5px; color:var(--sst-mu); word-break:break-all; }

  /* features */
  .sst-feats { display:flex; flex-wrap:wrap; gap:7px; }
  .sst-feat { display:inline-flex; align-items:center; gap:6px; font-size:12px; padding:5px 10px; border-radius:999px; border:1px solid var(--sst-bd); }
  .sst-feat-ok { border-color:rgba(5,150,105,.4); background:rgba(5,150,105,.08); }
  .sst-feat-lim { border-color:rgba(217,119,6,.4); background:rgba(217,119,6,.08); }

  /* bundles */
  .sst-bundles { display:flex; flex-direction:column; gap:7px; }
  .sst-bundle { display:flex; flex-wrap:wrap; align-items:center; gap:10px; padding:10px 12px; border:1px solid var(--sst-bd); border-radius:9px; }
  .sst-bundle-sug { border-color:${BRAND}; background:rgba(151,79,199,.06); }
  .sst-bn { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:12px; font-weight:600; word-break:break-all; flex:1 1 190px; }
  .sst-sug { font-size:10px; font-weight:700; letter-spacing:.04em; text-transform:uppercase; color:${BRAND}; background:rgba(151,79,199,.13); padding:3px 7px; border-radius:5px; white-space:nowrap; }

  /* PEF status, from the release's Pef custom resources. Colour-independent
     readers still get the word, so the hue is reinforcement rather than the
     only signal. Alpha backgrounds keep these legible on both themes. */
  .sst-pef { font-size:10px; font-weight:700; letter-spacing:.04em; text-transform:uppercase; padding:3px 7px; border-radius:5px; white-space:nowrap; }
  .sst-pef-stable { color:#15803d; background:rgba(21,128,61,.13); }
  .sst-pef-preview { color:#b45309; background:rgba(180,83,9,.14); }
  .sst-pef-deprecated { color:#b91c1c; background:rgba(185,28,28,.13); }
  :root[data-theme="dark"] .sst-pef-stable, html.dark .sst-pef-stable { color:#4ade80; background:rgba(74,222,128,.15); }
  :root[data-theme="dark"] .sst-pef-preview, html.dark .sst-pef-preview { color:#fbbf24; background:rgba(251,191,36,.16); }
  :root[data-theme="dark"] .sst-pef-deprecated, html.dark .sst-pef-deprecated { color:#f87171; background:rgba(248,113,113,.15); }

  /* deployment options — BYOC and speculative decoding */
  .sst-opts { display:flex; flex-direction:column; gap:6px; }
  .sst-opt { display:flex; flex-wrap:wrap; align-items:baseline; gap:8px; font-size:13px; }
  .sst-opt-k { color:var(--sst-mu); flex:0 0 auto; }
  .sst-opt-v { font-weight:600; }
  .sst-opt-v code { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:12px; font-weight:600; margin-right:6px; }

  .sst-link { font-size:13px; }
  .sst-empty { font-size:14px; color:var(--sst-mu); padding:20px 0; }

  @media (max-width:640px) {
    .sst-grid { grid-template-columns:1fr; }
    .sst-eps { grid-template-columns:1fr; }
  }
  `;
  const [q, setQ] = useState("");
  const [prov, setProv] = useState("all");
  const [mode, setMode] = useState("all");
  const [feat, setFeat] = useState("all");
  const FEATURES = [{
    key: "byoc",
    label: "Custom checkpoints"
  }, {
    key: "specdec",
    label: "Speculative decoding"
  }];
  const needle = q.trim().toLowerCase();
  const haystack = m => [m.id, m.provider, m.use, (m.modalities || []).join(" "), (m.capabilities || []).join(" "), m.byoc ? "byoc custom checkpoints" : "", m.specdec ? "speculative decoding draft model" : ""].join(" ").toLowerCase();
  const matchesFeat = m => feat === "all" || m[feat] === true;
  const afterSearch = models.filter(m => !needle || haystack(m).includes(needle));
  const byProv = afterSearch.filter(m => (mode === "all" || (m.modalities || []).includes(mode)) && matchesFeat(m));
  const byMode = afterSearch.filter(m => (prov === "all" || m.provider === prov) && matchesFeat(m));
  const byFeat = afterSearch.filter(m => (prov === "all" || m.provider === prov) && (mode === "all" || (m.modalities || []).includes(mode)));
  const shown = afterSearch.filter(m => (prov === "all" || m.provider === prov) && (mode === "all" || (m.modalities || []).includes(mode)) && matchesFeat(m));
  const providers = [...new Set(models.map(m => m.provider))].sort();
  const MODALITY_ORDER = ["Text", "Image", "Video", "Audio", "Embedding"];
  const present = new Set(models.flatMap(m => m.modalities || []));
  const modalities = [...MODALITY_ORDER.filter(x => present.has(x)), ...[...present].filter(x => !MODALITY_ORDER.includes(x)).sort()];
  const anchor = s => String(s).toLowerCase().replace(/[^a-z0-9_]+/g, "-").replace(/^-|-$/g, "");
  return <div className="sst">
      <style dangerouslySetInnerHTML={{
    __html: CSS
  }} />

      <div className="sst-controls">
        <input type="text" className="sst-search" value={q} onChange={e => setQ(e.target.value)} placeholder="Search by model ID, provider, or capability…" aria-label="Search models" />
      </div>

      <div className="sst-chips" role="group" aria-label="Filter by provider">
        <button type="button" className={prov === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setProv("all")} aria-pressed={prov === "all"}>
          All providers <span className="sst-chip-n">{byProv.length}</span>
        </button>
        {providers.map(pName => {
    const n = byProv.filter(m => m.provider === pName).length;
    return <button key={pName} type="button" className={prov === pName ? "sst-chip sst-chip-on" : "sst-chip"} onClick={() => setProv(pName)} aria-pressed={prov === pName}>
              <span className="sst-mark sst-mark-sm" style={{
      "--mk": markFor(pName).c,
      "--mkd": markFor(pName).d
    }} aria-hidden="true">
                <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
      __html: MARK_PATHS[markFor(pName).icon]
    }} />
              </span>
              {pName} <span className="sst-chip-n">{n}</span>
            </button>;
  })}
      </div>

      <div className="sst-chips" role="group" aria-label="Filter by modality">
        <button type="button" className={mode === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setMode("all")} aria-pressed={mode === "all"}>
          All modalities <span className="sst-chip-n">{byMode.length}</span>
        </button>
        {modalities.map(x => {
    const n = byMode.filter(m => (m.modalities || []).includes(x)).length;
    return <button key={x} type="button" className={mode === x ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setMode(x)} aria-pressed={mode === x}>
              {x} <span className="sst-chip-n">{n}</span>
            </button>;
  })}
      </div>

      {}
      {FEATURES.some(f => models.some(m => m[f.key] === true)) && <div className="sst-chips" role="group" aria-label="Filter by deployment option">
          <button type="button" className={feat === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setFeat("all")} aria-pressed={feat === "all"}>
            All options <span className="sst-chip-n">{byFeat.length}</span>
          </button>
          {FEATURES.filter(f => models.some(m => m[f.key] === true)).map(f => {
    const n = byFeat.filter(m => m[f.key] === true).length;
    return <button key={f.key} type="button" className={feat === f.key ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setFeat(f.key)} aria-pressed={feat === f.key}>
                {f.label} <span className="sst-chip-n">{n}</span>
              </button>;
  })}
        </div>}

      <p className="sst-count">
        {shown.length} {shown.length === 1 ? "model" : "models"} — select one to jump to its full
        configuration below
      </p>

      <div className="sst-grid">
        {shown.map(m => <a className="sst-mini" key={m.id} href={`#${anchor(m.id)}`}>
            <div className="sst-mini-top">
              <span className="sst-mark" style={{
    "--mk": markFor(m.provider).c,
    "--mkd": markFor(m.provider).d
  }} aria-hidden="true">
                <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
    __html: MARK_PATHS[markFor(m.provider).icon]
  }} />
              </span>
            </div>
            <span className="sst-prov">{m.provider}</span>
            <span className="sst-mini-id">{m.id}</span>
            <div className="sst-row">
              {(m.modalities || []).map(x => <span className="sst-tag sst-tag-mod" key={x}>
                  {x}
                </span>)}
            </div>
            <div className="sst-mini-foot">
              {m.context && m.context !== "\u2014" && <span className="sst-stat">{m.context} (max context length)</span>}
            </div>
          </a>)}
      </div>

      {shown.length === 0 && <p className="sst-empty">No models match those filters. Clear the search, or reset a filter above.</p>}
    </div>;
};

SambaStack supports a variety of models that can be deployed to both on-prem and hosted environments. Contact your system administrator to determine which models are available on your deployment. For definitions of the **Preview** and **Stable** designations, how models move between stages, the notice you receive before a model is retired, and the deprecation history, see [SambaStack model lifecycle](/docs/en/v2.2.3/sambastack/service-administration/model-deployment/model-lifecycle).

## All supported models

Search or filter to find a model, then select it to jump to its full configuration.

<StackModelExplorer
  models={[
{ id: "Meta-Llama-3.3-70B-Instruct", provider: "Meta", context: "128k", byoc: true, specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent, Text to SQL/Cipher" },
{ id: "Meta-Llama-3.1-8B-Instruct", provider: "Meta", context: "16k", byoc: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Gateway agent, Validation agent" },
{ id: "Meta-Llama-3.1-70B-Instruct", byoc: true, provider: "Meta", context: "32k", specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent" },
{ id: "Meta-Llama-3.1-405B-Instruct", byoc: true, provider: "Meta", context: "16k", specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent, Code generation" },
{ id: "Llama-4-Maverick-17B-128E-Instruct", provider: "Meta", context: "128k", modalities: ["Image","Text"], capabilities: ["Function calling","JSON mode"], use: "Image understanding, Task agent, Tool-calling agent" },
{ id: "MiniMax-M3", provider: "MiniMax", context: "1024k", modalities: ["Image","Text","Video"], capabilities: ["Image and video understanding","Optional thinking mode","Function calling"], use: "Coding agent, Long-context agentic tasks, Image and video understanding" },
{ id: "MiniMax-M2.7", byoc: true, provider: "MiniMax", context: "192k", modalities: ["Text"], capabilities: ["Function calling","Structured output"], use: "Coding agent" },
{ id: "MiniMax-M2.5", byoc: true, provider: "MiniMax", context: "160k", modalities: ["Text"], capabilities: ["Function calling","Structured output"], use: "Coding agent" },
{ id: "Mistral-Large-3-675B-Instruct-2512", byoc: true, provider: "Mistral AI", context: "32k", modalities: ["Text"], capabilities: ["Function calling"], use: "Multilingual instruction following, Task agent, Tool-calling agent" },
{ id: "DeepSeek-R1-0528", byoc: true, provider: "DeepSeek", context: "128k", modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Complex reasoning" },
{ id: "DeepSeek-R1-Distill-Llama-70B", provider: "DeepSeek", context: "128k", byoc: true, specdec: true, modalities: ["Text"], capabilities: ["Reasoning"], use: "Complex reasoning" },
{ id: "DeepSeek-V3-0324", byoc: true, provider: "DeepSeek", context: "128k", modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.1", byoc: true, provider: "DeepSeek", context: "128k", modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.2", byoc: true, provider: "DeepSeek", context: "128k", modalities: ["Text"], capabilities: ["Optional thinking mode","Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.1-Terminus", byoc: true, provider: "DeepSeek", context: "128k", modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Main/planner agent, Tool-calling agent" },
{ id: "gpt-oss-120b", byoc: true, provider: "OpenAI", context: "128k", modalities: ["Text"], capabilities: ["Reasoning","Function calling","JSON mode","Logit masking","logit_bias sampling parameter"], use: "Main/planner agent, Tool-calling agent" },
{ id: "gpt-oss-20b", byoc: true, provider: "OpenAI", context: "128k", modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent, Reasoning" },
{ id: "Whisper-Large-v3", provider: "OpenAI", context: "—", modalities: ["Audio"], capabilities: [], use: "Automatic speech recognition (ASR), Audio transcription" },
{ id: "gemma-3-27b-it", provider: "Google", context: "128k", modalities: ["Image","Text"], capabilities: ["Image understanding","JSON mode"], use: "Image understanding, Task agent" },
{ id: "gemma-3-12b-it", provider: "Google", context: "128k", modalities: ["Image","Text"], capabilities: ["Image understanding","JSON mode"], use: "Image understanding, Task agent" },
{ id: "gemma-4-31B-it", provider: "Google", context: "256k", modalities: ["Image","Text","Video"], capabilities: ["Image and video understanding","Optional thinking mode","Function calling","JSON mode"], use: "Image and video understanding, Task agent" },
{ id: "Qwen3-235B-A22B-Instruct-2507", byoc: true, provider: "Alibaba Cloud", context: "128k", modalities: ["Text"], capabilities: ["Reasoning"], use: "Agentic planner, Multilingual instruction following" },
{ id: "Qwen3-32B", provider: "Alibaba Cloud", context: "32k", modalities: ["Text"], capabilities: ["Reasoning"], use: "Task agent, Multilingual instruction following" },
{ id: "Qwen3-TTS-Talker", provider: "Alibaba Cloud", context: "4k", modalities: ["Audio","Text"], capabilities: [], use: "Text-to-speech" },
{ id: "Qwen3-TTS-Vocoder", provider: "Alibaba Cloud", context: "4k", modalities: ["Audio"], capabilities: [], use: "Audio synthesis" },
{ id: "Llama-3.3-Swallow-70B-Instruct-v0.4", byoc: true, provider: "Tokyotech-llm", context: "128k", specdec: true, modalities: ["Text"], capabilities: [], use: "Japanese instruction following, Task agent" },
{ id: "E5-Mistral-7B-Instruct", provider: "Other", context: "4k", modalities: ["Embedding"], capabilities: [], use: "Vector storage and retrieval (RAG)" },
]}
/>

## Custom checkpoints

All decoder-only text generation models support custom checkpoints, except the models in the following table. To list only the models that support them, select **Custom checkpoints** in the filter above. To convert a checkpoint, see [Checkpoint conversion tool](/docs/en/v2.2.3/sambastack/service-administration/model-deployment/checkpoint-conversion-tool).

| Model | Reason |
| - | - |
| `Llama-4-Maverick-17B-128E-Instruct` | Multimodal, not decoder-only text |
| `MiniMax-M3` | Multimodal, not decoder-only text |
| `gemma-3-27b-it` | Multimodal, not decoder-only text |
| `gemma-3-12b-it` | Multimodal, not decoder-only text |
| `gemma-4-31B-it` | Multimodal, not decoder-only text |
| `Qwen3-32B` | No conversion package available yet |
| `Whisper-Large-v3` | Not a text generation model |
| `Qwen3-TTS-Talker` | Not a text generation model |
| `Qwen3-TTS-Vocoder` | Not a text generation model |
| `E5-Mistral-7B-Instruct` | Not a text generation model |

## Finding models on your cluster

You can run the following command to discover available models in your cluster:

```shellscript theme={}
kubectl -n <namespace> get models
```

<Accordion title="kubectl get models does not return the name you send to the API">
  The command lists Kubernetes resource names (`metadata.name`). Inference requests must use the serving name (`spec.name`), which is what the **Model ID** headings below use.

  ```yaml theme={}
  apiVersion: sambanova.ai/v1alpha1
  kind: Model
  metadata:
    name: gemma-4-31b-it    # Kubernetes resource name – what kubectl get models returns
  spec:
    name: gemma-4-31B-it    # serving name – use this in API requests
  ```

  The two often differ only in case or punctuation, so the mismatch is easy to miss. In the example above they differ by a single character.

  To read the serving name of a deployed model, run:

  ```shellscript theme={}
  kubectl -n <namespace> describe model <resource-name>
  ```

  A model can also define `spec.aliases`, additional names that route to the same model. For the full `Model` field reference, see [Deploy custom checkpoints](/docs/en/v2.2.3/sambastack/service-administration/model-deployment/deploying-models-and-bundles/deploy-custom-checkpoints).

  Text to speech is one case where the names diverge: address `/v1/audio/speech` as `qwen3-tts`, not by the talker or vocoder resource names. See [Text to speech](/docs/en/features/speech).
</Accordion>

## Recommended model bundles

In SambaStack, a bundle is a packaged deployment that groups one or more models together with their associated configurations, such as batch size and sequence length. A single model can also be deployed on its own by pairing it with a model profile, without creating a bundle.

For example, deploying the `Meta‑Llama‑3.3‑70B` model with a batch size of 4 and a sequence length of 16K tokens constitutes a single configuration. A bundle, however, can contain multiple such configurations, either for the same model or for different models.

SambaNova’s RDU technology enables several models and configurations to be loaded simultaneously in a single deployment. This allows you to switch instantly between models and between batch‑/sequence‑size profiles as needed. In contrast to traditional GPU systems, where deployments are typically single‑model and static, SambaStack supports multi‑model, multi‑configuration bundles. This approach delivers higher efficiency, greater flexibility, and increased throughput while preserving low latency.

You can run the following command to discover available bundles in your cluster:

```shellscript theme={}
kubectl -n <namespace> get modelbundles
```

The table below lists the recommended bundles for the models currently available in SambaStack. Each entry pairs a model with its recommended deployment bundle.

<Note>
  If the bundles listed below do not satisfy your inference requirements, you can create [custom bundles](/docs/en/v2.2.3/sambastack/service-administration/model-deployment/deploying-models-and-bundles/create-a-custom-bundle) that combine any mix of models and configurations so long as they fit in DDR memory.
</Note>

### Deployment options

When deploying models in SambaStack, administrators can select from various context length and batch size combinations.

* Smaller batch sizes provide higher token throughput (tokens/second).
* Larger batch sizes provide better concurrency for multiple users.

SambaStack supports two deployment configurations – high-interactivity and high-throughput. See [High-throughput deployment](/docs/en/v2.2.3/sambastack/service-administration/model-deployment/deploying-models-and-bundles/deployment-configurations) for when to use each.

### Suggested bundles per model

For each model, this section lists the suggested bundle for typical use and any alternative bundles that trade off context length, batch size, or modality support. See [Bundle configurations](#bundle-configurations) below for the seq length/batch size details of each bundle.

| Provider | Model | Suggested bundle | Alternative bundles |
| - | - | - | - |
| Meta | `Meta-Llama-3.3-70B-Instruct` | `70b-3dot3-ss-4-8-16-32-64-128k` | `70b-3dot3-ss-full-whisper`, `us-agentic-rag-1-1`, `e5-mistral-70b-64k-128k` |
| Meta | `Meta-Llama-3.1-8B-Instruct` | `us-agentic-rag-1-1` | `qwen3-32b-llama405b-s-m` |
| Meta | `Meta-Llama-3.1-405B-Instruct` | `qwen3-32b-llama405b-s-m` | — |
| Meta | `Llama-4-Maverick-17B-128E-Instruct` | `llama-4-medium-8-16-32-64-128k` | `llama-4-medium-ss-16k-bs24`, `us-agentic-rag-1-1` |
| MiniMax | `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `minimax-m3-32-64-128-256-512k-1m` (up to 1M context) | `minimax-m3-512k` (up to 512K context), `minimax-m3-32k`, `minimax-m3-acb-prefix-caching` (adds [prompt caching](/docs/en/v2.2.3/sambastack/service-administration/performance/prompt-caching), text only) |
| MiniMax | `MiniMax-M2.7` | `dyt-minimax-m2p7-32-64-160-192k-pc` (adds [prompt caching](/docs/en/v2.2.3/sambastack/service-administration/performance/prompt-caching)) | `dyt-minimax-m2p7-32k-pc`, `dyt-minimax-m2p7-32-64-160-192k`, `dyt-minimax-m2p7-32-160-192k`, `dyt-minimax-m2p7-32k-v2` |
| MiniMax | `MiniMax-M2.5` | `dyt-minimax-m2p5-32-160k` | `dyt-minimax-m2p5-32k` |
| Mistral AI | `Mistral-Large-3-675B-Instruct-2512`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `mistral-large-3-fp8-8-16-32k` | `mistral-large-3-fp8-8k` |
| DeepSeek | `DeepSeek-R1-0528` | `deepseek-4in1-fp8-128k` (higher context length) | `deepseek-r1-v3-fp8-8k`, `deepseek-r1-v31-fp8-8k` |
| DeepSeek | `DeepSeek-V3-0324` | `deepseek-4in1-fp8-128k` (higher context length) | `deepseek-r1-v3-fp8-8k`, `deepseek-v3-v31-fp8-8k`, `deepseek-v3-v3termi-fp8-8k` |
| DeepSeek | `DeepSeek-V3.1` | `deepseek-4in1-fp8-128k` (higher context length) | `deepseek-r1-v31-fp8-8k`, `deepseek-v3-v31-fp8-8k` |
| DeepSeek | `DeepSeek-V3.1-Terminus` | `deepseek-4in1-fp8-128k` (higher context length) | `deepseek-v3-v3termi-fp8-8k` |
| OpenAI | `gpt-oss-120b` | `us-agentic-rag-1-1` | `cd-dyt-gpt-oss-120b-8-32-64-128k`, `gpt-gemma-whisper-mistral` |
| OpenAI | `gpt-oss-20b`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `dyt-gpt-oss-20b-32-64-128k` | — |
| OpenAI | `Whisper-Large-v3` | `qwen3-32b-whisper-e5-mistral` | `70b-3dot3-ss-full-whisper` (known issue: does not load within default startup time), `gpt-gemma-whisper-mistral` |
| Google | `gemma-3-27b-it`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `gemma3-27b-32-128k` | — |
| Google | `gemma-3-12b-it`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `gemma3-v3` | `gpt-gemma-whisper-mistral` |
| Google | `gemma-4-31B-it` | `gemma-4-31b-32-128-256k` (adds 256K context; text, image, and video support), `gemma-4-31b-32-128k` (up to 128K context; text, image, and video support, higher throughput). Both bundles support constrained decoding across all of their sequence lengths. Choose between them on context length and throughput, not capability. | `gemma-4-31b-mtp-cd-8-32-64-128k` (Preview; multi-token prediction, up to 128K context) |
| Alibaba Cloud | `Qwen3-235B-A22B-Instruct-2507` | `dyt-qwen3-235b-32-128k` | — |
| Alibaba Cloud | `Qwen3-32B` | `qwen3-32b-whisper-e5-mistral` | `qwen3-32b-llama405b-s-m` |
| Alibaba Cloud | `Qwen3-TTS-Talker`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `qwen3-tts-talker` | — |
| Alibaba Cloud | `Qwen3-TTS-Vocoder`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `qwen3-tts-vocoder` | — |
| Other | `E5-Mistral-7B-Instruct` | `us-agentic-rag-1-1` | `e5-mistral-70b-64k-128k`, `qwen3-32b-whisper-e5-mistral`, `gpt-gemma-whisper-mistral` |

### Bundle configurations

The table below lists the configuration details for each bundle.

| Model name | Bundle name | Bundle description | Bundle configuration |
| - | :- | :- | :- |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3.1` | deepseek-r1-v31-fp8-8k | <ul><li>Medium context length with low batch size</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details> |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3-0324` | deepseek-r1-v3-fp8-8k | <ul><li>Medium context length with low batch size</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details> |
| `DeepSeek-V3-0324` / <br />`DeepSeek-V3.1` | deepseek-v3-v31-fp8-8k | <ul><li>Medium context length with low batch size</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details> |
| `DeepSeek-V3-0324` / <br />`DeepSeek-V3.1-Terminus` | deepseek-v3-v3termi-fp8-8k | <ul><li>Medium context length with low batch size</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1-Terminus`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details> |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3-0324` / <br />`DeepSeek-V3.1` / <br />`DeepSeek-V3.1-Terminus` | deepseek-4in1-fp8-128k | <ul><li>Large context length with single batch size</li><li>Four DeepSeek models in one bundle</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3.1-Terminus`<ul><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details> |
| `E5-Mistral-7B-Instruct` / <br />`Meta-Llama-3.1-8B-Instruct` / <br />`Llama-4-Maverick-17B-128E-Instruct` / <br />`Meta-Llama-3.3-70B-Instruct` / <br />`gpt-oss-120b` | us-agentic-rag-1-1 | <ul><li>Small to medium context length with varied batch size</li><li>Speculative decoding supported for `Meta-Llama-3.3-70B`</li></ul> | <details><summary>View</summary><ul><li>`gpt-oss-120b`<ul><li>Seq Length: 32K, BS: 4</li><li>Seq Length: 64K, BS: 2</li><li>Seq Length: 128K, BS: 2</li></ul></li><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li></ul></li><li>`Meta-Llama-3.3-70B` (Target)/ `Meta-Llama-3.2-1B` (Draft)<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 1, 4, 8</li><li>Seq Length: 16K, BS: 1, 4</li><li>Seq Length: 32K, BS: 1, 4</li><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li><li>`Meta-Llama-3.1-8B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 16, 32</li><li>Seq Length: 8K, BS: 1, 4, 16, 32</li><li>Seq Length: 16K, BS: 1, 4, 8</li></ul></li><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li></ul></details> |
| `E5-Mistral-7B-Instruct` / <br />`Meta-Llama-3.3-70B` | e5-mistral-70b-64k-128k | <ul><li>Large context length with low batch size</li><li>Speculative decoding supported for `Meta-Llama-3.3-70B`</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li><li>`Meta-Llama-3.3-70B` (Target)/ `Meta-Llama-3.2-1B` (Draft)<ul><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details> |
| `gemma-3-27b-it` | gemma3-27b-32-128k | Homogeneous bundle containing `gemma-3-27b-it` configurations. | <details><summary>View</summary><ul><li>`gemma-3-27b-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `gemma-3-12b-it` | gemma3-v3 | <ul><li>Homogeneous bundle containing `gemma-3-12b-it` configurations.</li><li>Large context length with medium batch size</li></ul> | <details><summary>View</summary><ul><li>`gemma-3-12b-it`<ul><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `gemma-4-31B-it` | gemma-4-31b-32-128k | Homogeneous bundle with constrained decoding for `gemma-4-31B-it`. Supports text, image, and video input. | <details><summary>View</summary><ul><li>`gemma-4-31B-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `gemma-4-31B-it` | `gemma-4-31b-32-128-256k` | <ul><li>Homogeneous bundle with constrained decoding for `gemma-4-31B-it`. Adds context support up to 256K. Supports text, image, and video input at all sequence lengths.</li></ul> | <details><summary>View</summary><ul><li>`gemma-4-31B-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li><li>Seq Length: 256K, BS: 2</li></ul></li></ul></details> |
| `gemma-4-31B-it`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `gemma-4-31b-mtp-cd-8-32-64-128k` | <ul><li>Homogeneous bundle with multi-token prediction and constrained decoding for `gemma-4-31B-it`. Up to 128K context; text, image, and video input at all sequence lengths.</li><li>Multi-token prediction raises decode throughput; the model predicts several tokens per step and verifies them in the same pass.</li></ul> | <details><summary>View</summary><ul><li>`gemma-4-31B-it`<ul><li>Seq Length: 8K, BS: 2, 4, 6, 8</li><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `gpt-oss-120b` | `cd-dyt-gpt-oss-120b-8-32-64-128k` † | <ul><li>Homogeneous bundle with constrained decoding for `gpt-oss-120b`.</li><li>Covers 8K through 128K context.</li><li>Includes structured output (logit-masking) support.</li></ul> | <details><summary>View</summary><ul><li>`cd-dyt-gpt-oss-120b-8-32-64-128k`<ul><li>Seq Length: 8K, BS: 2, 4, 6, 8</li><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details> |
| `gpt-oss-20b` | `dyt-gpt-oss-20b-32-64-128k` † | Homogeneous bundle with constrained decoding for `gpt-oss-20b`. | <details><summary>View</summary><ul><li>`dyt-gpt-oss-20b-32-64-128k`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details> |
| `E5-Mistral-7B-Instruct` / <br />`Whisper-Large-v3` / <br />`gemma-3-12b-it` / <br />`gpt-oss-120b` | gpt-gemma-whisper-mistral | <ul><li>Small to medium context length combining embeddings, transcription, image understanding, and a tool-calling agent</li></ul> | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1</li></ul></li><li>`gemma-3-12b-it`<ul><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li><li>`gpt-oss-120b`<ul><li>Seq Length: 8K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `Llama-4-Maverick-17B-128E-Instruct` | llama-4-medium-8-16-32-64-128k | <ul><li>Homogeneous bundles containing `Llama-4-Maverick-17B-128E-Instruct` configurations.</li><li>Small to large context length with low batch</li></ul> | <details><summary>View</summary><ul><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1</li><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details> |
| `Llama-4-Maverick-17B-128E-Instruct` | llama-4-medium-ss-16k-bs24 | Homogeneous bundle containing `Llama-4-Maverick-17B-128E-Instruct` configurations; medium context length with higher batch size. | <details><summary>View</summary><ul><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 16K, BS: 2, 4</li></ul></li></ul></details> |
| `Meta-Llama-3.3-70B-Instruct` | 70b-3dot3-ss-4-8-16-32-64-128k | <ul><li>Medium to large context length with low batch size</li></ul> | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.3-70B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.2-1B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4; private: true</li><li>Seq Length: 128K, BS: 1; private: true</li></ul></li></ul></details> |
| `Meta-Llama-3.3-70B-Instruct` / <br />`Whisper-Large-v3` | 70b-3dot3-ss-full-whisper | <ul><li>Full context-length range with low to medium batch size, plus Whisper transcription</li><li>Speculative decoding supported for `Meta-Llama-3.3-70B-Instruct`</li></ul> | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.3-70B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1, 16, 32</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.2-1B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details> |
| `MiniMax-M2.5` | <ul><li>dyt-minimax-m2p5-32k</li><li>dyt-minimax-m2p5-32-160k</li></ul> | <ul><li>Homogeneous bundles containing `MiniMax-M2.5` configurations.</li><li>dyt-minimax-m2p5-32k is better for medium sequence lengths and high batching.</li><li>dyt-minimax-m2p5-32-160k is better for higher sequence lengths and low batching</li></ul> | <details><summary>View</summary><ul><li>dyt-minimax-m2p5-32k<ul><li>Seq Length: 4K-32K, BS: 2, 4, 6, 8</li></ul></li></ul><ul><li>dyt-minimax-m2p5-32-160k<ul><li>Seq Length: 32K, BS: 2</li><li>Seq Length: 160K, BS: 2</li></ul></li></ul></details> |
| `MiniMax-M2.7` | dyt-minimax-m2p7-32k-v2 | Homogeneous bundle containing `MiniMax-M2.7` configurations; medium context length with high batching. | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `MiniMax-M2.7` | dyt-minimax-m2p7-32-160-192k | Homogeneous bundle containing `MiniMax-M2.7` configurations; better for higher sequence lengths and low batching. | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details> |
| `MiniMax-M2.7` | dyt-minimax-m2p7-32-64-160-192k-pc | Homogeneous bundle with prompt caching for `MiniMax-M2.7`. | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details> |
| `MiniMax-M2.7` | dyt-minimax-m2p7-32-64-160-192k | <ul><li>Homogeneous bundle containing `MiniMax-M2.7` configurations; same context range as `dyt-minimax-m2p7-32-64-160-192k-pc` without prompt caching.</li></ul> | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details> |
| `MiniMax-M2.7` | dyt-minimax-m2p7-32k-pc | Homogeneous bundle with prompt caching for `MiniMax-M2.7`; medium context length with high batching. | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li></ul></li></ul></details> |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | minimax-m3-32k | Homogeneous bundle containing `MiniMax-M3` configurations; medium context length. Supports text, image, and video input. | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li></ul></li></ul></details> |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | minimax-m3-512k | <ul><li>Homogeneous bundle containing `MiniMax-M3` configurations.</li><li>Up to 512K context; text, image, and video input at all sequence lengths.</li></ul> | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1, 2, 4</li><li>Seq Length: 256K, BS: 1, 2, 4</li><li>Seq Length: 512K, BS: 1</li></ul></li></ul></details> |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | minimax-m3-32-64-128-256-512k-1m | <ul><li>Homogeneous bundle containing `MiniMax-M3` configurations.</li><li>Up to 1M context; text, image, and video input at all sequence lengths.</li></ul> | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1, 2, 4</li><li>Seq Length: 256K, BS: 1, 2, 4</li><li>Seq Length: 512K, BS: 1</li><li>Seq Length: 1M, BS: 1</li></ul></li></ul></details> |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | `minimax-m3-acb-prefix-caching` | <ul><li>Homogeneous bundle for `MiniMax-M3` with prompt caching, continuous batching, and constrained decoding. Up to 1M context.</li><li>Text input only. The other `MiniMax-M3` bundles serve text, image, and video.</li></ul> | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 1M, BS: 2</li></ul></li></ul></details> |
| `Mistral-Large-3-675B-Instruct-2512`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | mistral-large-3-fp8-8k | <ul><li>Homogeneous bundle containing `Mistral-Large-3-675B-Instruct-2512` configurations.</li><li>Preview model – text-only.</li></ul> | <details><summary>View</summary><ul><li>`Mistral-Large-3-675B-Instruct-2512`<ul><li>Seq Length: 8K, BS: 1</li></ul></li></ul></details> |
| `Mistral-Large-3-675B-Instruct-2512`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | mistral-large-3-fp8-8-16-32k | <ul><li>Homogeneous bundle containing `Mistral-Large-3-675B-Instruct-2512` configurations.</li><li>Wider context range than `mistral-large-3-fp8-8k`. Preview model – text-only.</li></ul> | <details><summary>View</summary><ul><li>`Mistral-Large-3-675B-Instruct-2512`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1, 2</li><li>Seq Length: 32K, BS: 1</li></ul></li></ul></details> |
| `Qwen3-235B-A22B-Instruct-2507` | dyt-qwen3-235b-32-128k | Homogeneous bundle containing `Qwen3-235B-A22B-Instruct-2507` configurations. | <details><summary>View</summary><ul><li>`Qwen3-235B-A22B-Instruct-2507`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details> |
| `Whisper-Large-v3` / <br />`Qwen3-32B` / <br />`E5-Mistral-7B-Instruct` | qwen3-32b-whisper-e5-mistral | <ul><li>Small to medium context length with varied batch size</li></ul> | <details><summary>View</summary><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li><li>`Qwen3-32B`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1, 2</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1, 16, 32</li></ul></li></ul></details> |
| `Qwen3-32B` / <br />`Meta-Llama-3.1-405B-Instruct` | qwen3-32b-llama405b-s-m | <ul><li>Small to medium context length</li><li>Speculative decoding supported for `Meta-Llama-3.1-405B-Instruct`</li></ul> | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.1-405B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 2, 4</li><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.1-8B-Instruct`<ul><li>Seq Length: 16K, BS: 1</li></ul></li><li>`Meta-Llama-3.2-3B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 2, 4</li><li>Seq Length: 8K, BS: 1</li></ul></li></ul><p><strong>Routable Models:</strong></p><ul><li>`Qwen3-32B`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1</li></ul></li></ul></details> |
| `Qwen3-TTS-Talker`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | qwen3-tts-talker | Homogeneous bundle containing `Qwen3-TTS-Talker` configurations. | <details><summary>View</summary><ul><li>`Qwen3-TTS-Talker`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16</li></ul></li></ul></details> |
| `Qwen3-TTS-Vocoder`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | qwen3-tts-vocoder | Homogeneous bundle containing `Qwen3-TTS-Vocoder` configurations. | <details><summary>View</summary><ul><li>`Qwen3-TTS-Vocoder`<ul><li>Codes length: 5, 8, 10, 16, BS: 16</li></ul></li></ul></details> |

† `cd-dyt-gpt-oss-120b-8-32-64-128k` supports sequence lengths from 8K–128K, and `dyt-gpt-oss-20b-32-64-128k` supports 32K–128K. If you require shorter context lengths (4K–16K) for either model, contact your SambaNova representative.

## Model details

Full configuration for each model, grouped by provider.

### Meta

#### Meta-Llama-3.3-70B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.3-70B-Instruct" use="Task agent, Tool-calling agent, Text to SQL/Cipher" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16, 32)","8K (1, 2, 4, 8, 16, 32)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1, 2, 4)","128K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct" />

#### Meta-Llama-3.1-8B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-8B-Instruct" use="Gateway agent, Validation agent" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16, 32, 64, 128)","8K (1, 2, 4, 8, 16, 32, 64)","16K (1, 2, 4, 8)"]} customCheckpoints={true} hf="https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct" />

#### Meta-Llama-3.1-70B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-70B-Instruct" use="Task agent, Tool-calling agent" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 2, 4, 8)","16K (1, 2, 4)","32K (1, 4)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.1-70B-Instruct" />

#### Meta-Llama-3.1-405B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-405B-Instruct" use="Task agent, Tool-calling agent, Code generation" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4)","8K (1)","16K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct" />

#### Llama-4-Maverick-17B-128E-Instruct

<StackModelCard provider="Meta" id="Llama-4-Maverick-17B-128E-Instruct" use="Image understanding, Task agent, Tool-calling agent" modalities={["Image","Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1)","16K (1, 2, 4)","32K (1)","64K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct" />

### MiniMax

#### MiniMax-M3

<StackModelCard provider="MiniMax" id="MiniMax-M3" use="Coding agent, Long-context agentic tasks, Image and video understanding" modalities={["Image","Text","Video"]} capabilities={["Image and video understanding","Optional thinking mode","Function calling"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"],["Responses","/v1/responses","Supported"]]} sequenceLengths={["8K-32K (1, 2, 4)","64K (1, 2, 4)","128K (1, 2, 4)","256K (1, 2, 4)","512K (1)","1M (1)"]} customCheckpoints={false} hf="https://huggingface.co/MiniMaxAI/MiniMax-M3" />

#### MiniMax-M2.7

<StackModelCard provider="MiniMax" id="MiniMax-M2.7" use="Coding agent" modalities={["Text"]} capabilities={["Function calling","Structured output"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","160K (2)","192K (2)"]} customCheckpoints={true} hf="https://huggingface.co/MiniMaxAI/MiniMax-M2.7" />

#### MiniMax-M2.5

<StackModelCard provider="MiniMax" id="MiniMax-M2.5" use="Coding agent" modalities={["Text"]} capabilities={["Function calling","Structured output"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","160K (2)"]} customCheckpoints={true} hf="https://huggingface.co/MiniMaxAI/MiniMax-M2.5" />

### Mistral AI

#### Mistral-Large-3-675B-Instruct-2512

<StackModelCard provider="Mistral AI" id="Mistral-Large-3-675B-Instruct-2512" use="Multilingual instruction following, Task agent, Tool-calling agent" modalities={["Text"]} capabilities={["Function calling"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1, 2)","32K (1)"]} customCheckpoints={true} hf="https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" />

### DeepSeek

#### DeepSeek-R1-0528

<StackModelCard provider="DeepSeek" id="DeepSeek-R1-0528" use="Complex reasoning" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-R1-0528" />

#### DeepSeek-R1-Distill-Llama-70B

<StackModelCard provider="DeepSeek" id="DeepSeek-R1-Distill-Llama-70B" use="Complex reasoning" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (2, 4, 8, 16, 32)","8K (2, 4, 8, 16, 32)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1, 2, 4)","128K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-70B" />

#### DeepSeek-V3-0324

<StackModelCard provider="DeepSeek" id="DeepSeek-V3-0324" use="Main/planner agent, Tool-calling agent" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3-0324" />

#### DeepSeek-V3.1

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.1" use="Main/planner agent, Tool-calling agent" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.1" />

#### DeepSeek-V3.2

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.2" use="Main/planner agent, Tool-calling agent" modalities={["Text"]} capabilities={["Optional thinking mode","Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.2" />

#### DeepSeek-V3.1-Terminus

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.1-Terminus" use="Main/planner agent, Tool-calling agent" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.1-Terminus" />

### OpenAI

#### gpt-oss-120b

<StackModelCard provider="OpenAI" id="gpt-oss-120b" use="Main/planner agent, Tool-calling agent" modalities={["Text"]} capabilities={["Reasoning","Function calling","JSON mode","Logit masking","logit_bias sampling parameter"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","128K (2)"]} customCheckpoints={true} hf="https://huggingface.co/openai/gpt-oss-120b" />

#### gpt-oss-20b

<StackModelCard provider="OpenAI" id="gpt-oss-20b" use="Main/planner agent, Tool-calling agent, Reasoning" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","128K (2)"]} customCheckpoints={true} hf="https://huggingface.co/openai/gpt-oss-20b" />

#### Whisper-Large-v3

<StackModelCard provider="OpenAI" id="Whisper-Large-v3" use="Automatic speech recognition (ASR), Audio transcription" modalities={["Audio"]} endpoints={[["Translation","/v1/audio/translations","Supported"],["Transcription","/v1/audio/transcriptions","Supported"]]} sequenceLengths={["448 (1, 16, 32)"]} customCheckpoints={false} hf="https://huggingface.co/openai/whisper-large-v3" />

### Google

#### gemma-3-27b-it

<StackModelCard provider="Google" id="gemma-3-27b-it" use="Image understanding, Task agent" modalities={["Image","Text"]} capabilities={["Image understanding","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K-128K (2, 4, 6, 8)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-3-27b-it" />

#### gemma-3-12b-it

<StackModelCard provider="Google" id="gemma-3-12b-it" use="Image understanding, Task agent" modalities={["Image","Text"]} capabilities={["Image understanding","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["128K (2, 4, 6, 8)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-3-12b-it" />

#### gemma-4-31B-it

<StackModelCard provider="Google" id="gemma-4-31B-it" use="Image and video understanding, Task agent" modalities={["Image","Text","Video"]} capabilities={["Image and video understanding","Optional thinking mode","Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (2, 4, 6, 8)","32K-128K (2, 4, 6, 8)","256K (2)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-4-31B-it" />

### Alibaba Cloud

#### Qwen3-235B-A22B-Instruct-2507

<StackModelCard provider="Alibaba Cloud" id="Qwen3-235B-A22B-Instruct-2507" use="Agentic planner, Multilingual instruction following" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["32K (2, 4, 6, 8)","128K (2)"]} customCheckpoints={true} hf="https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507" />

#### Qwen3-32B

<StackModelCard provider="Alibaba Cloud" id="Qwen3-32B" use="Task agent, Multilingual instruction following" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1, 2)"]} customCheckpoints={false} hf="https://huggingface.co/Qwen/Qwen3-32B" />

#### Qwen3-TTS-Talker

<StackModelCard provider="Alibaba Cloud" id="Qwen3-TTS-Talker" use="Text-to-speech" modalities={["Audio","Text"]} endpoints={[["Speech","","Supported"]]} sequenceLengths={["4K (2, 4, 8, 16)"]} customCheckpoints={false} hf="https://huggingface.co/collections/Qwen/qwen3-tts" />

#### Qwen3-TTS-Vocoder

<StackModelCard provider="Alibaba Cloud" id="Qwen3-TTS-Vocoder" use="Audio synthesis" modalities={["Audio"]} endpoints={[["Speech","","Supported"]]} sequenceLengths={["Codes length 5, 8, 10, 16 (16)","4K (1)"]} customCheckpoints={false} hf="https://huggingface.co/collections/Qwen/qwen3-tts" />

### Tokyotech-llm

#### Llama-3.3-Swallow-70B-Instruct-v0.4

<StackModelCard provider="Tokyotech-llm" id="Llama-3.3-Swallow-70B-Instruct-v0.4" use="Japanese instruction following, Task agent" modalities={["Text"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16)","8K (1, 2, 4, 8, 16)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1)","128K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/tokyotech-llm/Llama-3.3-Swallow-70B-Instruct-v0.4" />

### Other

#### E5-Mistral-7B-Instruct

<StackModelCard provider="Other" id="E5-Mistral-7B-Instruct" use="Vector storage and retrieval (RAG)" modalities={["Embedding"]} endpoints={[["Embeddings","/v1/embeddings","Supported"]]} sequenceLengths={["4K (1, 4, 8, 16, 32)"]} customCheckpoints={false} hf="https://huggingface.co/intfloat/e5-mistral-7b-instruct" />
