> ## Documentation Index
> Fetch the complete documentation index at: https://docs.sambanova.ai/docs/llms.txt
> Use this file to discover all available pages before exploring further.

# Supported models and bundles

export const StackModelCard = ({provider = "", id = "", use = "", context = "", modalities = [], capabilities = [], endpoints = [], features = [], bundles = [], hf = "", mark = "", customCheckpoints, draftModels = [], sequenceLengths = [], specDecoding = false}) => {
  const MARK_PATHS = {
    deepseek: `<path d="M23.748 4.651c-.254-.124-.364.113-.512.233-.051.04-.094.09-.137.137-.372.397-.806.657-1.373.626-.829-.046-1.537.214-2.163.848-.133-.782-.575-1.248-1.247-1.548-.352-.155-.708-.311-.955-.65-.172-.24-.219-.509-.305-.774-.055-.16-.11-.323-.293-.35-.2-.031-.278.136-.356.276-.313.572-.434 1.202-.422 1.84.027 1.436.633 2.58 1.838 3.393.137.094.172.187.129.323-.082.28-.18.553-.266.833-.055.179-.137.218-.328.14a5.5 5.5 0 0 1-1.737-1.179c-.857-.828-1.631-1.743-2.597-2.46a12 12 0 0 0-.689-.47c-.985-.957.13-1.743.387-1.836.27-.098.094-.433-.778-.428-.872.003-1.67.295-2.687.685a3 3 0 0 1-.465.136 9.6 9.6 0 0 0-2.883-.101c-1.885.21-3.39 1.1-4.497 2.622C.082 8.776-.231 10.854.152 13.02c.403 2.284 1.568 4.175 3.36 5.653 1.857 1.533 3.997 2.284 6.438 2.14 1.482-.085 3.132-.284 4.994-1.86.47.234.962.328 1.78.398.629.058 1.235-.031 1.705-.129.735-.155.684-.836.418-.961-2.155-1.004-1.682-.595-2.112-.926 1.095-1.295 2.768-3.598 3.284-6.733.05-.346.115-.834.108-1.114-.004-.171.035-.238.23-.257a4.2 4.2 0 0 0 1.545-.475c1.397-.763 1.96-2.016 2.093-3.517.02-.23-.004-.467-.247-.588M11.58 18.168c-2.088-1.642-3.101-2.183-3.52-2.16-.39.024-.32.472-.234.763.09.288.207.487.371.74.114.167.192.416-.113.603-.673.416-1.842-.14-1.897-.168-1.361-.801-2.5-1.86-3.301-3.306-.775-1.393-1.225-2.888-1.299-4.482-.02-.385.094-.522.477-.592a4.7 4.7 0 0 1 1.53-.038c2.131.311 3.946 1.264 5.467 2.774.868.86 1.525 1.887 2.202 2.89.72 1.066 1.494 2.082 2.48 2.915.348.291.626.513.892.677-.802.09-2.14.109-3.055-.615zm1.001-6.44a.306.306 0 0 1 .415-.287.3.3 0 0 1 .113.074.3.3 0 0 1 .086.214c0 .17-.136.307-.308.307a.303.303 0 0 1-.306-.307m3.11 1.596c-.2.081-.4.151-.591.16a1.25 1.25 0 0 1-.798-.254c-.274-.23-.47-.358-.551-.758a1.7 1.7 0 0 1 .015-.588c.07-.327-.007-.537-.238-.727-.188-.156-.426-.199-.689-.199a.6.6 0 0 1-.254-.078.253.253 0 0 1-.114-.358 1 1 0 0 1 .192-.21c.356-.202.767-.136 1.146.016.352.144.618.408 1.001.782.392.451.462.576.685.915.176.264.336.536.446.848.066.194-.02.353-.25.45"/>`,
    generic: `<path fill-rule="evenodd" clip-rule="evenodd" d="M12 1.5 21.5 7v10L12 22.5 2.5 17V7L12 1.5Zm0 2.31L4.5 8.16v7.68L12 20.19l7.5-4.35V8.16L12 3.81Z"/>`,
    google: `<path d="M12.48 10.92v3.28h7.84c-.24 1.84-.853 3.187-1.787 4.133-1.147 1.147-2.933 2.4-6.053 2.4-4.827 0-8.6-3.893-8.6-8.72s3.773-8.72 8.6-8.72c2.6 0 4.507 1.027 5.907 2.347l2.307-2.307C18.747 1.44 16.133 0 12.48 0 5.867 0 .307 5.387.307 12s5.56 12 12.173 12c3.573 0 6.267-1.173 8.373-3.36 2.16-2.16 2.84-5.213 2.84-7.667 0-.76-.053-1.467-.173-2.053H12.48z"/>`,
    meta: `<path d="M6.915 4.03c-1.968 0-3.683 1.28-4.871 3.113C.704 9.208 0 11.883 0 14.449c0 .706.07 1.369.21 1.973a6.624 6.624 0 0 0 .265.86 5.297 5.297 0 0 0 .371.761c.696 1.159 1.818 1.927 3.593 1.927 1.497 0 2.633-.671 3.965-2.444.76-1.012 1.144-1.626 2.663-4.32l.756-1.339.186-.325c.061.1.121.196.183.3l2.152 3.595c.724 1.21 1.665 2.556 2.47 3.314 1.046.987 1.992 1.22 3.06 1.22 1.075 0 1.876-.355 2.455-.843a3.743 3.743 0 0 0 .81-.973c.542-.939.861-2.127.861-3.745 0-2.72-.681-5.357-2.084-7.45-1.282-1.912-2.957-2.93-4.716-2.93-1.047 0-2.088.467-3.053 1.308-.652.57-1.257 1.29-1.82 2.05-.69-.875-1.335-1.547-1.958-2.056-1.182-.966-2.315-1.303-3.454-1.303zm10.16 2.053c1.147 0 2.188.758 2.992 1.999 1.132 1.748 1.647 4.195 1.647 6.4 0 1.548-.368 2.9-1.839 2.9-.58 0-1.027-.23-1.664-1.004-.496-.601-1.343-1.878-2.832-4.358l-.617-1.028a44.908 44.908 0 0 0-1.255-1.98c.07-.109.141-.224.211-.327 1.12-1.667 2.118-2.602 3.358-2.602zm-10.201.553c1.265 0 2.058.791 2.675 1.446.307.327.737.871 1.234 1.579l-1.02 1.566c-.757 1.163-1.882 3.017-2.837 4.338-1.191 1.649-1.81 1.817-2.486 1.817-.524 0-1.038-.237-1.383-.794-.263-.426-.464-1.13-.464-2.046 0-2.221.63-4.535 1.66-6.088.454-.687.964-1.226 1.533-1.533a2.264 2.264 0 0 1 1.088-.285z"/>`,
    minimax: `<path d="M11.43 3.92a.86.86 0 1 0-1.718 0v14.236a1.999 1.999 0 0 1-3.997 0V9.022a.86.86 0 1 0-1.718 0v3.87a1.999 1.999 0 0 1-3.997 0V11.49a.57.57 0 0 1 1.139 0v1.404a.86.86 0 0 0 1.719 0V9.022a1.999 1.999 0 0 1 3.997 0v9.134a.86.86 0 0 0 1.719 0V3.92a1.998 1.998 0 1 1 3.996 0v11.788a.57.57 0 1 1-1.139 0zm10.572 3.105a2 2 0 0 0-1.999 1.997v7.63a.86.86 0 0 1-1.718 0V3.923a1.999 1.999 0 0 0-3.997 0v16.16a.86.86 0 0 1-1.719 0V18.08a.57.57 0 1 0-1.138 0v2a1.998 1.998 0 0 0 3.996 0V3.92a.86.86 0 0 1 1.719 0v12.73a1.999 1.999 0 0 0 3.996 0V9.023a.86.86 0 1 1 1.72 0v6.686a.57.57 0 0 0 1.138 0V9.022a2 2 0 0 0-1.998-1.997"/>`,
    mistralai: `<path d="M17.143 3.429v3.428h-3.429v3.429h-3.428V6.857H6.857V3.43H3.43v13.714H0v3.428h10.286v-3.428H6.857v-3.429h3.429v3.429h3.429v-3.429h3.428v3.429h-3.428v3.428H24v-3.428h-3.43V3.429z"/>`,
    openai: `<path d="M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.0729zm-9.022 12.6081a4.4755 4.4755 0 0 1-2.8764-1.0408l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.872zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231l-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654l2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z"/>`,
    qwen: `<path d="M23.919 14.545 20.817 9.17l1.47-2.544a.56.56 0 0 0 0-.566l-1.633-2.83a.57.57 0 0 0-.49-.283h-6.207L12.487.402a.57.57 0 0 0-.49-.284H8.732a.56.56 0 0 0-.49.284L5.139 5.775h-2.94a.56.56 0 0 0-.49.284L.077 8.887a.56.56 0 0 0 0 .567L3.18 14.83l-1.47 2.545a.56.56 0 0 0 0 .566l1.634 2.83a.57.57 0 0 0 .49.283h6.205l1.47 2.545a.57.57 0 0 0 .49.284h3.266a.57.57 0 0 0 .49-.284l3.104-5.375h2.94a.57.57 0 0 0 .49-.283l1.634-2.828a.55.55 0 0 0-.004-.568M8.733.686l1.634 2.828-1.634 2.828H21.8L20.164 9.17H7.425L5.63 6.06Zm1.306 19.801-6.205-.002 1.634-2.83h3.265L2.201 6.344h3.267q3.182 5.517 6.367 11.032zm10.124-5.66L18.53 12l-6.532 11.315-1.634-2.83c2.129-3.673 4.25-7.351 6.373-11.028h3.592l3.102 5.374z"/>`
  };
  const PROVIDER_MARKS = {
    Meta: {
      icon: "meta",
      c: "#0467DF",
      d: "#2279E3"
    },
    MiniMax: {
      icon: "minimax",
      c: "#E73562",
      d: "#E73562"
    },
    "Mistral AI": {
      icon: "mistralai",
      c: "#FA520F",
      d: "#FA520F"
    },
    DeepSeek: {
      icon: "deepseek",
      c: "#5786FE",
      d: "#5786FE"
    },
    OpenAI: {
      icon: "openai",
      c: "#0D0D0D",
      d: "#F5F5F5"
    },
    Google: {
      icon: "google",
      c: "#4285F4",
      d: "#4285F4"
    },
    "Alibaba Cloud": {
      icon: "qwen",
      c: "#6950EF",
      d: "#7B65F1"
    }
  };
  const NEUTRAL = {
    icon: "generic",
    c: "#645D70",
    d: "#A9A1B6"
  };
  const mk = mark ? {
    ...NEUTRAL,
    icon: mark
  } : PROVIDER_MARKS[provider] || NEUTRAL;
  const dotFor = v => v === "Supported" ? "sst-dot sst-dot-ok" : v === "Limited" ? "sst-dot sst-dot-lim" : "sst-dot sst-dot-no";
  const featClass = v => v === "Supported" ? "sst-feat sst-feat-ok" : v === "Limited" ? "sst-feat sst-feat-lim" : "sst-feat";
  return <div className="sst">
      <div className="sst-card">
        <div className="sst-head">
          <div className="sst-head-l">
            <span className="sst-mark" style={{
    "--mk": mk.c,
    "--mkd": mk.d
  }} aria-hidden="true">
              {MARK_PATHS[mk.icon] ? <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
    __html: MARK_PATHS[mk.icon]
  }} /> : mk.m}
            </span>
            <div className="sst-head-txt">
              <div className="sst-prov">{provider}</div>
              <div className="sst-id">{id}</div>
              {use && <p className="sst-use">{use}</p>}
            </div>
          </div>
          <div className="sst-head-r">
            {context && <>
                <span className="sst-ctx">{context}</span>
                <span className="sst-ctx-l">max context</span>
              </>}
          </div>
        </div>

        <div className="sst-body">
          <div>
            <p className="sst-lbl">Modalities</p>
            <div className="sst-row">
              {modalities.length ? modalities.map(x => <span className="sst-tag sst-tag-mod" key={x}>
                    {x}
                  </span>) : <span className="sst-tag">—</span>}
            </div>
          </div>

          {capabilities.length > 0 && <div>
              <p className="sst-lbl">Capabilities</p>
              <div className="sst-row">
                {capabilities.map(x => <span className="sst-tag" key={x}>
                    {x}
                  </span>)}
              </div>
            </div>}

          {endpoints.length > 0 && <div>
              <p className="sst-lbl">Endpoints</p>
              <div className="sst-eps">
                {endpoints.map(([name, path, verdict]) => <div className={verdict ? "sst-ep" : "sst-ep sst-ep-off"} key={name}>
                    <span className={dotFor(verdict)} />
                    <div className="sst-ep-t">
                      <div className="sst-ep-n">{name}</div>
                      <div className="sst-ep-p">{path}</div>
                    </div>
                  </div>)}
              </div>
            </div>}

          {features.length > 0 && <div>
              <p className="sst-lbl">Features</p>
              <div className="sst-feats">
                {features.map(([name, verdict]) => <span className={featClass(verdict)} key={name}>
                    <span className={dotFor(verdict)} />
                    {name}
                    {verdict === "Limited" && " (limited)"}
                  </span>)}
              </div>
            </div>}

          {bundles.length > 0 && <div>
              <p className="sst-lbl">Bundle configurations</p>
              <div className="sst-bundles">
                {bundles.map(([name, ctx, isSuggested, status]) => <div className={isSuggested ? "sst-bundle sst-bundle-sug" : "sst-bundle"} key={name}>
                    <span className="sst-bn">{name}</span>
                    <span className="sst-stat">{ctx}</span>
                    {status && <span className={"sst-pef sst-pef-" + status} title={"PEF status: " + status}>
                        {status}
                      </span>}
                    {isSuggested && <span className="sst-sug">Suggested</span>}
                  </div>)}
              </div>
            </div>}

          {sequenceLengths.length > 0 && <div>
              <p className="sst-lbl">Context length (batch size)</p>
              <div className="sst-bundles">
                {sequenceLengths.map(sl => <div className="sst-bundle" key={sl}>
                    <span className="sst-bn">{sl}</span>
                  </div>)}
              </div>
            </div>}

          {(typeof customCheckpoints === "boolean" || draftModels.length > 0 || specDecoding) && <div>
              <p className="sst-lbl">Deployment options</p>
              <div className="sst-opts">
                {typeof customCheckpoints === "boolean" && <div className="sst-opt">
                    <span className="sst-opt-k">Custom checkpoints (BYOC)</span>
                    <span className="sst-opt-v">{customCheckpoints ? "Supported" : "Not supported"}</span>
                  </div>}
                {draftModels.length > 0 ? <div className="sst-opt">
                    <span className="sst-opt-k">Speculative decoding</span>
                    <span className="sst-opt-v">
                      Draft {draftModels.length > 1 ? "models" : "model"}:{" "}
                      {draftModels.map(d => <code key={d}>{d}</code>)}
                    </span>
                  </div> : specDecoding ? <div className="sst-opt">
                    <span className="sst-opt-k">Speculative decoding</span>
                    <span className="sst-opt-v">Supported</span>
                  </div> : null}
              </div>
            </div>}

          {hf && <a className="sst-link" href={hf} target="_blank" rel="noreferrer">
              View model card on Hugging Face ↗
            </a>}
        </div>
      </div>
    </div>;
};

export const StackModelExplorer = ({models = []}) => {
  const BRAND = "#974fc7";
  const MARK_PATHS = {
    deepseek: `<path d="M23.748 4.651c-.254-.124-.364.113-.512.233-.051.04-.094.09-.137.137-.372.397-.806.657-1.373.626-.829-.046-1.537.214-2.163.848-.133-.782-.575-1.248-1.247-1.548-.352-.155-.708-.311-.955-.65-.172-.24-.219-.509-.305-.774-.055-.16-.11-.323-.293-.35-.2-.031-.278.136-.356.276-.313.572-.434 1.202-.422 1.84.027 1.436.633 2.58 1.838 3.393.137.094.172.187.129.323-.082.28-.18.553-.266.833-.055.179-.137.218-.328.14a5.5 5.5 0 0 1-1.737-1.179c-.857-.828-1.631-1.743-2.597-2.46a12 12 0 0 0-.689-.47c-.985-.957.13-1.743.387-1.836.27-.098.094-.433-.778-.428-.872.003-1.67.295-2.687.685a3 3 0 0 1-.465.136 9.6 9.6 0 0 0-2.883-.101c-1.885.21-3.39 1.1-4.497 2.622C.082 8.776-.231 10.854.152 13.02c.403 2.284 1.568 4.175 3.36 5.653 1.857 1.533 3.997 2.284 6.438 2.14 1.482-.085 3.132-.284 4.994-1.86.47.234.962.328 1.78.398.629.058 1.235-.031 1.705-.129.735-.155.684-.836.418-.961-2.155-1.004-1.682-.595-2.112-.926 1.095-1.295 2.768-3.598 3.284-6.733.05-.346.115-.834.108-1.114-.004-.171.035-.238.23-.257a4.2 4.2 0 0 0 1.545-.475c1.397-.763 1.96-2.016 2.093-3.517.02-.23-.004-.467-.247-.588M11.58 18.168c-2.088-1.642-3.101-2.183-3.52-2.16-.39.024-.32.472-.234.763.09.288.207.487.371.74.114.167.192.416-.113.603-.673.416-1.842-.14-1.897-.168-1.361-.801-2.5-1.86-3.301-3.306-.775-1.393-1.225-2.888-1.299-4.482-.02-.385.094-.522.477-.592a4.7 4.7 0 0 1 1.53-.038c2.131.311 3.946 1.264 5.467 2.774.868.86 1.525 1.887 2.202 2.89.72 1.066 1.494 2.082 2.48 2.915.348.291.626.513.892.677-.802.09-2.14.109-3.055-.615zm1.001-6.44a.306.306 0 0 1 .415-.287.3.3 0 0 1 .113.074.3.3 0 0 1 .086.214c0 .17-.136.307-.308.307a.303.303 0 0 1-.306-.307m3.11 1.596c-.2.081-.4.151-.591.16a1.25 1.25 0 0 1-.798-.254c-.274-.23-.47-.358-.551-.758a1.7 1.7 0 0 1 .015-.588c.07-.327-.007-.537-.238-.727-.188-.156-.426-.199-.689-.199a.6.6 0 0 1-.254-.078.253.253 0 0 1-.114-.358 1 1 0 0 1 .192-.21c.356-.202.767-.136 1.146.016.352.144.618.408 1.001.782.392.451.462.576.685.915.176.264.336.536.446.848.066.194-.02.353-.25.45"/>`,
    generic: `<path fill-rule="evenodd" clip-rule="evenodd" d="M12 1.5 21.5 7v10L12 22.5 2.5 17V7L12 1.5Zm0 2.31L4.5 8.16v7.68L12 20.19l7.5-4.35V8.16L12 3.81Z"/>`,
    google: `<path d="M12.48 10.92v3.28h7.84c-.24 1.84-.853 3.187-1.787 4.133-1.147 1.147-2.933 2.4-6.053 2.4-4.827 0-8.6-3.893-8.6-8.72s3.773-8.72 8.6-8.72c2.6 0 4.507 1.027 5.907 2.347l2.307-2.307C18.747 1.44 16.133 0 12.48 0 5.867 0 .307 5.387.307 12s5.56 12 12.173 12c3.573 0 6.267-1.173 8.373-3.36 2.16-2.16 2.84-5.213 2.84-7.667 0-.76-.053-1.467-.173-2.053H12.48z"/>`,
    meta: `<path d="M6.915 4.03c-1.968 0-3.683 1.28-4.871 3.113C.704 9.208 0 11.883 0 14.449c0 .706.07 1.369.21 1.973a6.624 6.624 0 0 0 .265.86 5.297 5.297 0 0 0 .371.761c.696 1.159 1.818 1.927 3.593 1.927 1.497 0 2.633-.671 3.965-2.444.76-1.012 1.144-1.626 2.663-4.32l.756-1.339.186-.325c.061.1.121.196.183.3l2.152 3.595c.724 1.21 1.665 2.556 2.47 3.314 1.046.987 1.992 1.22 3.06 1.22 1.075 0 1.876-.355 2.455-.843a3.743 3.743 0 0 0 .81-.973c.542-.939.861-2.127.861-3.745 0-2.72-.681-5.357-2.084-7.45-1.282-1.912-2.957-2.93-4.716-2.93-1.047 0-2.088.467-3.053 1.308-.652.57-1.257 1.29-1.82 2.05-.69-.875-1.335-1.547-1.958-2.056-1.182-.966-2.315-1.303-3.454-1.303zm10.16 2.053c1.147 0 2.188.758 2.992 1.999 1.132 1.748 1.647 4.195 1.647 6.4 0 1.548-.368 2.9-1.839 2.9-.58 0-1.027-.23-1.664-1.004-.496-.601-1.343-1.878-2.832-4.358l-.617-1.028a44.908 44.908 0 0 0-1.255-1.98c.07-.109.141-.224.211-.327 1.12-1.667 2.118-2.602 3.358-2.602zm-10.201.553c1.265 0 2.058.791 2.675 1.446.307.327.737.871 1.234 1.579l-1.02 1.566c-.757 1.163-1.882 3.017-2.837 4.338-1.191 1.649-1.81 1.817-2.486 1.817-.524 0-1.038-.237-1.383-.794-.263-.426-.464-1.13-.464-2.046 0-2.221.63-4.535 1.66-6.088.454-.687.964-1.226 1.533-1.533a2.264 2.264 0 0 1 1.088-.285z"/>`,
    minimax: `<path d="M11.43 3.92a.86.86 0 1 0-1.718 0v14.236a1.999 1.999 0 0 1-3.997 0V9.022a.86.86 0 1 0-1.718 0v3.87a1.999 1.999 0 0 1-3.997 0V11.49a.57.57 0 0 1 1.139 0v1.404a.86.86 0 0 0 1.719 0V9.022a1.999 1.999 0 0 1 3.997 0v9.134a.86.86 0 0 0 1.719 0V3.92a1.998 1.998 0 1 1 3.996 0v11.788a.57.57 0 1 1-1.139 0zm10.572 3.105a2 2 0 0 0-1.999 1.997v7.63a.86.86 0 0 1-1.718 0V3.923a1.999 1.999 0 0 0-3.997 0v16.16a.86.86 0 0 1-1.719 0V18.08a.57.57 0 1 0-1.138 0v2a1.998 1.998 0 0 0 3.996 0V3.92a.86.86 0 0 1 1.719 0v12.73a1.999 1.999 0 0 0 3.996 0V9.023a.86.86 0 1 1 1.72 0v6.686a.57.57 0 0 0 1.138 0V9.022a2 2 0 0 0-1.998-1.997"/>`,
    mistralai: `<path d="M17.143 3.429v3.428h-3.429v3.429h-3.428V6.857H6.857V3.43H3.43v13.714H0v3.428h10.286v-3.428H6.857v-3.429h3.429v3.429h3.429v-3.429h3.428v3.429h-3.428v3.428H24v-3.428h-3.43V3.429z"/>`,
    openai: `<path d="M22.2819 9.8211a5.9847 5.9847 0 0 0-.5157-4.9108 6.0462 6.0462 0 0 0-6.5098-2.9A6.0651 6.0651 0 0 0 4.9807 4.1818a5.9847 5.9847 0 0 0-3.9977 2.9 6.0462 6.0462 0 0 0 .7427 7.0966 5.98 5.98 0 0 0 .511 4.9107 6.051 6.051 0 0 0 6.5146 2.9001A5.9847 5.9847 0 0 0 13.2599 24a6.0557 6.0557 0 0 0 5.7718-4.2058 5.9894 5.9894 0 0 0 3.9977-2.9001 6.0557 6.0557 0 0 0-.7475-7.0729zm-9.022 12.6081a4.4755 4.4755 0 0 1-2.8764-1.0408l.1419-.0804 4.7783-2.7582a.7948.7948 0 0 0 .3927-.6813v-6.7369l2.02 1.1686a.071.071 0 0 1 .038.052v5.5826a4.504 4.504 0 0 1-4.4945 4.4944zm-9.6607-4.1254a4.4708 4.4708 0 0 1-.5346-3.0137l.142.0852 4.783 2.7582a.7712.7712 0 0 0 .7806 0l5.8428-3.3685v2.3324a.0804.0804 0 0 1-.0332.0615L9.74 19.9502a4.4992 4.4992 0 0 1-6.1408-1.6464zM2.3408 7.8956a4.485 4.485 0 0 1 2.3655-1.9728V11.6a.7664.7664 0 0 0 .3879.6765l5.8144 3.3543-2.0201 1.1685a.0757.0757 0 0 1-.071 0l-4.8303-2.7865A4.504 4.504 0 0 1 2.3408 7.872zm16.5963 3.8558L13.1038 8.364 15.1192 7.2a.0757.0757 0 0 1 .071 0l4.8303 2.7913a4.4944 4.4944 0 0 1-.6765 8.1042v-5.6772a.79.79 0 0 0-.407-.667zm2.0107-3.0231l-.142-.0852-4.7735-2.7818a.7759.7759 0 0 0-.7854 0L9.409 9.2297V6.8974a.0662.0662 0 0 1 .0284-.0615l4.8303-2.7866a4.4992 4.4992 0 0 1 6.6802 4.66zM8.3065 12.863l-2.02-1.1638a.0804.0804 0 0 1-.038-.0567V6.0742a4.4992 4.4992 0 0 1 7.3757-3.4537l-.142.0805L8.704 5.459a.7948.7948 0 0 0-.3927.6813zm1.0976-2.3654l2.602-1.4998 2.6069 1.4998v2.9994l-2.5974 1.4997-2.6067-1.4997Z"/>`,
    qwen: `<path d="M23.919 14.545 20.817 9.17l1.47-2.544a.56.56 0 0 0 0-.566l-1.633-2.83a.57.57 0 0 0-.49-.283h-6.207L12.487.402a.57.57 0 0 0-.49-.284H8.732a.56.56 0 0 0-.49.284L5.139 5.775h-2.94a.56.56 0 0 0-.49.284L.077 8.887a.56.56 0 0 0 0 .567L3.18 14.83l-1.47 2.545a.56.56 0 0 0 0 .566l1.634 2.83a.57.57 0 0 0 .49.283h6.205l1.47 2.545a.57.57 0 0 0 .49.284h3.266a.57.57 0 0 0 .49-.284l3.104-5.375h2.94a.57.57 0 0 0 .49-.283l1.634-2.828a.55.55 0 0 0-.004-.568M8.733.686l1.634 2.828-1.634 2.828H21.8L20.164 9.17H7.425L5.63 6.06Zm1.306 19.801-6.205-.002 1.634-2.83h3.265L2.201 6.344h3.267q3.182 5.517 6.367 11.032zm10.124-5.66L18.53 12l-6.532 11.315-1.634-2.83c2.129-3.673 4.25-7.351 6.373-11.028h3.592l3.102 5.374z"/>`
  };
  const PROVIDER_MARKS = {
    Meta: {
      icon: "meta",
      c: "#0467DF",
      d: "#2279E3"
    },
    MiniMax: {
      icon: "minimax",
      c: "#E73562",
      d: "#E73562"
    },
    "Mistral AI": {
      icon: "mistralai",
      c: "#FA520F",
      d: "#FA520F"
    },
    DeepSeek: {
      icon: "deepseek",
      c: "#5786FE",
      d: "#5786FE"
    },
    OpenAI: {
      icon: "openai",
      c: "#0D0D0D",
      d: "#F5F5F5"
    },
    Google: {
      icon: "google",
      c: "#4285F4",
      d: "#4285F4"
    },
    "Alibaba Cloud": {
      icon: "qwen",
      c: "#6950EF",
      d: "#7B65F1"
    }
  };
  const NEUTRAL = {
    icon: "generic",
    c: "#645D70",
    d: "#A9A1B6"
  };
  const markFor = (p, override) => override ? {
    ...NEUTRAL,
    icon: override
  } : PROVIDER_MARKS[p] || NEUTRAL;
  const CSS = `
  .sst { --sst-bd: rgba(0,0,0,.10); --sst-mu: rgba(0,0,0,.55); --sst-sf: rgba(0,0,0,.025); --sst-sf2: rgba(0,0,0,.05); }
  .dark .sst { --sst-bd: rgba(255,255,255,.14); --sst-mu: rgba(255,255,255,.60); --sst-sf: rgba(255,255,255,.04); --sst-sf2: rgba(255,255,255,.07); }

  /* ---- controls ---- */
  .sst-controls { display:flex; flex-wrap:wrap; gap:10px; align-items:center; margin:4px 0 14px; }
  .sst-search { flex:1 1 300px; min-width:200px; font:inherit; font-size:14px; padding:10px 14px; border:1px solid var(--sst-bd); border-radius:10px; background:transparent; color:inherit; }
  .sst-search:focus { outline:none; border-color:${BRAND}; box-shadow:0 0 0 3px rgba(151,79,199,.14); }
  .sst-chips { display:flex; flex-wrap:wrap; gap:7px; margin-bottom:10px; }
  .sst-chip { display:inline-flex; align-items:center; gap:7px; font:inherit; font-size:12.5px; padding:5px 12px 5px 6px; border:1px solid var(--sst-bd); border-radius:999px; background:transparent; color:inherit; cursor:pointer; transition:.12s; }
  .sst-chip-plain { padding-left:12px; }
  .sst-chip:hover { border-color:${BRAND}; }
  .sst-chip-on { border-color:${BRAND}; background:rgba(151,79,199,.12); color:${BRAND}; font-weight:600; }
  .sst-chip-n { opacity:.55; font-variant-numeric:tabular-nums; }
  .sst-count { font-size:13px; color:var(--sst-mu); margin:0 0 14px; }

  /* ---- overview grid ---- */
  .sst-grid { display:grid; grid-template-columns:repeat(auto-fill,minmax(260px,1fr)); gap:12px; }
  .sst-mini { display:flex; flex-direction:column; gap:9px; padding:14px; border:1px solid var(--sst-bd); border-radius:12px; background:var(--sst-sf); text-decoration:none!important; color:inherit!important; transition:.12s; }
  .sst-mini:hover { border-color:${BRAND}; background:var(--sst-sf2); transform:translateY(-1px); }
  .sst-mini-top { display:flex; justify-content:space-between; align-items:center; gap:8px; }
  .sst-prov { font-size:10.5px; font-weight:700; letter-spacing:.06em; text-transform:uppercase; color:var(--sst-mu); }
  .sst-mini-id { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:13px; font-weight:600; word-break:break-all; line-height:1.35; }
  .sst-mini-foot { display:flex; flex-wrap:wrap; gap:6px; align-items:center; margin-top:auto; }

  /* ---- provider mark ---- */
  .sst-mark {
    width:34px; height:34px; flex-shrink:0;
    display:inline-flex; align-items:center; justify-content:center;
    font-family:ui-monospace,SFMono-Regular,Menlo,monospace;
    font-size:12px; font-weight:700; letter-spacing:-.02em; line-height:1;
    border-radius:9px;
    color:var(--mk);
    background:color-mix(in srgb, var(--mk) 13%, transparent);
    border:1px solid color-mix(in srgb, var(--mk) 24%, transparent);
  }
  .dark .sst-mark {
    color:var(--mkd);
    background:color-mix(in srgb, var(--mkd) 15%, transparent);
    border-color:color-mix(in srgb, var(--mkd) 28%, transparent);
  }
  .sst-mark-sm { width:20px; height:20px; border-radius:6px; font-size:9px; }
  /* Inline brand marks. The SVG is emitted with fill="currentColor", so the
     glyph takes the colour set on the tile — which is the provider hue, with a
     separate value for dark mode. A mask was used before, against an external
     SVG; that broke under the production /docs base path. */
  .sst-mark-img { width:18px; height:18px; display:block; }
  .sst-mark-sm .sst-mark-img { width:11px; height:11px; }

  /* ---- badges ---- */

  .sst-stat { font-size:11.5px; color:var(--sst-mu); font-variant-numeric:tabular-nums; }
  .sst-tag { font-size:11px; padding:2px 8px; border-radius:999px; border:1px solid var(--sst-bd); color:var(--sst-mu); white-space:nowrap; }
  .sst-tag-mod { border-color:transparent; background:var(--sst-sf2); color:inherit; font-weight:600; }

  /* ---- detail card ---- */
  .sst-card { border:1px solid var(--sst-bd); border-radius:14px; overflow:hidden; margin:2px 0 8px; }
  .sst-head { display:flex; flex-wrap:wrap; justify-content:space-between; align-items:flex-start; gap:12px; padding:16px 18px; background:var(--sst-sf); border-bottom:1px solid var(--sst-bd); }
  .sst-head-l { min-width:0; display:flex; gap:12px; align-items:flex-start; }
  .sst-head-txt { min-width:0; }
  .sst-id { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:15px; font-weight:700; word-break:break-all; line-height:1.3; }
  .sst-use { font-size:13px; color:var(--sst-mu); margin:5px 0 0; }
  .sst-head-r { display:flex; flex-direction:column; align-items:flex-end; gap:6px; flex-shrink:0; }
  .sst-ctx { font-size:19px; font-weight:700; line-height:1; font-variant-numeric:tabular-nums; }
  .sst-ctx-l { font-size:10px; font-weight:600; letter-spacing:.06em; text-transform:uppercase; color:var(--sst-mu); }

  .sst-body { padding:16px 18px; display:flex; flex-direction:column; gap:16px; }
  .sst-row { display:flex; flex-wrap:wrap; gap:7px; align-items:center; }
  .sst-lbl { font-size:10.5px; font-weight:700; letter-spacing:.07em; text-transform:uppercase; color:var(--sst-mu); margin:0 0 7px; }

  /* endpoints */
  .sst-eps { display:grid; grid-template-columns:repeat(auto-fill,minmax(215px,1fr)); gap:8px; }
  .sst-ep { display:flex; align-items:center; gap:9px; padding:9px 11px; border:1px solid var(--sst-bd); border-radius:9px; background:var(--sst-sf); }
  .sst-ep-off { opacity:.5; }
  .sst-dot { width:7px; height:7px; border-radius:50%; flex-shrink:0; }
  .sst-dot-ok { background:#059669; }
  .sst-dot-lim { background:#d97706; }
  .sst-dot-no { background:currentColor; opacity:.3; }
  .sst-ep-t { min-width:0; }
  .sst-ep-n { font-size:12.5px; font-weight:600; line-height:1.3; }
  .sst-ep-p { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:10.5px; color:var(--sst-mu); word-break:break-all; }

  /* features */
  .sst-feats { display:flex; flex-wrap:wrap; gap:7px; }
  .sst-feat { display:inline-flex; align-items:center; gap:6px; font-size:12px; padding:5px 10px; border-radius:999px; border:1px solid var(--sst-bd); }
  .sst-feat-ok { border-color:rgba(5,150,105,.4); background:rgba(5,150,105,.08); }
  .sst-feat-lim { border-color:rgba(217,119,6,.4); background:rgba(217,119,6,.08); }

  /* bundles */
  .sst-bundles { display:flex; flex-direction:column; gap:7px; }
  .sst-bundle { display:flex; flex-wrap:wrap; align-items:center; gap:10px; padding:10px 12px; border:1px solid var(--sst-bd); border-radius:9px; }
  .sst-bundle-sug { border-color:${BRAND}; background:rgba(151,79,199,.06); }
  .sst-bn { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:12px; font-weight:600; word-break:break-all; flex:1 1 190px; }
  .sst-sug { font-size:10px; font-weight:700; letter-spacing:.04em; text-transform:uppercase; color:${BRAND}; background:rgba(151,79,199,.13); padding:3px 7px; border-radius:5px; white-space:nowrap; }

  /* PEF status, from the release's Pef custom resources. Colour-independent
     readers still get the word, so the hue is reinforcement rather than the
     only signal. Alpha backgrounds keep these legible on both themes. */
  .sst-pef { font-size:10px; font-weight:700; letter-spacing:.04em; text-transform:uppercase; padding:3px 7px; border-radius:5px; white-space:nowrap; }
  .sst-pef-stable { color:#15803d; background:rgba(21,128,61,.13); }
  .sst-pef-preview { color:#b45309; background:rgba(180,83,9,.14); }
  .sst-pef-deprecated { color:#b91c1c; background:rgba(185,28,28,.13); }
  :root[data-theme="dark"] .sst-pef-stable, html.dark .sst-pef-stable { color:#4ade80; background:rgba(74,222,128,.15); }
  :root[data-theme="dark"] .sst-pef-preview, html.dark .sst-pef-preview { color:#fbbf24; background:rgba(251,191,36,.16); }
  :root[data-theme="dark"] .sst-pef-deprecated, html.dark .sst-pef-deprecated { color:#f87171; background:rgba(248,113,113,.15); }

  /* deployment options — BYOC and speculative decoding */
  .sst-opts { display:flex; flex-direction:column; gap:6px; }
  .sst-opt { display:flex; flex-wrap:wrap; align-items:baseline; gap:8px; font-size:13px; }
  .sst-opt-k { color:var(--sst-mu); flex:0 0 auto; }
  .sst-opt-v { font-weight:600; }
  .sst-opt-v code { font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:12px; font-weight:600; margin-right:6px; }

  .sst-link { font-size:13px; }
  .sst-empty { font-size:14px; color:var(--sst-mu); padding:20px 0; }

  @media (max-width:640px) {
    .sst-grid { grid-template-columns:1fr; }
    .sst-eps { grid-template-columns:1fr; }
    .sst-head-r { align-items:flex-start; }
  }
  `;
  const [q, setQ] = useState("");
  const [prov, setProv] = useState("all");
  const [mode, setMode] = useState("all");
  const [feat, setFeat] = useState("all");
  const FEATURES = [{
    key: "byoc",
    label: "Custom checkpoints"
  }, {
    key: "specdec",
    label: "Speculative decoding"
  }];
  const needle = q.trim().toLowerCase();
  const haystack = m => [m.id, m.provider, m.use, (m.modalities || []).join(" "), (m.capabilities || []).join(" "), m.byoc ? "byoc custom checkpoints" : "", m.specdec ? "speculative decoding draft model" : ""].join(" ").toLowerCase();
  const matchesFeat = m => feat === "all" || m[feat] === true;
  const afterSearch = models.filter(m => !needle || haystack(m).includes(needle));
  const byProv = afterSearch.filter(m => (mode === "all" || (m.modalities || []).includes(mode)) && matchesFeat(m));
  const byMode = afterSearch.filter(m => (prov === "all" || m.provider === prov) && matchesFeat(m));
  const byFeat = afterSearch.filter(m => (prov === "all" || m.provider === prov) && (mode === "all" || (m.modalities || []).includes(mode)));
  const shown = afterSearch.filter(m => (prov === "all" || m.provider === prov) && (mode === "all" || (m.modalities || []).includes(mode)) && matchesFeat(m));
  const providers = [...new Set(models.map(m => m.provider))].sort();
  const MODALITY_ORDER = ["Text", "Image", "Video", "Audio", "Embedding"];
  const present = new Set(models.flatMap(m => m.modalities || []));
  const modalities = [...MODALITY_ORDER.filter(x => present.has(x)), ...[...present].filter(x => !MODALITY_ORDER.includes(x)).sort()];
  const anchor = s => String(s).toLowerCase().replace(/[^a-z0-9_]+/g, "-").replace(/^-|-$/g, "");
  return <div className="sst">
      <style dangerouslySetInnerHTML={{
    __html: CSS
  }} />

      <div className="sst-controls">
        <input type="text" className="sst-search" value={q} onChange={e => setQ(e.target.value)} placeholder="Search by model ID, provider, or capability…" aria-label="Search models" />
      </div>

      <div className="sst-chips" role="group" aria-label="Filter by provider">
        <button type="button" className={prov === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setProv("all")} aria-pressed={prov === "all"}>
          All providers <span className="sst-chip-n">{byProv.length}</span>
        </button>
        {providers.map(pName => {
    const n = byProv.filter(m => m.provider === pName).length;
    return <button key={pName} type="button" className={prov === pName ? "sst-chip sst-chip-on" : "sst-chip"} onClick={() => setProv(pName)} aria-pressed={prov === pName}>
              <span className="sst-mark sst-mark-sm" style={{
      "--mk": markFor(pName).c,
      "--mkd": markFor(pName).d
    }} aria-hidden="true">
                {MARK_PATHS[markFor(pName).icon] ? <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
      __html: MARK_PATHS[markFor(pName).icon]
    }} /> : markFor(pName).m}
              </span>
              {pName} <span className="sst-chip-n">{n}</span>
            </button>;
  })}
      </div>

      <div className="sst-chips" role="group" aria-label="Filter by modality">
        <button type="button" className={mode === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setMode("all")} aria-pressed={mode === "all"}>
          All modalities <span className="sst-chip-n">{byMode.length}</span>
        </button>
        {modalities.map(x => {
    const n = byMode.filter(m => (m.modalities || []).includes(x)).length;
    return <button key={x} type="button" className={mode === x ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setMode(x)} aria-pressed={mode === x}>
              {x} <span className="sst-chip-n">{n}</span>
            </button>;
  })}
      </div>

      {}
      {FEATURES.some(f => models.some(m => m[f.key] === true)) && <div className="sst-chips" role="group" aria-label="Filter by deployment option">
          <button type="button" className={feat === "all" ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setFeat("all")} aria-pressed={feat === "all"}>
            All options <span className="sst-chip-n">{byFeat.length}</span>
          </button>
          {FEATURES.filter(f => models.some(m => m[f.key] === true)).map(f => {
    const n = byFeat.filter(m => m[f.key] === true).length;
    return <button key={f.key} type="button" className={feat === f.key ? "sst-chip sst-chip-plain sst-chip-on" : "sst-chip sst-chip-plain"} onClick={() => setFeat(f.key)} aria-pressed={feat === f.key}>
                {f.label} <span className="sst-chip-n">{n}</span>
              </button>;
  })}
        </div>}

      <p className="sst-count">
        {shown.length} {shown.length === 1 ? "model" : "models"} — select one to jump to its full
        configuration below
      </p>

      <div className="sst-grid">
        {shown.map(m => <a className="sst-mini" key={m.id} href={`#${anchor(m.id)}`}>
            <div className="sst-mini-top">
              <span className="sst-mark" style={{
    "--mk": markFor(m.provider, m.mark).c,
    "--mkd": markFor(m.provider, m.mark).d
  }} aria-hidden="true">
                {MARK_PATHS[markFor(m.provider, m.mark).icon] ? <svg className="sst-mark-img" viewBox="0 0 24 24" fill="currentColor" role="img" aria-hidden="true" dangerouslySetInnerHTML={{
    __html: MARK_PATHS[markFor(m.provider, m.mark).icon]
  }} /> : markFor(m.provider, m.mark).m}
              </span>
            </div>
            <span className="sst-prov">{m.provider}</span>
            <span className="sst-mini-id">{m.id}</span>
            <div className="sst-row">
              {(m.modalities || []).map(x => <span className="sst-tag sst-tag-mod" key={x}>
                  {x}
                </span>)}
            </div>
            <div className="sst-mini-foot">
              <span className="sst-stat">{m.context} context</span>
              <span className="sst-stat">·</span>
              <span className="sst-stat">
                {m.bundles} {m.bundles === 1 ? "bundle" : "bundles"}
              </span>
            </div>
          </a>)}
      </div>

      {shown.length === 0 && <p className="sst-empty">No models match those filters. Clear the search, or reset a filter above.</p>}
    </div>;
};

SambaStack supports a variety of models that can be deployed to both on-prem and hosted environments. Contact your system administrator to determine which models are available on your deployment. For definitions of the **Preview** and **Production** designations, see the [Glossary](/docs/en/resources/glossary).

## Deployment options

When deploying models in SambaStack, administrators can select from various context length and batch size combinations.

* Smaller batch sizes provide higher token throughput (tokens/second).
* Larger batch sizes provide better concurrency for multiple users.

SambaStack supports two deployment configurations – high-interactivity and high-throughput. See [High-throughput deployment](/docs/en/v2.1.2/sambastack/service-administration/model-deployment/deploying-models-and-bundles/deployment-configurations) for when to use each.

## Finding models on your cluster

You can run the following command to discover available models in your cluster:

```shellscript theme={}
kubectl -n <namespace> get models
```

<Accordion title="kubectl get models does not return the name you send to the API">
  The command lists Kubernetes resource names (`metadata.name`). Inference requests must use the serving name (`spec.name`), which is what the **Model ID** headings below use.

  ```yaml theme={}
  apiVersion: sambanova.ai/v1alpha1
  kind: Model
  metadata:
    name: gemma-4-31b-it    # Kubernetes resource name – what kubectl get models returns
  spec:
    name: gemma-4-31B-it    # serving name – use this in API requests
  ```

  The two often differ only in case or punctuation, so the mismatch is easy to miss. In the example above they differ by a single character.

  To read the serving name of a deployed model, run:

  ```shellscript theme={}
  kubectl -n <namespace> describe model <resource-name>
  ```

  A model can also define `spec.aliases`, additional names that route to the same model. For the full `Model` field reference, see [Deploy custom checkpoints](/docs/en/v2.1.2/sambastack/service-administration/model-deployment/deploying-models-and-bundles/deploy-custom-checkpoints).

  Text to speech is one case where the names diverge: address `/v1/audio/speech` as `qwen3-tts`, not by the talker or vocoder resource names. See [Text to speech](/docs/en/features/speech).
</Accordion>

## All supported models

Search or filter to find a model, then select it to jump to its full configuration.

<StackModelExplorer
  models={[
{ id: "Meta-Llama-3.3-70B-Instruct", provider: "Meta", context: "128k", bundles: 0, byoc: true, specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent, Text to SQL/Cipher" },
{ id: "Meta-Llama-3.1-8B-Instruct", provider: "Meta", context: "16k", bundles: 0, byoc: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Gateway agent, Validation agent" },
{ id: "Meta-Llama-3.1-70B-Instruct", provider: "Meta", context: "32k", bundles: 0, specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent" },
{ id: "Meta-Llama-3.1-405B-Instruct", provider: "Meta", context: "16k", bundles: 0, specdec: true, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Task agent, Tool-calling agent, Code generation" },
{ id: "Llama-4-Maverick-17B-128E-Instruct", provider: "Meta", context: "128k", bundles: 0, modalities: ["Image","Text"], capabilities: ["Function calling","JSON mode"], use: "Image understanding, Task agent, Tool-calling agent" },
{ id: "MiniMax-M3", provider: "MiniMax", context: "512k", bundles: 0, modalities: ["Image","Text","Video"], capabilities: ["Image and video understanding","Optional thinking mode","Function calling"], use: "Coding agent, Long-context agentic tasks, [Image](/docs/en/features/vision) and [video](/docs/en/features/video) understanding" },
{ id: "MiniMax-M2.7", provider: "MiniMax", context: "192k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","Structured output"], use: "Coding agent" },
{ id: "MiniMax-M2.5", provider: "MiniMax", context: "160k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","Structured output"], use: "Coding agent" },
{ id: "Mistral-Large-3-675B-Instruct-2512", provider: "Mistral AI", context: "8k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling"], use: "Multilingual instruction following, Task agent, Tool-calling agent" },
{ id: "DeepSeek-R1-0528", provider: "DeepSeek", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Complex reasoning" },
{ id: "DeepSeek-R1-Distill-Llama-70B", provider: "DeepSeek", context: "128k", bundles: 0, byoc: true, specdec: true, modalities: ["Text"], capabilities: ["Reasoning"], use: "Complex reasoning" },
{ id: "DeepSeek-V3-0324", provider: "DeepSeek", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.1", provider: "DeepSeek", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.2", provider: "DeepSeek", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Optional thinking mode","Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent" },
{ id: "DeepSeek-V3.1-Terminus", provider: "DeepSeek", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","JSON mode","Reasoning"], use: "Main/planner agent, Tool-calling agent" },
{ id: "gpt-oss-120b", provider: "OpenAI", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Reasoning","Function calling","JSON mode","Logit masking","`logit_bias` sampling parameter (see logit_bias)"], use: "Main/planner agent, Tool-calling agent" },
{ id: "gpt-oss-20b", provider: "OpenAI", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Function calling","JSON mode"], use: "Main/planner agent, Tool-calling agent, Reasoning" },
{ id: "Whisper-Large-v3", provider: "OpenAI", context: "—", bundles: 0, modalities: ["Audio"], capabilities: [], use: "Automatic speech recognition (ASR), Audio transcription" },
{ id: "gemma-3-27b-it", provider: "Google", context: "128k", bundles: 0, modalities: ["Image","Text"], capabilities: ["Image understanding","JSON mode"], use: "Image understanding, Task agent" },
{ id: "gemma-3-12b-it", provider: "Google", context: "128k", bundles: 0, modalities: ["Image","Text"], capabilities: ["Image understanding","JSON mode"], use: "Image understanding, Task agent" },
{ id: "gemma-4-31B-it", provider: "Google", context: "256k", bundles: 0, modalities: ["Image","Text","Video"], capabilities: ["Image and video understanding","Optional thinking mode","Function calling","JSON mode"], use: "[Image](/docs/en/features/vision) and [video](/docs/en/features/video) understanding, Task agent" },
{ id: "Qwen3-235B-A22B-Instruct-2507", provider: "Alibaba Cloud", context: "128k", bundles: 0, modalities: ["Text"], capabilities: ["Reasoning"], use: "Agentic planner, Multilingual instruction following" },
{ id: "Qwen3-32B", provider: "Alibaba Cloud", context: "32k", bundles: 0, modalities: ["Text"], capabilities: ["Reasoning"], use: "Task agent, Multilingual instruction following" },
{ id: "Qwen3-TTS-Talker", provider: "Alibaba Cloud", context: "4k", bundles: 0, modalities: ["Audio","Text"], capabilities: [], use: "Text-to-speech" },
{ id: "Qwen3-TTS-Vocoder", provider: "Alibaba Cloud", context: "4k", bundles: 0, modalities: ["Audio"], capabilities: [], use: "Audio synthesis" },
{ id: "Llama-3.3-Swallow-70B-Instruct-v0.4", provider: "Tokyotech-llm", context: "128k", bundles: 0, specdec: true, modalities: ["Text"], capabilities: [], use: "Japanese instruction following, Task agent" },
{ id: "E5-Mistral-7B-Instruct", provider: "Other", context: "4k", bundles: 0, modalities: ["Embedding"], capabilities: [], use: "Vector storage and retrieval (RAG)" },
]}
/>

## Meta

### Meta-Llama-3.3-70B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.3-70B-Instruct" use="Task agent, Tool-calling agent, Text to SQL/Cipher" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16, 32)","8K (1, 2, 4, 8, 16, 32)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1, 2, 4)","128K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct" />

### Meta-Llama-3.1-8B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-8B-Instruct" use="Gateway agent, Validation agent" context="16k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16, 32, 64, 128)","8K (1, 2, 4, 8, 16, 32, 64)","16K (1, 2, 4, 8)"]} customCheckpoints={true} hf="https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct" />

### Meta-Llama-3.1-70B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-70B-Instruct" use="Task agent, Tool-calling agent" context="32k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 2, 4, 8)","16K (1, 2, 4)","32K (1, 4)"]} customCheckpoints={false} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.1-70B-Instruct" />

### Meta-Llama-3.1-405B-Instruct

<StackModelCard provider="Meta" id="Meta-Llama-3.1-405B-Instruct" use="Task agent, Tool-calling agent, Code generation" context="16k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4)","8K (1)","16K (1)"]} customCheckpoints={false} specDecoding={true} hf="https://huggingface.co/meta-llama/Llama-3.1-405B-Instruct" />

### Llama-4-Maverick-17B-128E-Instruct

<StackModelCard provider="Meta" id="Llama-4-Maverick-17B-128E-Instruct" use="Image understanding, Task agent, Tool-calling agent" context="128k" modalities={["Image","Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1)","16K (1, 2, 4)","32K (1)","64K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct" />

## MiniMax

### MiniMax-M3

<StackModelCard provider="MiniMax" id="MiniMax-M3" use="Coding agent, Long-context agentic tasks, [Image](/docs/en/features/vision) and [video](/docs/en/features/video) understanding" context="1024k" modalities={["Image","Text","Video"]} capabilities={["Image and video understanding","Optional thinking mode","Function calling"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"],["Responses","/v1/responses","Supported"]]} sequenceLengths={["8K-32K (1, 2, 4)","64K (1, 2, 4)","128K (1, 2, 4)","256K (1, 2, 4)","512K (1)","1M (1)"]} customCheckpoints={false} hf="https://huggingface.co/MiniMaxAI/MiniMax-M3" />

### MiniMax-M2.7

<StackModelCard provider="MiniMax" id="MiniMax-M2.7" use="Coding agent" context="192k" modalities={["Text"]} capabilities={["Function calling","Structured output"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","160K (2)","192K (2)"]} customCheckpoints={false} hf="https://huggingface.co/MiniMaxAI/MiniMax-M2.7" />

### MiniMax-M2.5

<StackModelCard provider="MiniMax" id="MiniMax-M2.5" use="Coding agent" context="160k" modalities={["Text"]} capabilities={["Function calling","Structured output"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","160K (2)"]} customCheckpoints={false} hf="https://huggingface.co/MiniMaxAI/MiniMax-M2.5" />

## Mistral AI

### Mistral-Large-3-675B-Instruct-2512

<StackModelCard provider="Mistral AI" id="Mistral-Large-3-675B-Instruct-2512" use="Multilingual instruction following, Task agent, Tool-calling agent" context="8k" modalities={["Text"]} capabilities={["Function calling"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1)"]} customCheckpoints={false} hf="https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" />

## DeepSeek

### DeepSeek-R1-0528

<StackModelCard provider="DeepSeek" id="DeepSeek-R1-0528" use="Complex reasoning" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/deepseek-ai/DeepSeek-R1" />

### DeepSeek-R1-Distill-Llama-70B

<StackModelCard provider="DeepSeek" id="DeepSeek-R1-Distill-Llama-70B" use="Complex reasoning" context="128k" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (2, 4, 8, 16, 32)","8K (2, 4, 8, 16, 32)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1, 2, 4)","128K (1)"]} customCheckpoints={true} specDecoding={true} hf="https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Llama-70B" />

### DeepSeek-V3-0324

<StackModelCard provider="DeepSeek" id="DeepSeek-V3-0324" use="Main/planner agent, Tool-calling agent" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3-0324" />

### DeepSeek-V3.1

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.1" use="Main/planner agent, Tool-calling agent" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.1" />

### DeepSeek-V3.2

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.2" use="Main/planner agent, Tool-calling agent" context="128k" modalities={["Text"]} capabilities={["Optional thinking mode","Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.2" />

### DeepSeek-V3.1-Terminus

<StackModelCard provider="DeepSeek" id="DeepSeek-V3.1-Terminus" use="Main/planner agent, Tool-calling agent" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode","Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1)","128K (1)"]} customCheckpoints={false} hf="https://huggingface.co/deepseek-ai/DeepSeek-V3.1-Terminus" />

## OpenAI

### gpt-oss-120b

<StackModelCard provider="OpenAI" id="gpt-oss-120b" use="Main/planner agent, Tool-calling agent" context="128k" modalities={["Text"]} capabilities={["Reasoning","Function calling","JSON mode","Logit masking","`logit_bias` sampling parameter (see logit_bias)"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","128K (2)"]} customCheckpoints={false} hf="https://huggingface.co/openai/gpt-oss-120b" />

### gpt-oss-20b

<StackModelCard provider="OpenAI" id="gpt-oss-20b" use="Main/planner agent, Tool-calling agent, Reasoning" context="128k" modalities={["Text"]} capabilities={["Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K-32K (2, 4, 6, 8)","64K (2, 4)","128K (2)"]} customCheckpoints={false} hf="https://huggingface.co/openai/gpt-oss-20b" />

### Whisper-Large-v3

<StackModelCard provider="OpenAI" id="Whisper-Large-v3" use="Automatic speech recognition (ASR), Audio transcription" context="—" modalities={["Audio"]} endpoints={[["Translation","/v1/audio/translations","Supported"],["Transcription","/v1/audio/transcriptions","Supported"]]} sequenceLengths={["448 (1, 16, 32)"]} customCheckpoints={false} hf="https://huggingface.co/openai/whisper-large-v3" />

## Google

### gemma-3-27b-it

<StackModelCard provider="Google" id="gemma-3-27b-it" use="Image understanding, Task agent" context="128k" modalities={["Image","Text"]} capabilities={["Image understanding","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K-128K (2, 4, 6, 8)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-3-27b-it" />

### gemma-3-12b-it

<StackModelCard provider="Google" id="gemma-3-12b-it" use="Image understanding, Task agent" context="128k" modalities={["Image","Text"]} capabilities={["Image understanding","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["128K (2, 4, 6, 8)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-3-12b-it" />

### gemma-4-31B-it

<StackModelCard provider="Google" id="gemma-4-31B-it" use="[Image](/docs/en/features/vision) and [video](/docs/en/features/video) understanding, Task agent" context="256k" modalities={["Image","Text","Video"]} capabilities={["Image and video understanding","Optional thinking mode","Function calling","JSON mode"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["32K-128K (2, 4, 6, 8)","256K (2)"]} customCheckpoints={false} hf="https://huggingface.co/google/gemma-4-31B-it" />

## Alibaba Cloud

### Qwen3-235B-A22B-Instruct-2507

<StackModelCard provider="Alibaba Cloud" id="Qwen3-235B-A22B-Instruct-2507" use="Agentic planner, Multilingual instruction following" context="128k" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["32K (2, 4, 6, 8)","128K (2)"]} customCheckpoints={false} hf="https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507" />

### Qwen3-32B

<StackModelCard provider="Alibaba Cloud" id="Qwen3-32B" use="Task agent, Multilingual instruction following" context="32k" modalities={["Text"]} capabilities={["Reasoning"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["8K (1, 4)","16K (1)","32K (1, 2)"]} customCheckpoints={false} hf="https://huggingface.co/Qwen/Qwen3-32B" />

### Qwen3-TTS-Talker

<StackModelCard provider="Alibaba Cloud" id="Qwen3-TTS-Talker" use="Text-to-speech" context="4k" modalities={["Audio","Text"]} endpoints={[["Speech","","Supported"]]} sequenceLengths={["4K (2, 4, 8, 16)"]} customCheckpoints={false} hf="https://huggingface.co/collections/Qwen/qwen3-tts" />

### Qwen3-TTS-Vocoder

<StackModelCard provider="Alibaba Cloud" id="Qwen3-TTS-Vocoder" use="Audio synthesis" context="4k" modalities={["Audio"]} endpoints={[["Speech","","Supported"]]} sequenceLengths={["Codes length 5, 8, 10, 16 (16)","4K (1)"]} customCheckpoints={false} hf="https://huggingface.co/collections/Qwen/qwen3-tts" />

## Tokyotech-llm

### Llama-3.3-Swallow-70B-Instruct-v0.4

<StackModelCard provider="Tokyotech-llm" id="Llama-3.3-Swallow-70B-Instruct-v0.4" use="Japanese instruction following, Task agent" context="128k" modalities={["Text"]} endpoints={[["Chat completions","/v1/chat/completions","Supported"]]} sequenceLengths={["4K (1, 2, 4, 8, 16)","8K (1, 2, 4, 8, 16)","16K (1, 2, 4)","32K (1, 2, 4)","64K (1)","128K (1)"]} customCheckpoints={false} specDecoding={true} hf="https://huggingface.co/tokyotech-llm/Llama-3.3-Swallow-70B-Instruct-v0.4" />

## Other

### E5-Mistral-7B-Instruct

<StackModelCard provider="Other" id="E5-Mistral-7B-Instruct" use="Vector storage and retrieval (RAG)" context="4k" modalities={["Embedding"]} endpoints={[["Embeddings","/v1/embeddings","Supported"]]} sequenceLengths={["4K (1, 4, 8, 16, 32)"]} customCheckpoints={false} hf="https://huggingface.co/intfloat/e5-mistral-7b-instruct" />

## Recommended model bundles

In SambaStack, a bundle is a packaged deployment that groups one or more models together with their associated configurations, such as batch size and sequence length. A single model can also be deployed on its own by pairing it with a model profile, without creating a bundle.

For example, deploying the `Meta‑Llama‑3.3‑70B` model with a batch size of 4 and a sequence length of 16K tokens constitutes a single configuration. A bundle, however, can contain multiple such configurations, either for the same model or for different models.

SambaNova’s RDU technology enables several models and configurations to be loaded simultaneously in a single deployment. This allows you to switch instantly between models and between batch‑/sequence‑size profiles as needed. In contrast to traditional GPU systems, where deployments are typically single‑model and static, SambaStack supports multi‑model, multi‑configuration bundles. This approach delivers higher efficiency, greater flexibility, and increased throughput while preserving low latency.

You can run the following command to discover available bundles in your cluster:

```shellscript theme={}
kubectl -n <namespace> get modelbundles
```

The table below lists the recommended bundles for the models currently available in SambaStack. Each entry pairs a model with its recommended deployment bundle.

<Note>
  If the bundles listed below do not satisfy your inference requirements, you can create [custom bundles](/docs/en/v2.1.2/sambastack/service-administration/model-deployment/deploying-models-and-bundles/create-a-custom-bundle) that combine any mix of models and configurations so long as they fit in DDR memory.
</Note>

### Suggested bundles per model

For each model, this section lists the suggested bundle for typical use and any alternative bundles that trade off context length, batch size, or modality support. See [Bundle configurations](#bundle-configurations) below for the seq length/batch size details of each bundle.

#### Meta

<CardGroup cols={2}>
  <Card>
    **`Meta-Llama-3.3-70B-Instruct`**

    **Suggested:** `70b-3dot3-ss-4-8-16-32-64-128k`

    **Alternatives:** `70b-3dot3-ss-full-whisper`, `us-agentic-rag-1-1`, `e5-mistral-70b-64k-128k`
  </Card>

  <Card>
    **`Meta-Llama-3.1-8B-Instruct`**

    **Suggested:** `us-agentic-rag-1-1`

    **Alternatives:** `qwen3-32b-llama405b-s-m`
  </Card>

  <Card>
    **`Meta-Llama-3.1-405B-Instruct`**

    **Suggested:** `qwen3-32b-llama405b-s-m`
  </Card>

  <Card>
    **`Llama-4-Maverick-17B-128E-Instruct`**

    **Suggested:** `llama-4-medium-8-16-32-64-128k`

    **Alternatives:** `llama-4-medium-ss-16k-bs24`, `us-agentic-rag-1-1`
  </Card>
</CardGroup>

#### MiniMax

<CardGroup cols={2}>
  <Card>
    **`MiniMax-M3`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `minimax-m3-32-64-128-256-512k-1m` (full 32K–1M context range)

    **Alternatives:** `minimax-m3-512k` (32K–512K context range), `minimax-m3-32k`
  </Card>

  <Card>
    **`MiniMax-M2.7`**

    **Suggested:** `dyt-minimax-m2p7-32-64-160-192k-pc` (adds [prompt caching](/docs/en/v2.1.2/sambastack/service-administration/performance/prompt-caching))

    **Alternatives:** `dyt-minimax-m2p7-32k-pc`, `dyt-minimax-m2p7-32-64-160-192k`, `dyt-minimax-m2p7-32-160-192k`, `dyt-minimax-m2p7-32k-v2`
  </Card>

  <Card>
    **`MiniMax-M2.5`**

    **Suggested:** `dyt-minimax-m2p5-32-160k`

    **Alternatives:** `dyt-minimax-m2p5-32k`
  </Card>
</CardGroup>

#### Mistral AI

<CardGroup cols={2}>
  <Card>
    **`Mistral-Large-3-675B-Instruct-2512`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `mistral-large-3-fp8-8-16-32k`

    **Alternatives:** `mistral-large-3-fp8-8k`
  </Card>
</CardGroup>

#### DeepSeek

<CardGroup cols={2}>
  <Card>
    **`DeepSeek-R1-0528`**

    **Suggested:**

    * `deepseek-4in1-fp8-128k` (higher context length)

    **Alternatives:** `deepseek-r1-v3-fp8-8k`, `deepseek-r1-v31-fp8-8k`
  </Card>

  <Card>
    **`DeepSeek-V3-0324`**

    **Suggested:**

    * `deepseek-4in1-fp8-128k` (higher context length)

    **Alternatives:** `deepseek-r1-v3-fp8-8k`, `deepseek-v3-v31-fp8-8k`, `deepseek-v3-v3termi-fp8-8k`
  </Card>

  <Card>
    **`DeepSeek-V3.1`**

    **Suggested:**

    * `deepseek-4in1-fp8-128k` (higher context length)

    **Alternatives:** `deepseek-r1-v31-fp8-8k`, `deepseek-v3-v31-fp8-8k`
  </Card>

  <Card>
    **`DeepSeek-V3.1-Terminus`**

    **Suggested:**

    * `deepseek-4in1-fp8-128k` (higher context length)

    **Alternatives:** `deepseek-v3-v3termi-fp8-8k`
  </Card>
</CardGroup>

#### OpenAI

<CardGroup cols={2}>
  <Card>
    **`gpt-oss-120b`**

    **Suggested:** `us-agentic-rag-1-1`

    **Alternatives:** `cd-dyt-gpt-oss-120b-8-32-64-128k`, `gpt-gemma-whisper-mistral`
  </Card>

  <Card>
    **`gpt-oss-20b`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `dyt-gpt-oss-20b-32-64-128k`
  </Card>

  <Card>
    **`Whisper-Large-v3`**

    **Suggested:** `qwen3-32b-whisper-e5-mistral`

    **Alternatives:** `70b-3dot3-ss-full-whisper` (known issue: does not load within default startup time), `gpt-gemma-whisper-mistral`
  </Card>
</CardGroup>

#### Google

<CardGroup cols={2}>
  <Card>
    **`gemma-3-27b-it`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `gemma3-27b-32-128k`
  </Card>

  <Card>
    **`gemma-3-12b-it`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `gemma3-v3`

    **Alternatives:** `gpt-gemma-whisper-mistral`
  </Card>

  <Card>
    **`gemma-4-31B-it`**

    **Suggested:**

    * `gemma-4-31b-32-128-256k` (adds 256K context; text, image, and video support)
    * `gemma-4-31b-32-128k` (up to 128K context; text, image, and video support, higher throughput)

    Both bundles support constrained decoding across all of their sequence lengths. Choose between them on context length and throughput, not capability.
  </Card>
</CardGroup>

#### Alibaba Cloud

<CardGroup cols={2}>
  <Card>
    **`Qwen3-235B-A22B-Instruct-2507`**

    **Suggested:** `dyt-qwen3-235b-32-128k`
  </Card>

  <Card>
    **`Qwen3-32B`**

    **Suggested:** `qwen3-32b-whisper-e5-mistral`

    **Alternatives:** `qwen3-32b-llama405b-s-m`
  </Card>

  <Card>
    **`Qwen3-TTS-Talker`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `qwen3-tts-talker`
  </Card>

  <Card>
    **`Qwen3-TTS-Vocoder`**<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>

    **Suggested:** `qwen3-tts-vocoder`
  </Card>
</CardGroup>

#### Other

<CardGroup cols={2}>
  <Card>
    **`E5-Mistral-7B-Instruct`**

    **Suggested:** `us-agentic-rag-1-1`

    **Alternatives:** `e5-mistral-70b-64k-128k`, `qwen3-32b-whisper-e5-mistral`, `gpt-gemma-whisper-mistral`
  </Card>
</CardGroup>

### Bundle configurations

The table below lists the configuration details for each bundle referenced above.

| Model name                                                                                                                                                                                                          | Bundle name                                                             | Bundle description                                                                                                                                                                                                                                                          | Bundle configuration                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                           |
| ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :---------------------------------------------------------------------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3.1`                                                                                                                                                                          | deepseek-r1-v31-fp8-8k                                                  | <ul><li>Medium context length with low batch size</li></ul>                                                                                                                                                                                                                 | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                      |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3-0324`                                                                                                                                                                       | deepseek-r1-v3-fp8-8k                                                   | <ul><li>Medium context length with low batch size</li></ul>                                                                                                                                                                                                                 | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   |
| `DeepSeek-V3-0324` / <br />`DeepSeek-V3.1`                                                                                                                                                                          | deepseek-v3-v31-fp8-8k                                                  | <ul><li>Medium context length with low batch size</li></ul>                                                                                                                                                                                                                 | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                      |
| `DeepSeek-V3-0324` / <br />`DeepSeek-V3.1-Terminus`                                                                                                                                                                 | deepseek-v3-v3termi-fp8-8k                                              | <ul><li>Medium context length with low batch size</li></ul>                                                                                                                                                                                                                 | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li><li>`DeepSeek-V3.1-Terminus`<ul><li>Seq Length: 8K, BS: 1, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
| `DeepSeek-R1-0528` / <br />`DeepSeek-V3-0324` / <br />`DeepSeek-V3.1` / <br />`DeepSeek-V3.1-Terminus`                                                                                                              | deepseek-4in1-fp8-128k                                                  | <ul><li>Large context length with single batch size</li><li>Four DeepSeek models in one bundle</li></ul>                                                                                                                                                                    | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`DeepSeek-R1-0528`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3-0324`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3.1`<ul><li>Seq Length: 128K, BS: 1</li></ul></li><li>`DeepSeek-V3.1-Terminus`<ul><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          |
| `E5-Mistral-7B-Instruct` / <br />`Meta-Llama-3.1-8B-Instruct` / <br />`Llama-4-Maverick-17B-128E-Instruct` / <br />`Meta-Llama-3.3-70B-Instruct` / <br />`gpt-oss-120b`                                             | us-agentic-rag-1-1                                                      | <ul><li>Small to medium context length with varied batch size</li><li>[Speculative decoding](/docs/en/v2.1.2/sambastack/service-administration/performance/deploy-with-speculative-decoding) supported for `Meta-Llama-3.3-70B`</li></ul>                                        | <details><summary>View</summary><ul><li>`gpt-oss-120b`<ul><li>Seq Length: 32K, BS: 4</li><li>Seq Length: 64K, BS: 2</li><li>Seq Length: 128K, BS: 2</li></ul></li><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li></ul></li><li>`Meta-Llama-3.3-70B` (Target)/ `Meta-Llama-3.2-1B` (Draft)<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 1, 4, 8</li><li>Seq Length: 16K, BS: 1, 4</li><li>Seq Length: 32K, BS: 1, 4</li><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li><li>`Meta-Llama-3.1-8B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 16, 32</li><li>Seq Length: 8K, BS: 1, 4, 16, 32</li><li>Seq Length: 16K, BS: 1, 4, 8</li></ul></li><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li></ul></details> |
| `E5-Mistral-7B-Instruct` / <br />`Meta-Llama-3.3-70B`                                                                                                                                                               | e5-mistral-70b-64k-128k                                                 | <ul><li>Large context length with low batch size</li><li>[Speculative decoding](/docs/en/v2.1.2/sambastack/service-administration/performance/deploy-with-speculative-decoding) supported for `Meta-Llama-3.3-70B`</li></ul>                                                     | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li><li>`Meta-Llama-3.3-70B` (Target)/ `Meta-Llama-3.2-1B` (Draft)<ul><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                            |
| `gemma-3-27b-it`                                                                                                                                                                                                    | gemma3-27b-32-128k                                                      | Homogeneous bundle containing `gemma-3-27b-it` configurations.                                                                                                                                                                                                              | <details><summary>View</summary><ul><li>`gemma-3-27b-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         |
| `gemma-3-12b-it`                                                                                                                                                                                                    | gemma3-v3                                                               | <ul><li>Homogeneous bundle containing `gemma-3-12b-it` configurations.</li><li>Large context length with medium batch size</li></ul>                                                                                                                                        | <details><summary>View</summary><ul><li>`gemma-3-12b-it`<ul><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 |
| `gemma-4-31B-it`                                                                                                                                                                                                    | gemma-4-31b-32-128k                                                     | Homogeneous bundle with constrained decoding for `gemma-4-31B-it`. Supports text, image, and video input.                                                                                                                                                                   | <details><summary>View</summary><ul><li>`gemma-4-31B-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         |
| `gemma-4-31B-it`                                                                                                                                                                                                    | `gemma-4-31b-32-128-256k`                                               | <ul><li>Homogeneous bundle with constrained decoding for `gemma-4-31B-it`. Adds 32K–256K context support. Supports text, image, and video input at all sequence lengths.</li></ul>                                                                                          | <details><summary>View</summary><ul><li>`gemma-4-31B-it`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2, 4, 6, 8</li><li>Seq Length: 256K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         |
| `gpt-oss-120b`                                                                                                                                                                                                      | `cd-dyt-gpt-oss-120b-8-32-64-128k` †                                    | <ul><li>Homogeneous bundle with constrained decoding for `gpt-oss-120b`.</li><li>Covers 8K through 128K context.</li><li>Includes structured output (logit-masking) support.</li></ul>                                                                                      | <details><summary>View</summary><ul><li>`cd-dyt-gpt-oss-120b-8-32-64-128k`<ul><li>Seq Length: 8K, BS: 2, 4, 6, 8</li><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                       |
| `gpt-oss-20b`                                                                                                                                                                                                       | `dyt-gpt-oss-20b-32-64-128k` †                                          | Homogeneous bundle with constrained decoding for `gpt-oss-20b`.                                                                                                                                                                                                             | <details><summary>View</summary><ul><li>`dyt-gpt-oss-20b-32-64-128k`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                    |
| `E5-Mistral-7B-Instruct` / <br />`Whisper-Large-v3` / <br />`gemma-3-12b-it` / <br />`gpt-oss-120b`                                                                                                                 | gpt-gemma-whisper-mistral                                               | <ul><li>Small to medium context length combining embeddings, transcription, image understanding, and a tool-calling agent</li></ul>                                                                                                                                         | <details><summary>View</summary><p><strong>Models:</strong></p><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1</li></ul></li><li>`gemma-3-12b-it`<ul><li>Seq Length: 128K, BS: 2, 4, 6, 8</li></ul></li><li>`gpt-oss-120b`<ul><li>Seq Length: 8K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 |
| `Llama-4-Maverick-17B-128E-Instruct`                                                                                                                                                                                | llama-4-medium-8-16-32-64-128k                                          | <ul><li>Homogeneous bundles containing `Llama-4-Maverick-17B-128E-Instruct` configurations.</li><li>Small to large context length with low batch</li></ul>                                                                                                                  | <details><summary>View</summary><ul><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1</li><li>Seq Length: 64K, BS: 1</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                           |
| `Llama-4-Maverick-17B-128E-Instruct`                                                                                                                                                                                | llama-4-medium-ss-16k-bs24                                              | Homogeneous bundle containing `Llama-4-Maverick-17B-128E-Instruct` configurations; medium context length with higher batch size.                                                                                                                                            | <details><summary>View</summary><ul><li>`Llama-4-Maverick-17B-128E-Instruct`<ul><li>Seq Length: 16K, BS: 2, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                    |
| `Meta-Llama-3.3-70B-Instruct`                                                                                                                                                                                       | 70b-3dot3-ss-4-8-16-32-64-128k                                          | <ul><li>Medium to large context length with low batch size</li></ul>                                                                                                                                                                                                        | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.3-70B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.2-1B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4; private: true</li><li>Seq Length: 128K, BS: 1; private: true</li></ul></li></ul></details>                                                                                                               |
| `Meta-Llama-3.3-70B-Instruct` / <br />`Whisper-Large-v3`                                                                                                                                                            | 70b-3dot3-ss-full-whisper                                               | <ul><li>Full context-length range with low to medium batch size, plus Whisper transcription</li><li>[Speculative decoding](/docs/en/v2.1.2/sambastack/service-administration/performance/deploy-with-speculative-decoding) supported for `Meta-Llama-3.3-70B-Instruct`</li></ul> | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.3-70B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1, 16, 32</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.2-1B-Instruct`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 8K, BS: 2, 4, 8, 16, 32</li><li>Seq Length: 16K, BS: 1, 2, 4</li><li>Seq Length: 32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1</li></ul></li></ul></details>                                                                                   |
| `MiniMax-M2.5`                                                                                                                                                                                                      | <ul><li>dyt-minimax-m2p5-32k</li><li>dyt-minimax-m2p5-32-160k</li></ul> | <ul><li>Homogeneous bundles containing `MiniMax-M2.5` configurations.</li><li>dyt-minimax-m2p5-32k is better for medium sequence lengths and high batching.</li><li>dyt-minimax-m2p5-32-160k is better for higher sequence lengths and low batching</li></ul>               | <details><summary>View</summary><ul><li>dyt-minimax-m2p5-32k<ul><li>Seq Length: 4K-32K, BS: 2, 4, 6, 8</li></ul></li></ul><ul><li>dyt-minimax-m2p5-32-160k<ul><li>Seq Length: 32K, BS: 2</li><li>Seq Length: 160K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                         |
| `MiniMax-M2.7`                                                                                                                                                                                                      | dyt-minimax-m2p7-32k-v2                                                 | Homogeneous bundle containing `MiniMax-M2.7` configurations; medium context length with high batching.                                                                                                                                                                      | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 |
| `MiniMax-M2.7`                                                                                                                                                                                                      | dyt-minimax-m2p7-32-160-192k                                            | Homogeneous bundle containing `MiniMax-M2.7` configurations; better for higher sequence lengths and low batching.                                                                                                                                                           | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                            |
| `MiniMax-M2.7`                                                                                                                                                                                                      | dyt-minimax-m2p7-32-64-160-192k-pc                                      | Homogeneous bundle with prompt caching for `MiniMax-M2.7`.                                                                                                                                                                                                                  | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2, 4</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                          |
| `MiniMax-M2.7`                                                                                                                                                                                                      | dyt-minimax-m2p7-32-64-160-192k                                         | <ul><li>Homogeneous bundle containing `MiniMax-M2.7` configurations; same context range as `dyt-minimax-m2p7-32-64-160-192k-pc` without prompt caching.</li></ul>                                                                                                           | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li><li>Seq Length: 64K, BS: 2</li><li>Seq Length: 160K-192K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
| `MiniMax-M2.7`                                                                                                                                                                                                      | dyt-minimax-m2p7-32k-pc                                                 | Homogeneous bundle with prompt caching for `MiniMax-M2.7`; medium context length with high batching.                                                                                                                                                                        | <details><summary>View</summary><ul><li>`MiniMax-M2.7`<ul><li>Seq Length: 8K-32K, BS: 2, 4, 6, 8</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>                         | minimax-m3-32k                                                          | Homogeneous bundle containing `MiniMax-M3` configurations; medium context length. Supports text, image, and video input.                                                                                                                                                    | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                      |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>                         | minimax-m3-512k                                                         | <ul><li>Homogeneous bundle containing `MiniMax-M3` configurations.</li><li>32K–512K context range; text, image, and video input at all sequence lengths.</li></ul>                                                                                                          | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1, 2, 4</li><li>Seq Length: 256K, BS: 1, 2, 4</li><li>Seq Length: 512K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                     |
| `MiniMax-M3`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>                         | minimax-m3-32-64-128-256-512k-1m                                        | <ul><li>Homogeneous bundle containing `MiniMax-M3` configurations.</li><li>Full 32K–1M context range; text, image, and video input at all sequence lengths.</li></ul>                                                                                                       | <details><summary>View</summary><ul><li>`MiniMax-M3`<ul><li>Seq Length: 8K-32K, BS: 1, 2, 4</li><li>Seq Length: 64K, BS: 1, 2, 4</li><li>Seq Length: 128K, BS: 1, 2, 4</li><li>Seq Length: 256K, BS: 1, 2, 4</li><li>Seq Length: 512K, BS: 1</li><li>Seq Length: 1M, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                       |
| `Mistral-Large-3-675B-Instruct-2512`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | mistral-large-3-fp8-8k                                                  | <ul><li>Homogeneous bundle containing `Mistral-Large-3-675B-Instruct-2512` configurations.</li><li>Preview model – text-only.</li></ul>                                                                                                                                     | <details><summary>View</summary><ul><li>`Mistral-Large-3-675B-Instruct-2512`<ul><li>Seq Length: 8K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                        |
| `Mistral-Large-3-675B-Instruct-2512`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup> | mistral-large-3-fp8-8-16-32k                                            | <ul><li>Homogeneous bundle containing `Mistral-Large-3-675B-Instruct-2512` configurations.</li><li>Wider context range than `mistral-large-3-fp8-8k`. Preview model – text-only.</li></ul>                                                                                  | <details><summary>View</summary><ul><li>`Mistral-Large-3-675B-Instruct-2512`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1, 2</li><li>Seq Length: 32K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                    |
| `Qwen3-235B-A22B-Instruct-2507`                                                                                                                                                                                     | dyt-qwen3-235b-32-128k                                                  | Homogeneous bundle containing `Qwen3-235B-A22B-Instruct-2507` configurations.                                                                                                                                                                                               | <details><summary>View</summary><ul><li>`Qwen3-235B-A22B-Instruct-2507`<ul><li>Seq Length: 32K, BS: 2, 4, 6, 8</li><li>Seq Length: 128K, BS: 2</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                   |
| `Whisper-Large-v3` / <br />`Qwen3-32B` / <br />`E5-Mistral-7B-Instruct`                                                                                                                                             | qwen3-32b-whisper-e5-mistral                                            | <ul><li>Small to medium context length with varied batch size</li></ul>                                                                                                                                                                                                     | <details><summary>View</summary><ul><li>`E5-Mistral-7B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 4, 8, 16, 32</li></ul></li><li>`Qwen3-32B`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1, 2</li></ul></li><li>`Whisper-Large-v3`<ul><li>BS: 1, 16, 32</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                             |
| `Qwen3-32B` / <br />`Meta-Llama-3.1-405B-Instruct`                                                                                                                                                                  | qwen3-32b-llama405b-s-m                                                 | <ul><li>Small to medium context length</li><li>[Speculative decoding](/docs/en/v2.1.2/sambastack/service-administration/performance/deploy-with-speculative-decoding) supported for `Meta-Llama-3.1-405B-Instruct`</li></ul>                                                     | <details><summary>View</summary><p><strong>Target Models:</strong></p><ul><li>`Meta-Llama-3.1-405B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 2, 4</li><li>Seq Length: 8K, BS: 1</li><li>Seq Length: 16K, BS: 1</li></ul></li></ul><p><strong>Draft Models:</strong></p><ul><li>`Meta-Llama-3.1-8B-Instruct`<ul><li>Seq Length: 16K, BS: 1</li></ul></li><li>`Meta-Llama-3.2-3B-Instruct`<ul><li>Seq Length: 4K, BS: 1, 2, 4</li><li>Seq Length: 8K, BS: 1</li></ul></li></ul><p><strong>Routable Models:</strong></p><ul><li>`Qwen3-32B`<ul><li>Seq Length: 8K, BS: 1, 4</li><li>Seq Length: 16K, BS: 1</li><li>Seq Length: 32K, BS: 1</li></ul></li></ul></details>                                                                                                                                                                                             |
| `Qwen3-TTS-Talker`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>                   | qwen3-tts-talker                                                        | Homogeneous bundle containing `Qwen3-TTS-Talker` configurations.                                                                                                                                                                                                            | <details><summary>View</summary><ul><li>`Qwen3-TTS-Talker`<ul><li>Seq Length: 4K, BS: 2, 4, 8, 16</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                |
| `Qwen3-TTS-Vocoder`<sup style={{background:"#4338ca",color:"#ffffff",fontWeight:700,fontSize:"0.62em",letterSpacing:"0.03em",padding:"1px 4px",borderRadius:"3px",marginLeft:"3px"}}>PREVIEW</sup>                  | qwen3-tts-vocoder                                                       | Homogeneous bundle containing `Qwen3-TTS-Vocoder` configurations.                                                                                                                                                                                                           | <details><summary>View</summary><ul><li>`Qwen3-TTS-Vocoder`<ul><li>Codes length: 5, 8, 10, 16, BS: 16</li></ul></li></ul></details>                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                            |

† `cd-dyt-gpt-oss-120b-8-32-64-128k` supports sequence lengths from 8K–128K, and `dyt-gpt-oss-20b-32-64-128k` supports 32K–128K. If you require shorter context lengths (4K–16K) for either model, contact your SambaNova representative.
