> ## Documentation Index
> Fetch the complete documentation index at: https://docs.together.ai/llms.txt
> Use this file to discover all available pages before exploring further.

# Supported models

> View the supported models you can deploy or fine-tune for dedicated model inference.

export const SupportedModelsTable = ({type}) => {
  const models = [{
    id: "arch_CcrDm3fX42Vj5LbcLm3Mv",
    organization: "DeepSeek",
    name: "DeepSeek V3.1",
    apiName: "deepseek-ai/DeepSeek-V3.1",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_Cd8c2oXGowXBu3coVsaQY",
    organization: "DeepSeek",
    name: "DeepSeek V4 Flash",
    apiName: "deepseek-ai/DeepSeek-V4-Flash",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }, {
    id: "arch_CdiL8STh1ttQDZn7LEFcn",
    organization: "DeepSeek",
    name: "DeepSeek V4 Flash 0731",
    apiName: "deepseek-ai/DeepSeek-V4-Flash-0731",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }, {
    id: "arch_CcrDkxyjFbuo4sxJRe1k9",
    organization: "DeepSeek",
    name: "DeepSeek V4 Pro",
    apiName: "deepseek-ai/DeepSeek-V4-Pro",
    type: "chat",
    minHardware: "8xnvidia-b200-180gb"
  }, {
    id: "arch_Ce3jkUxWAzJJ7fRsWjAe3",
    organization: "DeepSeek",
    name: "DeepSeek V4 Pro 0813",
    apiName: "deepseek-ai/DeepSeek-V4-Pro-0813",
    type: "chat",
    minHardware: "8xnvidia-b200-180gb"
  }, {
    id: "arch_CcqaviADkb6NJGW71mxX5",
    organization: "Google",
    name: "Gemma 4 26B A4B IT",
    apiName: "google/gemma-4-26B-A4B-it",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqavk2q2uhoAmjqJdABf",
    organization: "Google",
    name: "Gemma 4 31B IT",
    apiName: "google/gemma-4-31B-it",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_CcqavmoVcrn1fJHj8mtsj",
    organization: "Google",
    name: "Gemma 4 E4B IT",
    apiName: "google/gemma-4-E4B-it",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_CcqavoedEn2SPa7mpb6pK",
    organization: "Meta",
    name: "Llama 3.1 8B Instruct",
    apiName: "meta-llama/Llama-3.1-8B-Instruct",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_CcqavqNJtov9XTbZfYRyi",
    organization: "Meta",
    name: "Llama 3.3 70B Instruct",
    apiName: "meta-llama/Llama-3.3-70B-Instruct",
    type: "chat",
    minHardware: "4xnvidia-h100-80gb"
  }, {
    id: "arch_CcrDkrM7ufb4zpkwTu5vN",
    organization: "Meta",
    name: "Llama 4 Maverick 17B 128E Instruct",
    apiName: "meta-llama/Llama-4-Maverick-17B-128E-Instruct",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_Ccqavs6z3mq1S1iVU9YDv",
    organization: "Meta",
    name: "Llama 4 Scout 17B 16E Instruct",
    apiName: "meta-llama/Llama-4-Scout-17B-16E-Instruct",
    type: "chat",
    minHardware: "4xnvidia-h100-80gb"
  }, {
    id: "arch_Cdv3TZ8NaZLx43TETi2Wk",
    organization: "Meta",
    name: "Muse Glimmer 30B",
    apiName: "meta-models/Muse-Glimmer-30B",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_CcrDkwTDBLCXTiQyC3H2q",
    organization: "MiniMaxAI",
    name: "MiniMax M2.7",
    apiName: "MiniMaxAI/MiniMax-M2.7",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }, {
    id: "arch_CcrDkuoEUagAqrEKzY9mY",
    organization: "MiniMaxAI",
    name: "MiniMax M3",
    apiName: "MiniMaxAI/MiniMax-M3",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CcrDeX9oFqf9CzB81ebSp",
    organization: "Mistral AI",
    name: "Mistral 7B Instruct v0.3",
    apiName: "mistralai/Mistral-7B-Instruct-v0.3",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_CcrDm1zoYmw1kKE3REG9M",
    organization: "Mistral AI",
    name: "Mixtral 8x7B Instruct v0.1",
    apiName: "mistralai/Mixtral-8x7B-Instruct-v0.1",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_CcxzRVeR9hUH6hztfKdXh",
    organization: "Moonshot AI",
    name: "Kimi K2.6",
    apiName: "moonshotai/Kimi-K2.6",
    type: "chat",
    minHardware: "8xnvidia-b200-180gb"
  }, {
    id: "arch_CcqavtovsFmzkggHcjDcY",
    organization: "Moonshot AI",
    name: "Kimi K2.7 Code",
    apiName: "moonshotai/Kimi-K2.7-Code",
    type: "chat",
    minHardware: "8xnvidia-b200-180gb"
  }, {
    id: "arch_CcqavvY7YFTQSqxK8wzG5",
    organization: "NVIDIA",
    name: "NVIDIA Nemotron 3 Ultra 550B A55B",
    apiName: "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-NVFP4",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CdvzNy734Yg2gp8hHmmSG",
    organization: "NVIDIA",
    name: "NVIDIA Nemotron 3.5 Lightning 30B",
    apiName: "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-FP8",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_CcxverjKT1Sgkgo38EPEE",
    organization: "OpenAI",
    name: "OpenAI GPT-OSS 120B",
    apiName: "openai/gpt-oss-120b",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_CcsNkeWHG5TM3BzaJA1dT",
    organization: "OpenAI",
    name: "OpenAI GPT-OSS 20B",
    apiName: "openai/gpt-oss-20b",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqavx876DvG8DQLYmVmc",
    organization: "Qwen",
    name: "Qwen2.5 7B Instruct",
    apiName: "Qwen/Qwen2.5-7B-Instruct",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_CcsNkhUdJPVktsLgNkxbJ",
    organization: "Qwen",
    name: "Qwen3 235B A22B Instruct 2507",
    apiName: "Qwen/Qwen3-235B-A22B-Instruct-2507",
    type: "chat",
    minHardware: "4xnvidia-h100-80gb"
  }, {
    id: "arch_CcsNkjGmYPDhMo9SXfKzh",
    organization: "Qwen",
    name: "Qwen3 235B A22B Thinking 2507",
    apiName: "Qwen/Qwen3-235B-A22B-Thinking-2507",
    type: "chat",
    minHardware: "4xnvidia-h100-80gb"
  }, {
    id: "arch_CcqavybAYykq86YQizrW8",
    organization: "Qwen",
    name: "Qwen3 Coder 480B A35B Instruct",
    apiName: "Qwen/Qwen3-Coder-480B-A35B-Instruct",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CcsNkoMM7VTon2UZcwTSR",
    organization: "Qwen",
    name: "Qwen3 Coder Next",
    apiName: "Qwen/Qwen3-Coder-Next",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }, {
    id: "arch_CcsNkpYnHduFoxQbNvfQj",
    organization: "Qwen",
    name: "Qwen3 Next 80B A3B Instruct",
    apiName: "Qwen/Qwen3-Next-80B-A3B-Instruct",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqaw18SLVCjmW9dtCmhT",
    organization: "Qwen",
    name: "Qwen3 VL 8B Instruct",
    apiName: "Qwen/Qwen3-VL-8B-Instruct",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqaw2YGZuFkyXxizoXjv",
    organization: "Qwen",
    name: "Qwen3.5 122B A10B",
    apiName: "Qwen/Qwen3.5-122B-A10B",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }, {
    id: "arch_Ccqaw4917Ds8BdSGmwc16",
    organization: "Qwen",
    name: "Qwen3.5 27B",
    apiName: "Qwen/Qwen3.5-27B",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqaw65qFVXbtnceEDwyE",
    organization: "Qwen",
    name: "Qwen3.5 35B A3B",
    apiName: "Qwen/Qwen3.5-35B-A3B",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_CcrDm5X8pwzEWrnTmjmsQ",
    organization: "Qwen",
    name: "Qwen3.5 397B A17B",
    apiName: "Qwen/Qwen3.5-397B-A17B",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CcqTKyTw8XamS4VedT5US",
    organization: "Qwen",
    name: "Qwen3.5 9B",
    apiName: "Qwen/Qwen3.5-9B",
    type: "chat",
    minHardware: "1xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqaw7eqfHPzgMgV414JN",
    organization: "Qwen",
    name: "Qwen3.6 27B",
    apiName: "Qwen/Qwen3.6-27B",
    type: "chat",
    minHardware: "2xnvidia-h100-80gb"
  }, {
    id: "arch_Ccqaw9EMHakbn3cBnX1Rs",
    organization: "Qwen",
    name: "Qwen3.6 35B A3B",
    apiName: "Qwen/Qwen3.6-35B-A3B",
    type: "chat",
    minHardware: "1xnvidia-b200-180gb"
  }, {
    id: "arch_CeN2FRmFkvmXSbW8ps48g",
    organization: "Qwen",
    name: "Qwen3.8 27B",
    apiName: "Qwen/Qwen3.8-27B",
    type: "chat",
    minHardware: "1xnvidia-b200-180gb"
  }, {
    id: "arch_CdZQygcnWvTWDscbs1emG",
    organization: "Thinking Machines",
    name: "Inkling Small",
    apiName: "thinkingmachines/Inkling-Small",
    type: "chat",
    minHardware: "8xnvidia-h100-80gb"
  }, {
    id: "arch_CcqawBhBM7pwUuAEyrjKc",
    organization: "Z.ai",
    name: "GLM 5.1",
    apiName: "zai-org/GLM-5.1",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CcsNkmePoNXoyvSsCv3om",
    organization: "Z.ai",
    name: "GLM 5.2",
    apiName: "zai-org/GLM-5.2",
    type: "chat",
    minHardware: "4xnvidia-b200-180gb"
  }, {
    id: "arch_CeXJBDXqwPCAku2BFsHgW",
    organization: "Z.ai",
    name: "GLM 5.3",
    apiName: "zai-org/GLM-5.3",
    type: "chat",
    minHardware: "8xnvidia-b200-180gb"
  }, {
    id: "arch_CedHsbQFjuwnACJGTKMpp",
    organization: "Z.ai",
    name: "GLM 5.3 Flash",
    apiName: "zai-org/GLM-5.3-Flash",
    type: "chat",
    minHardware: "2xnvidia-b200-180gb"
  }];
  const typeLabels = {
    audio: "Audio",
    chat: "Chat",
    code: "Code",
    embedding: "Embedding",
    image: "Image",
    language: "Language",
    moderation: "Moderation",
    rerank: "Rerank",
    transcribe: "Transcription",
    video: "Video"
  };
  const listedModels = models.filter(m => type === undefined || m.type === type).sort((a, b) => {
    if (a.type !== b.type) return a.type.localeCompare(b.type);
    if (a.organization !== b.organization) {
      if (a.organization === "") return 1;
      if (b.organization === "") return -1;
      return a.organization.localeCompare(b.organization);
    }
    return a.apiName.localeCompare(b.apiName);
  });
  return <table className="w-full">
      <thead>
        <tr>
          {type === undefined && <th>Type</th>}
          <th>Organization</th>
          <th>Model name</th>
          <th>API model name</th>
          <th>Deployable hardware</th>
        </tr>
      </thead>
      <tbody>
        {listedModels.map(model => <tr key={model.id}>
            {type === undefined && <td>{typeLabels[model.type] ?? model.type}</td>}
            <td>{model.organization}</td>
            <td>{model.name}</td>
            <td>{model.apiName}</td>
            <td>{model.minHardware ? <code>{model.minHardware}</code> : "—"}</td>
          </tr>)}
      </tbody>
    </table>;
};

This page lists the supported models hosted by Together AI for dedicated model inference. To upload a model you fine-tuned, see [Upload a fine-tuned model](/docs/dedicated-endpoints/custom-models).

<Tip>
  If you're not sure which model to use, check out our list of [recommended models](/docs/recommended-models) by use case.
</Tip>

## Models

The **Deployable hardware** column shows the [instance type](/docs/dedicated-endpoints/concepts#instance-type) of each model's smallest published [deployment profile](/docs/dedicated-endpoints/concepts#deployment-profile). A model may offer other profiles on different hardware. See [Pricing](/docs/dedicated-endpoints/pricing#supported-hardware) for the per-hour cost of each instance type.

<SupportedModelsTable />

## List supported models programmatically

The table above is generated from Together's model catalog. To fetch the same catalog from the command line, list the platform-supported models with `tg beta models public`. Filter by product surface (`--product`), input modality (`--modality`), or a search term (`--search`):

```bash CLI theme={null}
# All models available for dedicated inference
tg beta models public --product DEDICATED

# Narrow by modality or search term
tg beta models public --modality TEXT --search qwen
```

Add `--json` to see the full record for each model, including the `deploymentProfiles` array. Each profile is a certified model-and-config pair Together publishes for that model, so it gives you a vetted `model` and `config` that you can pass straight into [creating a deployment](/docs/dedicated-endpoints/manage#create-a-deployment).

The response looks like this:

```json theme={null}
{
  "data": [
    {
      "id": "arch_abc123",
      "name": "zai-org/GLM-5.2",
      "displayName": "GLM 5.2",
      "displayType": "chat",
      "deploymentProfiles": [
        {
          "profileId": "cfg_a",
          "certifiedConfigRevisionId": "cr_certified",
          "certifiedModelRevisionId": "rv_snap",
          "modelName": "zai-org/GLM-5.2-FP8",
          "config": "projects/proj_cfg/configs/cr_certified",
          "model": "projects/proj_weights/models/ml_weight/revisions/rv_snap",
          "parallelism": "TP8",
          "gpuType": "H100",
          "gpuCount": 8
        }
      ]
    }
  ],
  "object": "list"
}
```

Each architecture includes these identity fields:

| Field         | Description                                                                                                                          |
| ------------- | ------------------------------------------------------------------------------------------------------------------------------------ |
| `id`          | Architecture UID (`arch_...`). Pass to `retrieve_supported` to fetch a single entry. The catalog also accepts the architecture slug. |
| `name`        | Catalog-controlled Hugging Face model ID (for example `zai-org/GLM-5.2`).                                                            |
| `displayName` | Catalog-controlled human-readable display name (for example `GLM 5.2`).                                                              |

Each architecture also includes `displayType`, the model's category. Possible values are `chat`, `language`, `code`, `image`, `embedding`, `rerank`, `moderation`, `audio`, `video`, and `transcribe`.

Each deployment profile includes these fields:

| Field         | Description                                                                                                                                                                                                                                                                                                                                                                                     |
| ------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `modelName`   | Name of the deploy model this profile references (`<project_slug>/<model_name>`). This is the model name you pass when you deploy, and it identifies which precision weight the profile uses (for example `zai-org/GLM-5.2-FP8`).                                                                                                                                                               |
| `config`      | Resource name of the certified config revision: `projects/{project_id}/configs/{config_revision_id}`. `{project_id}` is the config's owning project, which is often a platform project rather than your project. Empty when the profile has no config pinned or its owning project is unresolved.                                                                                               |
| `model`       | Resource name of the deployable weight model for this profile (the quantization-specific build, not the architecture's base model): `projects/{project_id}/models/{model_id}[/revisions/{revision_id}]`. This field also exposes the deploy model ID, which the bare `certifiedModelRevisionId` alone does not. Empty when the profile has no model pinned or its owning project is unresolved. |
| `parallelism` | The catalog's free-form parallelism spec (for example `TP8`, `TP4`, `EP`, or `PD`). Not every value is a tensor-parallel degree.                                                                                                                                                                                                                                                                |

The bare `certifiedConfigRevisionId` and `certifiedModelRevisionId` fields remain populated alongside the resource names. Copy `config` and `model` from a profile directly into the `config` and `model` fields when you [create a deployment](/docs/dedicated-endpoints/manage#create-a-deployment).

To list the profiles published for a specific model instead, see [Choose a deployment profile](/docs/dedicated-endpoints/configs).
