{
  "slug": "together-ai",
  "name": "Together AI",
  "url": "https://www.together.ai",
  "cat": "ai",
  "group": "ai",
  "tagline": "Open-model inference and GPU cloud",
  "price": "Serverless per token from ~$0.14/M (DeepSeek V4 Flash) to ~$1.04/M (Llama 3.3 70B); embeddings $0.02/M. Dedicated H100 $5.49/GPU-hr, B200 $8.99. Clusters from $3.19/hr reserved.",
  "tier": "mixed",
  "pick": false,
  "what": "An <b>inference host</b> and GPU cloud for open models: serverless per-token endpoints for 100+ open-weight LLMs, plus fine-tuning, provisioned throughput and raw H100/B200 clusters. It runs other labs' weights rather than training frontier models itself.",
  "why": "The natural graduation path when a prompt works on an open model and you now need it cheap, fine-tuned and at volume. One account covers experimentation, LoRA training and dedicated capacity.",
  "warn": "Serverless rate limits scale with sustained traffic, so a launch spike can get throttled. Dedicated GPUs bill for idle time — a forgotten H100 endpoint is roughly $130 a day.",
  "install": "pip install together",
  "category": {
    "key": "ai",
    "name": "Model APIs, routers & observability",
    "desc": "Providers, gateways, inference hosts and eval layers — constantly conflated, listed separately here.",
    "group": "ai",
    "groupName": "Put AI inside it"
  },
  "freshness": null,
  "health": {
    "ok": true,
    "status_code": 200,
    "final_url": "https://www.together.ai/",
    "checked_at": "2026-08-18T23:43:12.623+00:00"
  },
  "corrected_at": null,
  "upstream_changed_at": null
}