{
  "slug": "fireworks-ai",
  "name": "Fireworks AI",
  "url": "https://fireworks.ai",
  "cat": "ai",
  "group": "ai",
  "tagline": "Open-model inference host",
  "price": "Per token serverless across Standard/Priority/Fast tiers; embeddings $0.008-$0.10/M. On-demand GPUs billed per second: H100/H200 ~$7-8/hr, B200 ~$10-13/hr. Fine-tuning $0.50-$40 per M training tokens. $1 free credit. New pricing includes: $3/M Input, $15/M Output for one model; $0.15/M Input, $0.5/M Output for another; $1.4/M Input, $4.4/M Output for a third; and $0.3/M Input, $1.2/M Output for a fourth, with varying context windows.",
  "tier": "mixed",
  "pick": false,
  "what": "A speed-focused inference host for open-weight models, offering serverless token endpoints, per-second on-demand GPU deployments and managed fine-tuning. It competes with Together and Groq on throughput and cold-start behaviour rather than on proprietary models.",
  "why": "Good when you need a specific open model served reliably with tunable speed and price tiers, or want LoRA fine-tunes deployed without managing GPUs. Serverless has no cold starts, which suits bursty consumer traffic.",
  "warn": "Only $1 in free credit, on-demand GPU rates are scheduled to rise on 1 September, and region-restricted deployments carry a 1.5x premium. Dedicated deployments bill while idle.",
  "install": "pip install fireworks-ai",
  "format": "text",
  "category": {
    "key": "ai",
    "name": "Model APIs, routers & observability",
    "desc": "Providers, gateways, inference hosts and eval layers — constantly conflated, listed separately here.",
    "group": "ai",
    "groupName": "Put AI inside it"
  },
  "freshness": null,
  "health": {
    "ok": true,
    "status_code": 200,
    "final_url": "https://fireworks.ai/",
    "checked_at": "2026-10-03T09:30:08.926+00:00"
  },
  "corrected_at": "2026-09-26T23:45:04.69+00:00",
  "upstream_changed_at": null
}