{
  "slug": "fireworks-ai",
  "name": "Fireworks AI",
  "url": "https://fireworks.ai",
  "cat": "ai",
  "group": "ai",
  "tagline": "Open-model inference host",
  "price": "Per token serverless across Standard/Priority/Fast tiers; embeddings $0.008-$0.10/M. On-demand GPUs billed per second: H100/H200 ~$7-8/hr, B200 ~$10-13/hr. Fine-tuning $0.50-$40 per M training tokens. $1 free credit.",
  "tier": "mixed",
  "pick": false,
  "what": "A speed-focused <b>inference host</b> for open-weight models, offering serverless token endpoints, per-second on-demand GPU deployments and managed fine-tuning. It competes with Together and Groq on throughput and cold-start behaviour rather than on proprietary models.",
  "why": "Good when you need a specific open model served reliably with tunable speed and price tiers, or want LoRA fine-tunes deployed without managing GPUs. Serverless has no cold starts, which suits bursty consumer traffic.",
  "warn": "Only $1 in free credit, on-demand GPU rates are scheduled to rise on 1 September, and region-restricted deployments carry a 1.5x premium. Dedicated deployments bill while idle.",
  "install": "pip install fireworks-ai",
  "category": {
    "key": "ai",
    "name": "Model APIs, routers & observability",
    "desc": "Providers, gateways, inference hosts and eval layers — constantly conflated, listed separately here.",
    "group": "ai",
    "groupName": "Put AI inside it"
  },
  "freshness": null,
  "health": {
    "ok": true,
    "status_code": 200,
    "final_url": "https://fireworks.ai/",
    "checked_at": "2026-08-18T18:09:22.039+00:00"
  },
  "corrected_at": null,
  "upstream_changed_at": null
}