{
  "slug": "groq",
  "name": "Groq",
  "url": "https://groq.com",
  "cat": "ai",
  "group": "ai",
  "tagline": "Ultra-fast open-model inference",
  "price": "Per token: Llama 3.1 8B $0.05/$0.08 per M, GPT-OSS 120B $0.15/$0.60, Llama 3.3 70B $0.59/$0.79. Free tier ~30 RPM / 6k TPM / 14.4k RPD. Dev tier adds 10x limits and 25% off.",
  "tier": "mixed",
  "pick": false,
  "what": "An <b>inference host</b> running open-weight models on custom LPU hardware, not a model lab. It serves Llama, Qwen, GPT-OSS and Whisper at several hundred tokens per second — often an order of magnitude faster than GPU-based providers — behind an OpenAI-compatible endpoint.",
  "why": "When latency is the product — voice agents, autocomplete, live chat — Groq makes an open model feel instant. The free tier needs no credit card, so it is the fastest path to a working LLM call.",
  "warn": "No GPT, Claude or Gemini — open weights only, so quality tops out below frontier. Free-tier limits apply per organization, and extra API keys will not raise them.",
  "category": {
    "key": "ai",
    "name": "Model APIs, routers & observability",
    "desc": "Providers, gateways, inference hosts and eval layers — constantly conflated, listed separately here.",
    "group": "ai",
    "groupName": "Put AI inside it"
  },
  "freshness": null,
  "health": {
    "ok": true,
    "status_code": 200,
    "final_url": "https://groq.com/",
    "checked_at": "2026-08-18T18:09:22.039+00:00"
  },
  "corrected_at": null,
  "upstream_changed_at": null
}