{"id":"9d840539-7fcc-4716-b4e6-f3b74cac98fa","name":"Fireworks AI","slug":"fireworks-ai","description":"Fireworks AI is a San Mateo inference platform founded in 2022 by a group of former Meta PyTorch engineers led by CEO Lin Qiao, who ran the PyTorch team before spinning out. Rather than building a proprietary foundation model or standing up its own datacenter campuses, Fireworks positions itself as a serving and fine-tuning layer for open-weight models: developers point at models like Llama, DeepSeek, Qwen, and Mixtral and get low-latency inference, LoRA fine-tuning, and reserved GPU capacity behind a single API. Jensen Huang has publicly called the company \"the TSMC of AI factories,\" a framing Fireworks itself now uses in marketing.\n\nThe company is inference-focused rather than a training-hyperscaler like CoreWeave or Crusoe, and it sits closer to the software layer than to the metal. Its published on-demand GPU rates run $7 to $8 per hour for H100 and H200 (rising in September 2026), $10 to $13 per hour for B200, and up to $20 per hour for GB300, with region-restricted deployments carrying a 1.5x premium. Named customers include Cursor, Vercel, Notion, Sourcegraph, and UiPath, most of them agentic-coding or productivity products where token latency is the binding constraint. Fireworks does not publicly disclose GPU counts or MW footprint; capacity appears to sit on Nvidia silicon rented and reserved from colocation partners rather than owned campuses.\n\nInvestor demand has run well ahead of disclosed operating detail. Sequoia led a $52 million Series B in July 2024; Lightspeed and Index led a $250 million Series C in October 2025 at a $4 billion valuation; and in July 2026 Atreides Management and Index led a $1.51 billion Series D at a $17.5 billion valuation, roughly a 4x mark-up in nine months and one of the largest inference-layer rounds on record.","status":"active","founded":2022,"hq":"San Mateo, California, USA","hqLat":null,"hqLng":null,"hqGeocodedAt":null,"hqAddress":null,"hqPrecision":null,"website":"https://fireworks.ai","fundingTotal":"1812000000","type":null,"roles":[],"reviewStatus":"reviewed","heatIndex":0,"lifecycleState":"active","supersededByCompanyId":null,"sources":[{"url":"https://fireworks.ai/","title":"Fireworks AI home page","publisher":"Fireworks AI"},{"url":"https://fireworks.ai/pricing","title":"Fireworks AI GPU pricing","publisher":"Fireworks AI"},{"url":"https://en.wikipedia.org/wiki/Fireworks_AI","title":"Fireworks AI","publisher":"Wikipedia"}],"keyFacts":[{"label":"Headquarters","value":"San Mateo, California","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"},{"label":"Founded","value":"2022 by ex-Meta PyTorch team led by Lin Qiao","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"},{"label":"Positioning","value":"Inference-focused platform for open-weight LLMs; fine-tuning + serving layer rather than training hyperscaler","sourceUrl":"https://fireworks.ai/"},{"label":"Announced hourly pricing (through Aug 31, 2026)","value":"H100 80GB and H200 141GB at $7.00/hr on-demand; B200 180GB at $10.00/hr; B300 288GB at $12.00/hr; GB300 288GB at $18.00/hr","sourceUrl":"https://fireworks.ai/pricing"},{"label":"Announced hourly pricing (from Sep 1, 2026)","value":"H100/H200 rising to $8.00/hr; B200 to $13.00/hr; B300 to $15.00/hr; GB300 to $20.00/hr; region-restricted deployments carry a 1.5x premium","sourceUrl":"https://fireworks.ai/pricing"},{"label":"Silicon partnerships","value":"Nvidia-primary fleet (H100, H200, B200, B300, GB300); no AMD or custom silicon disclosed on the public pricing page","sourceUrl":"https://fireworks.ai/pricing"},{"label":"GPU inventory and MW footprint","value":"Not publicly disclosed; capacity is reserved and colocated rather than run from owned campuses","sourceUrl":"https://fireworks.ai/"},{"label":"Named customers","value":"Cursor, Vercel, Notion, Sourcegraph, UiPath; concentration in agentic-coding and productivity products","sourceUrl":"https://fireworks.ai/"},{"label":"Series B (Jul 2024)","value":"$52M led by Sequoia Capital","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"},{"label":"Series C (Oct 2025)","value":"$250M at $4B valuation, led by Lightspeed Venture Partners and Index Ventures","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"},{"label":"Series D (Jul 2026)","value":"$1.51B at $17.5B valuation, led by Atreides Management and Index Ventures; roughly a 4x mark-up in nine months","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"},{"label":"Public / private status","value":"Private, venture-backed","sourceUrl":"https://en.wikipedia.org/wiki/Fireworks_AI"}],"aliases":[],"collisionRisk":"low","reviewNote":null,"atsProvider":null,"atsSlug":null,"factoryLocations":null,"manufacturingCapacity":null,"orgType":null,"burnSignal":null,"litigationEvents":null,"insurancePostureDisclosed":null,"unitEconomicsDisclosed":null,"tickerSymbol":null,"stockExchange":null,"secCik":null,"revenueUsd":null,"revenueBasis":null,"revenueAsOf":null,"employeeCount":null,"employeeCountBasis":null,"employeeCountAsOf":null,"marketCapUsd":null,"marketCapAsOf":null,"marketCapSource":null,"createdAt":"2026-08-28T01:48:21.547Z","updatedAt":"2026-08-28T03:45:17.421Z","jsonLd":{"@context":"https://schema.org","@type":"Organization","@id":"https://registry.deploy.report/companies/fireworks-ai","url":"https://registry.deploy.report/companies/fireworks-ai","name":"Fireworks AI","description":"Fireworks AI is a San Mateo inference platform founded in 2022 by a group of former Meta PyTorch engineers led by CEO Lin Qiao, who ran the PyTorch team before spinning out. Rather than building a proprietary foundation model or standing up its own datacenter campuses, Fireworks positions itself as a serving and fine-tuning layer for open-weight models: developers point at models like Llama, DeepSeek, Qwen, and Mixtral and get low-latency inference, LoRA fine-tuning, and reserved GPU capacity behind a single API. Jensen Huang has publicly called the company \"the TSMC of AI factories,\" a framing Fireworks itself now uses in marketing.\n\nThe company is inference-focused rather than a training-hyperscaler like CoreWeave or Crusoe, and it sits closer to the software layer than to the metal. Its published on-demand GPU rates run $7 to $8 per hour for H100 and H200 (rising in September 2026), $10 to $13 per hour for B200, and up to $20 per hour for GB300, with region-restricted deployments carrying a 1.5x premium. Named customers include Cursor, Vercel, Notion, Sourcegraph, and UiPath, most of them agentic-coding or productivity products where token latency is the binding constraint. Fireworks does not publicly disclose GPU counts or MW footprint; capacity appears to sit on Nvidia silicon rented and reserved from colocation partners rather than owned campuses.\n\nInvestor demand has run well ahead of disclosed operating detail. Sequoia led a $52 million Series B in July 2024; Lightspeed and Index led a $250 million Series C in October 2025 at a $4 billion valuation; and in July 2026 Atreides Management and Index led a $1.51 billion Series D at a $17.5 billion valuation, roughly a 4x mark-up in nine months and one of the largest inference-layer rounds on record.","identifier":"9d840539-7fcc-4716-b4e6-f3b74cac98fa","foundingDate":"2022","address":"San Mateo, California, USA","publisher":{"@id":"https://deploy.report/#organization"}},"framework_metadata":{"framework_schema_version":"0.1.0","verification_status":"verified","maturity_stage":null,"lifecycle_state":null,"architectural_position":{"cohort":null,"sub_cohorts":[]},"within_cohort_verified_vs_claimed_pair":null,"cap_flags":[],"verification_depth":{"sources_count":3,"primary_source_types":["knowledge-base"]}}}