1//! The Workers AI models that could be the LLM, and what they cost. 2 3use crate::{MAX_TOKENS, Usage}; 4 5/// Workers AI's free allocation, shared by the whole account, reset at 00:00 6/// UTC (developers.cloudflare.com/workers-ai/platform/pricing, 2026-10-02). 7pub const FREE_NEURONS_PER_DAY: f64 = 10_000.0; 8 9/// A text model with function calling, and its price in neurons per million 10/// tokens, from the same page on the same day. 11#[derive(Clone, Copy, Debug, PartialEq)] 12pub struct Model { 13 pub id: &'static str, 14 pub neurons_per_m_in: f64, 15 pub neurons_per_m_out: f64, 16} 17 18/// Every candidate under Haiku's price, cheapest first by a typical call 19/// (about 700 tokens in, 200 out). The eval walks this list in order. 20pub const CANDIDATES: &[Model] = &[ 21 Model { id: "@cf/ibm-granite/granite-4.0-h-micro", neurons_per_m_in: 1542.0, neurons_per_m_out: 10158.0 }, 22 Model { id: "@cf/qwen/qwen3-30b-a3b-fp8", neurons_per_m_in: 4625.0, neurons_per_m_out: 30475.0 }, 23 Model { id: "@cf/zai-org/glm-4.7-flash", neurons_per_m_in: 5500.0, neurons_per_m_out: 36400.0 }, 24 Model { id: "@cf/google/gemma-4-26b-a4b-it", neurons_per_m_in: 9091.0, neurons_per_m_out: 27273.0 }, 25 Model { id: "@cf/zai-org/glm-5.3-flash", neurons_per_m_in: 13636.0, neurons_per_m_out: 45455.0 }, 26 Model { id: "@cf/openai/gpt-oss-20b", neurons_per_m_in: 18182.0, neurons_per_m_out: 27273.0 }, 27 Model { id: "@cf/meta/llama-4-scout-17b-16e-instruct", neurons_per_m_in: 24545.0, neurons_per_m_out: 77273.0 }, 28 Model { id: "@cf/mistralai/mistral-small-3.1-24b-instruct", neurons_per_m_in: 31876.0, neurons_per_m_out: 50488.0 }, 29 Model { id: "@cf/openai/gpt-oss-120b", neurons_per_m_in: 31818.0, neurons_per_m_out: 68182.0 }, 30 Model { id: "@cf/meta/llama-3.3-70b-instruct-fp8-fast", neurons_per_m_in: 26668.0, neurons_per_m_out: 204805.0 }, 31]; 32 33impl Model { 34 pub fn find(id: &str) -> Option<&'static Model> { 35 CANDIDATES.iter().find(|model| model.id == id) 36 } 37 38 fn priced(&self, tokens_in: f64, tokens_out: f64) -> f64 { 39 (tokens_in * self.neurons_per_m_in + tokens_out * self.neurons_per_m_out) / 1e6 40 } 41 42 /// What a reply cost: Cloudflare's count when it gave one, the price list 43 /// otherwise. 44 pub fn neurons(&self, usage: Usage) -> f64 { 45 usage.neurons.unwrap_or_else(|| self.priced(usage.prompt_tokens as f64, usage.completion_tokens as f64)) 46 } 47 48 /// The most a request can cost, before it is sent: the prompt at one 49 /// token per two bytes (generous for English and for JSON), and a reply 50 /// that runs to the token ceiling. 51 pub fn worst_case_neurons(&self, request: &str) -> f64 { 52 self.priced(request.len() as f64 / 2.0, f64::from(MAX_TOKENS)) 53 } 54}