lmjtfy.git / packages / llm / src / models.rs
1//! The Workers AI models that could be the LLM, and what they cost.
2
3use crate::{MAX_TOKENS, Usage};
4
5/// Workers AI's free allocation, shared by the whole account, reset at 00:00
6/// UTC (developers.cloudflare.com/workers-ai/platform/pricing, 2026-10-02).
7pub const FREE_NEURONS_PER_DAY: f64 = 10_000.0;
8
9/// A text model with function calling, and its price in neurons per million
10/// tokens, from the same page on the same day.
11#[derive(Clone, Copy, Debug, PartialEq)]
12pub struct Model {
13    pub id: &'static str,
14    pub neurons_per_m_in: f64,
15    pub neurons_per_m_out: f64,
16}
17
18/// Every candidate under Haiku's price, cheapest first by a typical call
19/// (about 700 tokens in, 200 out). The eval walks this list in order.
20pub const CANDIDATES: &[Model] = &[
21    Model { id: "@cf/ibm-granite/granite-4.0-h-micro", neurons_per_m_in: 1542.0, neurons_per_m_out: 10158.0 },
22    Model { id: "@cf/qwen/qwen3-30b-a3b-fp8", neurons_per_m_in: 4625.0, neurons_per_m_out: 30475.0 },
23    Model { id: "@cf/zai-org/glm-4.7-flash", neurons_per_m_in: 5500.0, neurons_per_m_out: 36400.0 },
24    Model { id: "@cf/google/gemma-4-26b-a4b-it", neurons_per_m_in: 9091.0, neurons_per_m_out: 27273.0 },
25    Model { id: "@cf/zai-org/glm-5.3-flash", neurons_per_m_in: 13636.0, neurons_per_m_out: 45455.0 },
26    Model { id: "@cf/openai/gpt-oss-20b", neurons_per_m_in: 18182.0, neurons_per_m_out: 27273.0 },
27    Model { id: "@cf/meta/llama-4-scout-17b-16e-instruct", neurons_per_m_in: 24545.0, neurons_per_m_out: 77273.0 },
28    Model { id: "@cf/mistralai/mistral-small-3.1-24b-instruct", neurons_per_m_in: 31876.0, neurons_per_m_out: 50488.0 },
29    Model { id: "@cf/openai/gpt-oss-120b", neurons_per_m_in: 31818.0, neurons_per_m_out: 68182.0 },
30    Model { id: "@cf/meta/llama-3.3-70b-instruct-fp8-fast", neurons_per_m_in: 26668.0, neurons_per_m_out: 204805.0 },
31];
32
33impl Model {
34    pub fn find(id: &str) -> Option<&'static Model> {
35        CANDIDATES.iter().find(|model| model.id == id)
36    }
37
38    fn priced(&self, tokens_in: f64, tokens_out: f64) -> f64 {
39        (tokens_in * self.neurons_per_m_in + tokens_out * self.neurons_per_m_out) / 1e6
40    }
41
42    /// What a reply cost: Cloudflare's count when it gave one, the price list
43    /// otherwise.
44    pub fn neurons(&self, usage: Usage) -> f64 {
45        usage.neurons.unwrap_or_else(|| self.priced(usage.prompt_tokens as f64, usage.completion_tokens as f64))
46    }
47
48    /// The most a request can cost, before it is sent: the prompt at one
49    /// token per two bytes (generous for English and for JSON), and a reply
50    /// that runs to the token ceiling.
51    pub fn worst_case_neurons(&self, request: &str) -> f64 {
52        self.priced(request.len() as f64 / 2.0, f64::from(MAX_TOKENS))
53    }
54}