1//! Input tokens estimated from characters (contract *Execution*, "A 2//! request fits the model's context"). The tokenizer is unpublished and 3//! may not be derived (MCA §2.3(c)), so the ratio is learned from what 4//! answers reported: characters sent per `usage.input_tokens`, per model, 5//! falling back to jev-axi's conservative 2.6 when nothing is known. 6//! 7//! A request is `REQUEST_TOKENS` of fixed overhead plus its characters 8//! over the ratio, so the ratio is fitted to the tokens above that 9//! overhead, in the same form the estimate uses. 10 11/// Tokens of fixed overhead in every request (Hume's measurement). 12pub const REQUEST_TOKENS: f64 = 267.0; 13 14/// Attempts a request may be billed for: the first and two retries, 15/// since there is no idempotency key (contract *Cost and safety*). 16pub const BILLED_ATTEMPTS: f64 = 3.0; 17 18/// Characters per token: learned, or the fallback. 19#[derive(Clone, Copy, Debug, PartialEq)] 20pub struct TokenRatio(f64); 21 22/// One answered request: the characters it sent and the input tokens 23/// its answer reported. 24#[derive(Clone, Copy, Debug)] 25pub struct Sample { 26 pub chars: u64, 27 pub input_tokens: u64, 28} 29 30impl TokenRatio { 31 /// jev-axi's conservative figure, when no answer has been seen. 32 pub const FALLBACK: TokenRatio = TokenRatio(2.6); 33 34 /// The characters of every sample over their tokens above the 35 /// request overhead, so large requests weigh as what they cost. The 36 /// fallback when the samples carry no tokens or no characters above 37 /// it, since nothing was learned. 38 pub fn learn(samples: impl IntoIterator<Item = Sample>) -> TokenRatio { 39 let (chars, tokens) = samples.into_iter().fold((0.0, 0.0), |(c, t), s| { 40 (c + s.chars as f64, t + (s.input_tokens as f64 - REQUEST_TOKENS).max(0.0)) 41 }); 42 if chars > 0.0 && tokens > 0.0 { TokenRatio(chars / tokens) } else { TokenRatio::FALLBACK } 43 } 44 45 pub fn chars_per_token(self) -> f64 { 46 self.0 47 } 48 49 /// Input tokens of `requests` requests carrying `chars` characters. 50 pub fn tokens(self, requests: f64, chars: f64) -> f64 { 51 requests * REQUEST_TOKENS + chars / self.0 52 } 53 54 /// Dollars `requests` requests carrying `chars` characters cost at 55 /// worst: every one billed [`BILLED_ATTEMPTS`] times, at 56 /// `price_per_mtok` dollars per million input tokens. 57 pub fn worst_case(self, requests: f64, chars: f64, price_per_mtok: f64) -> f64 { 58 self.tokens(requests, chars) * BILLED_ATTEMPTS * price_per_mtok / 1e6 59 } 60} 61 62#[cfg(test)] 63mod tests { 64 use super::*; 65 66 #[test] 67 fn nothing_learned_is_the_fallback() { 68 assert_eq!(TokenRatio::learn([]), TokenRatio::FALLBACK); 69 // All overhead: no characters' worth of tokens to divide by. 70 assert_eq!(TokenRatio::learn([Sample { chars: 100, input_tokens: 267 }]), TokenRatio::FALLBACK); 71 assert_eq!(TokenRatio::learn([Sample { chars: 0, input_tokens: 500 }]), TokenRatio::FALLBACK); 72 } 73 74 #[test] 75 fn learns_characters_per_token_above_the_overhead() { 76 let ratio = TokenRatio::learn([ 77 Sample { chars: 400, input_tokens: 367 }, 78 Sample { chars: 800, input_tokens: 467 }, 79 ]); 80 assert_eq!(ratio.chars_per_token(), 4.0); 81 // The estimate in the same form reproduces what was reported. 82 assert_eq!(ratio.tokens(2.0, 1200.0), 834.0); 83 } 84 85 #[test] 86 fn the_fallback_estimate() { 87 assert_eq!(TokenRatio::FALLBACK.tokens(1.0, 26.0), 277.0); 88 } 89 90 #[test] 91 fn the_worst_case_bills_three_attempts() { 92 assert_eq!(TokenRatio::FALLBACK.worst_case(1.0, 26.0, 1e6), 831.0); 93 } 94}