postjevsql.git / crates / postjevsql-core / src / token_ratio.rs
1//! Input tokens estimated from characters (contract *Execution*, "A
2//! request fits the model's context"). The tokenizer is unpublished and
3//! may not be derived (MCA §2.3(c)), so the ratio is learned from what
4//! answers reported: characters sent per `usage.input_tokens`, per model,
5//! falling back to jev-axi's conservative 2.6 when nothing is known.
6//!
7//! A request is `REQUEST_TOKENS` of fixed overhead plus its characters
8//! over the ratio, so the ratio is fitted to the tokens above that
9//! overhead, in the same form the estimate uses.
10
11/// Tokens of fixed overhead in every request (Hume's measurement).
12pub const REQUEST_TOKENS: f64 = 267.0;
13
14/// Attempts a request may be billed for: the first and two retries,
15/// since there is no idempotency key (contract *Cost and safety*).
16pub const BILLED_ATTEMPTS: f64 = 3.0;
17
18/// Characters per token: learned, or the fallback.
19#[derive(Clone, Copy, Debug, PartialEq)]
20pub struct TokenRatio(f64);
21
22/// One answered request: the characters it sent and the input tokens
23/// its answer reported.
24#[derive(Clone, Copy, Debug)]
25pub struct Sample {
26    pub chars: u64,
27    pub input_tokens: u64,
28}
29
30impl TokenRatio {
31    /// jev-axi's conservative figure, when no answer has been seen.
32    pub const FALLBACK: TokenRatio = TokenRatio(2.6);
33
34    /// The characters of every sample over their tokens above the
35    /// request overhead, so large requests weigh as what they cost. The
36    /// fallback when the samples carry no tokens or no characters above
37    /// it, since nothing was learned.
38    pub fn learn(samples: impl IntoIterator<Item = Sample>) -> TokenRatio {
39        let (chars, tokens) = samples.into_iter().fold((0.0, 0.0), |(c, t), s| {
40            (c + s.chars as f64, t + (s.input_tokens as f64 - REQUEST_TOKENS).max(0.0))
41        });
42        if chars > 0.0 && tokens > 0.0 { TokenRatio(chars / tokens) } else { TokenRatio::FALLBACK }
43    }
44
45    pub fn chars_per_token(self) -> f64 {
46        self.0
47    }
48
49    /// Input tokens of `requests` requests carrying `chars` characters.
50    pub fn tokens(self, requests: f64, chars: f64) -> f64 {
51        requests * REQUEST_TOKENS + chars / self.0
52    }
53
54    /// Dollars `requests` requests carrying `chars` characters cost at
55    /// worst: every one billed [`BILLED_ATTEMPTS`] times, at
56    /// `price_per_mtok` dollars per million input tokens.
57    pub fn worst_case(self, requests: f64, chars: f64, price_per_mtok: f64) -> f64 {
58        self.tokens(requests, chars) * BILLED_ATTEMPTS * price_per_mtok / 1e6
59    }
60}
61
62#[cfg(test)]
63mod tests {
64    use super::*;
65
66    #[test]
67    fn nothing_learned_is_the_fallback() {
68        assert_eq!(TokenRatio::learn([]), TokenRatio::FALLBACK);
69        // All overhead: no characters' worth of tokens to divide by.
70        assert_eq!(TokenRatio::learn([Sample { chars: 100, input_tokens: 267 }]), TokenRatio::FALLBACK);
71        assert_eq!(TokenRatio::learn([Sample { chars: 0, input_tokens: 500 }]), TokenRatio::FALLBACK);
72    }
73
74    #[test]
75    fn learns_characters_per_token_above_the_overhead() {
76        let ratio = TokenRatio::learn([
77            Sample { chars: 400, input_tokens: 367 },
78            Sample { chars: 800, input_tokens: 467 },
79        ]);
80        assert_eq!(ratio.chars_per_token(), 4.0);
81        // The estimate in the same form reproduces what was reported.
82        assert_eq!(ratio.tokens(2.0, 1200.0), 834.0);
83    }
84
85    #[test]
86    fn the_fallback_estimate() {
87        assert_eq!(TokenRatio::FALLBACK.tokens(1.0, 26.0), 277.0);
88    }
89
90    #[test]
91    fn the_worst_case_bills_three_attempts() {
92        assert_eq!(TokenRatio::FALLBACK.worst_case(1.0, 26.0, 1e6), 831.0);
93    }
94}