plano/crates/common/src/tokenizer.rs

use log::debug;

#[allow(dead_code)]
pub fn token_count(model_name: &str, text: &str) -> Result<usize, String> {
    debug!("TOKENIZER: computing token count for model={}", model_name);
    //HACK: add support for tokenizing mistral and other models
    //filed issue https://github.com/katanemo/arch/issues/222

    let updated_model = match model_name.starts_with("gpt-4") {
        false => {
            debug!(
                "tiktoken_rs: unsupported model: {}, using gpt-4 to compute token count",
                model_name
            );
            "gpt-4o"
        }
        true => {
            if model_name.starts_with("gpt-4.1") {
                "gpt-4o"
            } else {
                model_name
            }
        }
    };

    // Consideration: is it more expensive to instantiate the BPE object every time, or to contend the singleton?
    let bpe = tiktoken_rs::get_bpe_from_model(updated_model).map_err(|e| e.to_string())?;
    Ok(bpe.encode_ordinary(text).len())
}

#[cfg(test)]
mod test {
    use super::*;

    #[test]
    fn encode_ordinary() {
        let model_name = "gpt-3.5-turbo";
        let text = "How many tokens does this sentence have?";
        assert_eq!(
            8,
            token_count(model_name, text).expect("correct tokenization")
        );
    }
}
Integrate Arch-Function-Chat (#449) 2025-04-15 14:39:12 -07:00			`use log::debug;`
Add Ratelimit on request tokens (#44) Signed-off-by: José Ulises Niño Rivera <junr03@users.noreply.github.com> 2024-09-04 17:28:12 -07:00
			`#[allow(dead_code)]`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`pub fn token_count(model_name: &str, text: &str) -> Result<usize, String> {`
add support for v1/messages and transformations (#558) * pushing draft PR * transformations are working. Now need to add some tests next * updated tests and added necessary response transformations for Anthropics' message response object * fixed bugs for integration tests * fixed doc tests * fixed serialization issues with enums on response * adding some debug logs to help * fixed issues with non-streaming responses * updated the stream_context to update response bytes * the serialized bytes length must be set in the response side * fixed the debug statement that was causing the integration tests for wasm to fail * fixing json parsing errors * intentionally removing the headers * making sure that we convert the raw bytes to the correct provider type upstream * fixing non-streaming responses to tranform correctly * /v1/messages works with transformations to and from /v1/chat/completions * updating the CLI and demos to support anthropic vs. claude * adding the anthropic key to the preference based routing tests * fixed test cases and added more structured logs * fixed integration tests and cleaned up logs * added python client tests for anthropic and openai * cleaned up logs and fixed issue with connectivity for llm gateway in weather forecast demo * fixing the tests. python dependency order was broken * updated the openAI client to fix demos * removed the raw response debug statement * fixed the dup cloning issue and cleaned up the ProviderRequestType enum and traits * fixing logs * moved away from string literals to consts * fixed streaming from Anthropic Client to OpenAI * removed debug statement that would likely trip up integration tests * fixed integration tests for llm_gateway * cleaned up test cases and removed unnecessary crates * fixing comments from PR * fixed bug whereby we were sending an OpenAIChatCompletions request object to llm_gateway even though the request may have been AnthropicMessages --------- Co-authored-by: Salman Paracha <salmanparacha@MacBook-Pro-4.local> Co-authored-by: Salman Paracha <salmanparacha@MacBook-Pro-9.local> Co-authored-by: Salman Paracha <salmanparacha@MacBook-Pro-10.local> Co-authored-by: Salman Paracha <salmanparacha@MacBook-Pro-41.local> Co-authored-by: Salman Paracha <salmanparacha@MacBook-Pro-136.local> 2025-09-10 07:40:30 -07:00			`debug!("TOKENIZER: computing token count for model={}", model_name);`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`//HACK: add support for tokenizing mistral and other models`
			`//filed issue https://github.com/katanemo/arch/issues/222`

release 0.3.2 (#507) 2025-06-13 17:02:20 -07:00			`let updated_model = match model_name.starts_with("gpt-4") {`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`false => {`
Use better logs (#452) 2025-03-27 10:40:20 -07:00			`debug!(`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`"tiktoken_rs: unsupported model: {}, using gpt-4 to compute token count",`
			`model_name`
			`);`
release 0.3.2 (#507) 2025-06-13 17:02:20 -07:00			`"gpt-4o"`
			`}`
			`true => {`
			`if model_name.starts_with("gpt-4.1") {`
			`"gpt-4o"`
			`} else {`
			`model_name`
			`}`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`}`
			`};`

Add Ratelimit on request tokens (#44) Signed-off-by: José Ulises Niño Rivera <junr03@users.noreply.github.com> 2024-09-04 17:28:12 -07:00			`// Consideration: is it more expensive to instantiate the BPE object every time, or to contend the singleton?`
use passed in model name in chat completion request (#445) 2025-03-21 15:56:17 -07:00			`let bpe = tiktoken_rs::get_bpe_from_model(updated_model).map_err(\|e\| e.to_string())?;`
Add Ratelimit on request tokens (#44) Signed-off-by: José Ulises Niño Rivera <junr03@users.noreply.github.com> 2024-09-04 17:28:12 -07:00			`Ok(bpe.encode_ordinary(text).len())`
			`}`

			`#[cfg(test)]`
			`mod test {`
			`use super::*;`

			`#[test]`
			`fn encode_ordinary() {`
			`let model_name = "gpt-3.5-turbo";`
			`let text = "How many tokens does this sentence have?";`
			`assert_eq!(`
			`8,`
			`token_count(model_name, text).expect("correct tokenization")`
			`);`
			`}`
			`}`