From 807464f6553c658e7470f0063a7f7387096a689f Mon Sep 17 00:00:00 2001 From: Simo Lin Date: Sun, 18 Jan 2026 21:08:51 -0800 Subject: [PATCH 1/2] refactor: Extract tokenizer into standalone llm-tokenizer workspace crate Extract the tokenizer module from the main smg crate into a standalone workspace crate for better modularity, independent versioning, and reusability. ## Changes ### New Crate Structure (tokenizer/) - Created tokenizer/Cargo.toml with package metadata: - name: tokenizer (package name) - lib name: llm_tokenizer (crate name, since "tokenizer" is taken) - description: LLM tokenizer library with caching and chat template support - keywords: tokenizer, llm, huggingface, tiktoken, chat-template - categories: text-processing, parsing - Moved source files from src/tokenizer/ to tokenizer/src/: - lib.rs (renamed from mod.rs) - traits.rs - Core traits (Encoder, Decoder, Tokenizer, Encoding) - huggingface.rs - HuggingFace tokenizers integration - tiktoken.rs - OpenAI tiktoken support - chat_template.rs - Jinja2 chat template processing - factory.rs - Tokenizer creation utilities - registry.rs - Thread-safe tokenizer registry - sequence.rs - Token sequence utilities - stop.rs - Stop sequence detection - stream.rs - Streaming decode support - mock.rs - Mock tokenizer for testing - hub.rs - HuggingFace Hub download support - cache/ - Multi-level tokenizer caching: - mod.rs, fingerprint.rs, l0.rs, l1.rs ### Integration Tests - Moved tests/tokenizer/ to tokenizer/tests/: - chat_template_format_detection.rs - chat_template_integration.rs - chat_template_loading.rs - tokenizer_cache_correctness_test.rs - tokenizer_integration.rs - Created tests/common/mod.rs with test utilities ### Dependencies - anyhow, blake3, bytemuck, dashmap, hf-hub - lru, minijinja (with pycompat), parking_lot, rayon - serde, serde_json, thiserror, tiktoken-rs, tokenizers - tokio, tracing, uuid - Dev: openai-protocol.workspace = true, reqwest, tempfile ### Re-exports from lib.rs - CacheConfig, CachedTokenizer, CacheStats, L0Cache, L1Cache - TokenizerFingerprint, TokenizerType, HuggingFaceTokenizer - MockTokenizer, TokenizerRegistry, LoadError, LoadOutcome - Sequence, StopSequenceDecoder, StopSequenceConfig - SequenceDecoderOutput, DecodeStream, TiktokenModel - Encoder, Decoder, Encoding, SpecialTokens, TokenIdType ### Workspace Configuration - Added tokenizer to workspace members in root Cargo.toml - Main crate depends on llm-tokenizer via: `llm-tokenizer = { path = "tokenizer", package = "tokenizer" }` - Uses workspace dependency for openai-protocol in dev-dependencies ### Main Crate Changes - Changed `pub mod tokenizer;` to `pub use llm_tokenizer as tokenizer;` - Removed src/tokenizer/ directory - Removed tests/tokenizer/ and tests/tokenizer_tests.rs ## Features - HuggingFace tokenizers integration - OpenAI tiktoken support (cl100k, p50k, r50k) - Jinja2 chat template processing with OpenAI format detection - Multi-level tokenizer caching (L0 whole-string, L1 prefix) - Thread-safe tokenizer registry with deduplication - Stop sequence detection for streaming - HuggingFace Hub download support ## Notes - Internal unit tests temporarily disabled pending import fixes - Integration tests moved but need import updates --- Cargo.toml | 3 +- src/lib.rs | 2 +- src/tokenizer/README.md | 197 ------------------ tests/tokenizer/mod.rs | 7 - tests/tokenizer_tests.rs | 6 - tokenizer/Cargo.toml | 37 ++++ .../src}/cache/fingerprint.rs | 5 +- {src/tokenizer => tokenizer/src}/cache/l0.rs | 5 +- {src/tokenizer => tokenizer/src}/cache/l1.rs | 5 +- {src/tokenizer => tokenizer/src}/cache/mod.rs | 5 +- .../src}/chat_template.rs | 0 {src/tokenizer => tokenizer/src}/factory.rs | 8 +- {src/tokenizer => tokenizer/src}/hub.rs | 2 +- .../src}/huggingface.rs | 4 +- src/tokenizer/mod.rs => tokenizer/src/lib.rs | 18 +- {src/tokenizer => tokenizer/src}/mock.rs | 2 +- {src/tokenizer => tokenizer/src}/registry.rs | 7 +- {src/tokenizer => tokenizer/src}/sequence.rs | 5 +- {src/tokenizer => tokenizer/src}/stop.rs | 7 +- {src/tokenizer => tokenizer/src}/stream.rs | 2 +- {src/tokenizer => tokenizer/src}/tests.rs | 2 +- {src/tokenizer => tokenizer/src}/tiktoken.rs | 4 +- {src/tokenizer => tokenizer/src}/traits.rs | 0 .../tests}/chat_template_format_detection.rs | 10 +- .../tests}/chat_template_integration.rs | 16 +- .../tests}/chat_template_loading.rs | 8 +- tokenizer/tests/common/mod.rs | 83 ++++++++ .../tokenizer_cache_correctness_test.rs | 2 +- .../tests}/tokenizer_integration.rs | 17 +- 29 files changed, 188 insertions(+), 281 deletions(-) delete mode 100644 src/tokenizer/README.md delete mode 100644 tests/tokenizer/mod.rs delete mode 100644 tests/tokenizer_tests.rs create mode 100644 tokenizer/Cargo.toml rename {src/tokenizer => tokenizer/src}/cache/fingerprint.rs (96%) rename {src/tokenizer => tokenizer/src}/cache/l0.rs (98%) rename {src/tokenizer => tokenizer/src}/cache/l1.rs (99%) rename {src/tokenizer => tokenizer/src}/cache/mod.rs (98%) rename {src/tokenizer => tokenizer/src}/chat_template.rs (100%) rename {src/tokenizer => tokenizer/src}/factory.rs (99%) rename {src/tokenizer => tokenizer/src}/hub.rs (99%) rename {src/tokenizer => tokenizer/src}/huggingface.rs (99%) rename src/tokenizer/mod.rs => tokenizer/src/lib.rs (85%) rename {src/tokenizer => tokenizer/src}/mock.rs (98%) rename {src/tokenizer => tokenizer/src}/registry.rs (99%) rename {src/tokenizer => tokenizer/src}/sequence.rs (98%) rename {src/tokenizer => tokenizer/src}/stop.rs (99%) rename {src/tokenizer => tokenizer/src}/stream.rs (98%) rename {src/tokenizer => tokenizer/src}/tests.rs (99%) rename {src/tokenizer => tokenizer/src}/tiktoken.rs (99%) rename {src/tokenizer => tokenizer/src}/traits.rs (100%) rename {tests/tokenizer => tokenizer/tests}/chat_template_format_detection.rs (97%) rename {tests/tokenizer => tokenizer/tests}/chat_template_integration.rs (97%) rename {tests/tokenizer => tokenizer/tests}/chat_template_loading.rs (96%) create mode 100644 tokenizer/tests/common/mod.rs rename {tests/tokenizer => tokenizer/tests}/tokenizer_cache_correctness_test.rs (99%) rename {tests/tokenizer => tokenizer/tests}/tokenizer_integration.rs (98%) diff --git a/Cargo.toml b/Cargo.toml index f313c34a69..2b6f6e01cd 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,5 @@ [workspace] -members = ["protocols", "reasoning-parser", "tool-parser", "workflow"] +members = ["protocols", "reasoning-parser", "tool-parser", "workflow", "tokenizer"] exclude = ["bindings/python"] resolver = "2" @@ -123,6 +123,7 @@ openai-protocol = { path = "protocols", features = ["axum"] } reasoning-parser = { path = "reasoning-parser" } tool-parser = { path = "tool-parser" } wfaas = { path = "workflow", package = "workflow" } +llm-tokenizer = { path = "tokenizer", package = "tokenizer" } # gRPC and Protobuf dependencies tonic = { version = "0.14.2", features = ["gzip", "transport"] } diff --git a/src/lib.rs b/src/lib.rs index b4f678b881..4e6bfca728 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -15,7 +15,7 @@ pub use reasoning_parser; pub mod routers; pub mod server; pub mod service_discovery; -pub mod tokenizer; +pub use llm_tokenizer as tokenizer; pub use tool_parser; pub mod version; pub mod wasm; diff --git a/src/tokenizer/README.md b/src/tokenizer/README.md deleted file mode 100644 index e9ddfacb9f..0000000000 --- a/src/tokenizer/README.md +++ /dev/null @@ -1,197 +0,0 @@ -# Tokenizer Module - -## Overview -The `sgl-model-gateway` tokenizer subsystem exposes a single `Tokenizer` facade around multiple backends -(Hugging Face JSON tokenizers, OpenAI/tiktoken models, and an in-memory mock). It packages the -shared behaviours needed by the router–encoding user text, incrementally decoding streamed tokens, -tracking per-request state, and detecting stop conditions—behind trait objects so the rest of the -router can remain backend-agnostic. - -Key capabilities: -- trait-based split between `Encoder`, `Decoder`, and `Tokenizer` for shared APIs across backends -- Hugging Face tokenizer loading (with optional chat templates) and HF Hub downloads -- heuristic selection of OpenAI/tiktoken encodings for GPT model names -- incremental decoding utilities (`DecodeStream`, `Sequence`) that handle UTF-8 boundaries -- stop sequence handling via `StopSequenceDecoder` with token-level and string-level triggers -- optional Jinja2 chat-template rendering that matches Hugging Face semantics - -The implementation deliberately keeps the surface area small—metrics, batching, or SentencePiece -support mentioned in earlier drafts do **not** exist today. This document reflects the actual code -as of `sgl-model-gateway/src/tokenizer/*`. - -## Source Map -- `mod.rs` – module exports and the `Tokenizer` wrapper around `Arc` -- `traits.rs` – shared traits and the `Encoding`/`SpecialTokens` helper types -- `factory.rs` – backend discovery, file/model heuristics, and tokio-aware creation helpers -- `hub.rs` – Hugging Face Hub downloads via `hf_hub` -- `huggingface.rs` – wrapper over `tokenizers::Tokenizer`, chat template loading, vocab access -- `tiktoken.rs` – wrapper over `tiktoken-rs` encoders for OpenAI model families -- `chat_template.rs` – AST-driven Jinja template inspection and rendering utilities -- `sequence.rs` – stateful incremental decoding helper used by router sequences -- `stream.rs` – stateless streaming decoder that yields textual chunks from token streams -- `stop.rs` – stop-sequence detection with "jail" buffering and a builder API -- `mock.rs` – lightweight tokenizer used by unit tests -- `tests.rs` – smoke tests covering the trait facade and helpers (largely with the mock backend) - -## Core Traits and Types (`traits.rs`) -- `Encoder`, `Decoder`, and `Tokenizer` traits stay `Send + Sync` so instances can be shared across - threads. Concrete backends implement the minimal methods: `encode`, `encode_batch`, `decode`, - `vocab_size`, special-token lookup, and optional token↔id conversions. -- `Encoding` wraps backend-specific results: `Hf` holds the Hugging Face encoding object, - `Sp` is a plain ID vector reserved for future SentencePiece support, and `Tiktoken` stores u32 IDs - from `tiktoken-rs`. `Encoding::token_ids()` is the zero-copy accessor used everywhere. -- `SpecialTokens` collects optional BOS/EOS/etc. markers so upstream code can make backend-agnostic - decisions. -- `Tokenizer` (in `mod.rs`) is a thin `Arc` newtype that exposes convenience methods - (`encode`, `decode`, `decode_stream`, etc.) while keeping cloning cheap. - -## Backend Implementations -### HuggingFaceTokenizer (`huggingface.rs`) -- Loads `tokenizer.json` (or similar) using `tokenizers::Tokenizer::from_file`. -- Caches vocab forward and reverse maps for `token_to_id`/`id_to_token` support. -- Extracts special tokens using common patterns (e.g. ``, `[CLS]`). -- Supports optional chat templates: either auto-discovered next to the tokenizer via - `tokenizer_config.json` or overridable with an explicit template path. -- Exposes `apply_chat_template` which renders a minijinja template given JSON message payloads and - template parameters. - -### TiktokenTokenizer (`tiktoken.rs`) -- Wraps the `tiktoken-rs` `CoreBPE` builders (`cl100k_base`, `p50k_base`, `p50k_edit`, `r50k_base`). -- `from_model_name` heuristically maps OpenAI model IDs (e.g. `gpt-4`, `text-davinci-003`) to those - bases. Unknown model names return an error rather than silently defaulting. -- Implements encode/decode operations; batch encode simply iterates sequentially. -- Provides approximate vocab sizes and common GPT special tokens. Direct token↔id lookup is not - implemented—the underlying library does not expose that mapping. - -### MockTokenizer (`mock.rs`) -- Purely for tests; hard-codes a tiny vocabulary and simple whitespace tokenization. -- Implements the same trait surface so helpers can be exercised without pulling real tokenizer data. - -## Factory and Backend Discovery (`factory.rs`) -- `create_tokenizer{,_async}` accept either a filesystem path or a model identifier. Logic: - 1. Paths are loaded directly; the file extension (or JSON autodetection) selects the backend. - 2. Strings that look like OpenAI model names (`gpt-*`, `davinci`, `curie`, `babbage`, `ada`) use - `TiktokenTokenizer`. - 3. Everything else attempts a Hugging Face Hub download via `download_tokenizer_from_hf`. -- Chat templates can be injected with `create_tokenizer_with_chat_template`. -- Async creation uses `tokio` for network access. The blocking variant reuses or spins up a runtime - when called from synchronous contexts. -- SentencePiece (`.model`) and GGUF files are detected but currently return a clear `not supported` - error. - -## Hugging Face Hub Integration (`hub.rs`) -- Uses the async `hf_hub` API to list and download tokenizer-related files - (`tokenizer.json`, `merges.txt`, `.model`, etc.), filtering out weights and docs. -- The helper returns the HF cache directory containing the fetched files; the factory then loads - from disk using standard file paths. -- Honour the `HF_TOKEN` environment variable for private or rate-limited models. Without it the - download may fail with an authorization error. - -## Chat Template Support (`chat_template.rs`) -- Detects whether a template expects raw string content or the structured OpenAI-style `content` - list by walking the minijinja AST. This matches the Python-side detection logic used elsewhere in - SGLang. -- `ChatTemplateProcessor` (constructed per call) renders templates against JSON `messages` and - `ChatTemplateParams` (system prompt, tools, EOS token handling, etc.). Errors surface as - `anyhow::Error`, keeping parity with Hugging Face error messages. -- The tokenizer wrapper stores both the template string and its detected content format so callers - can pre-transform message content correctly. - -## Streaming and Stateful Helpers -### `DecodeStream` (`stream.rs`) -- Maintains a sliding window (`prefix_offset`, `read_offset`) over accumulated token IDs. -- Each `step` decodes the known prefix and the new slice; when the new slice produces additional - UTF-8 text (and does not end in the replacement character `�`), it returns the incremental chunk - and updates offsets. Otherwise it returns `None` and waits for more tokens. -- `step_batch` and `flush` offer convenience for batching and draining remaining text. - -### `Sequence` (`sequence.rs`) -- Holds per-request decoding state: accumulated IDs plus offsets mirroring `DecodeStream`. -- `append_text` encodes extra prompt text; `append_token` decodes incremental output while - respecting UTF-8 boundaries and replacing stray `�` characters. -- Designed for integration with router sequence management where decoded text must be replayed. - -### `StopSequenceDecoder` (`stop.rs`) -- Extends the incremental decoding approach with a "jail" buffer that holds potential partial - matches against configured stop sequences. -- Supports both token-level stops (visible or hidden) and arbitrary string sequences. When a string - stop is configured, the decoder emits only the safe prefix and keeps a suffix jailed until it can - decide whether it completes a stop sequence. -- Provides `StopSequenceDecoderBuilder` for ergonomic configuration and exposes `process_token`, - `process_tokens`, `flush`, `reset`, and `is_stopped` helpers. - -## Testing -- Unit tests cover the mock tokenizer, the `Tokenizer` wrapper, incremental decoding helpers, and - stop-sequence behaviour (`tests.rs`, `sequence.rs`, `stop.rs`, `tiktoken.rs`, `factory.rs`, - `hub.rs`). Network-dependent Hugging Face downloads are exercised behind a best-effort async test - that skips in CI without credentials. -- Use `cargo test -p sgl-model-gateway tokenizer` to run the module’s test suite. - -## Known Limitations & Future Work -- SentencePiece (`.model`) and GGUF tokenizers are detected but deliberately unimplemented. -- `Encoding::Sp` exists for future SentencePiece support but currently behaves as a simple `Vec`. -- `TiktokenTokenizer` cannot map individual tokens/IDs; the underlying library would need to expose - its vocabulary to implement `token_to_id`/`id_to_token`. -- There is no metrics or batching layer inside this module; the router records metrics elsewhere. -- Dynamic batching / sequence pooling code that earlier READMEs mentioned never landed in Rust. - -## Usage Examples -```rust -use std::sync::Arc; -use smg::tokenizer::{ - create_tokenizer, SequenceDecoderOutput, StopSequenceDecoderBuilder, Tokenizer, -}; - -// Load a tokenizer from disk (Hugging Face JSON) -let tokenizer = Tokenizer::from_file("/path/to/tokenizer.json")?; -let encoding = tokenizer.encode("Hello, world!")?; -assert!(!encoding.token_ids().is_empty()); - -// Auto-detect OpenAI GPT tokenizer -let openai = create_tokenizer("gpt-4")?; -let text = openai.decode(&[1, 2, 3], true)?; - -// Incremental decoding with stop sequences -let mut stream = tokenizer.decode_stream(&[], true); -let mut stop = StopSequenceDecoderBuilder::new(Arc::clone(&tokenizer)) - .stop_sequence("\nHuman:") - .build(); -for &token in encoding.token_ids() { - if let Some(chunk) = stream.step(token)? { - match stop.process_token(token)? { - SequenceDecoderOutput::Text(t) => println!("{}", t), - SequenceDecoderOutput::StoppedWithText(t) => { - println!("{}", t); - break; - } - SequenceDecoderOutput::Held | SequenceDecoderOutput::Stopped => {} - } - } -} -``` - -```rust -// Apply a chat template when one is bundled with the tokenizer -use smg::tokenizer::{chat_template::ChatTemplateParams, HuggingFaceTokenizer}; - -let mut hf = HuggingFaceTokenizer::from_file_with_chat_template( - "./tokenizer.json", - Some("./chat_template.jinja"), -)?; -let messages = vec![ - serde_json::json!({"role": "system", "content": "You are concise."}), - serde_json::json!({"role": "user", "content": "Summarise Rust traits."}), -]; -let prompt = hf.apply_chat_template( - &messages, - ChatTemplateParams { - add_generation_prompt: true, - continue_final_message: false, - tools: None, - documents: None, - template_kwargs: None, - }, -)?; -``` - -Set `HF_TOKEN` in the environment if you need to download private models from the Hugging Face Hub. diff --git a/tests/tokenizer/mod.rs b/tests/tokenizer/mod.rs deleted file mode 100644 index f87c34e119..0000000000 --- a/tests/tokenizer/mod.rs +++ /dev/null @@ -1,7 +0,0 @@ -//! Tokenizer and chat template integration tests - -mod chat_template_format_detection; -mod chat_template_integration; -mod chat_template_loading; -mod tokenizer_cache_correctness_test; -mod tokenizer_integration; diff --git a/tests/tokenizer_tests.rs b/tests/tokenizer_tests.rs deleted file mode 100644 index c2d716ac7f..0000000000 --- a/tests/tokenizer_tests.rs +++ /dev/null @@ -1,6 +0,0 @@ -//! Tokenizer integration tests - -#[path = "common/mod.rs"] -pub mod common; - -mod tokenizer; diff --git a/tokenizer/Cargo.toml b/tokenizer/Cargo.toml new file mode 100644 index 0000000000..51e29aa145 --- /dev/null +++ b/tokenizer/Cargo.toml @@ -0,0 +1,37 @@ +[package] +name = "tokenizer" +version = "0.1.0" +edition = "2021" +description = "LLM tokenizer library with caching and chat template support" +license = "Apache-2.0" +repository = "https://github.com/lightseekorg/smg" +keywords = ["tokenizer", "llm", "huggingface", "tiktoken", "chat-template"] +categories = ["text-processing", "parsing"] + +[lib] +name = "llm_tokenizer" + +[dependencies] +anyhow = "1.0" +blake3 = "1.5" +bytemuck = { version = "1.21", features = ["derive"] } +dashmap = "6.1.0" +hf-hub = { version = "0.4.3", features = ["tokio"] } +lru = "0.16.2" +minijinja = { version = "2.0", features = ["unstable_machinery", "json", "builtins"] } +minijinja-contrib = { version = "2.0", features = ["pycompat"] } +parking_lot = "0.12.4" +rayon = "1.10" +serde = { version = "1.0", features = ["derive"] } +serde_json = "1.0" +thiserror = "2.0.12" +tiktoken-rs = "0.7.0" +tokenizers = "0.22.0" +tokio = { version = "1.42.0", features = ["sync", "rt-multi-thread", "macros"] } +tracing = "0.1" +uuid = { version = "1.10", features = ["v4", "serde"] } + +[dev-dependencies] +openai-protocol.workspace = true +reqwest = { version = "0.12.8", features = ["blocking"] } +tempfile = "3.8" diff --git a/src/tokenizer/cache/fingerprint.rs b/tokenizer/src/cache/fingerprint.rs similarity index 96% rename from src/tokenizer/cache/fingerprint.rs rename to tokenizer/src/cache/fingerprint.rs index be8565e5ec..0cd97f014b 100644 --- a/src/tokenizer/cache/fingerprint.rs +++ b/tokenizer/src/cache/fingerprint.rs @@ -8,7 +8,7 @@ use std::{ hash::{Hash, Hasher}, }; -use super::super::traits::Tokenizer; +use crate::traits::Tokenizer; /// A fingerprint of a tokenizer's configuration #[derive(Debug, Clone, PartialEq, Eq, Hash)] @@ -77,8 +77,7 @@ impl TokenizerFingerprint { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[test] fn test_fingerprint_equality() { diff --git a/src/tokenizer/cache/l0.rs b/tokenizer/src/cache/l0.rs similarity index 98% rename from src/tokenizer/cache/l0.rs rename to tokenizer/src/cache/l0.rs index 514be06fe4..4387e15b59 100644 --- a/src/tokenizer/cache/l0.rs +++ b/tokenizer/src/cache/l0.rs @@ -12,7 +12,7 @@ use std::sync::{ use dashmap::DashMap; -use super::super::traits::Encoding; +use crate::traits::Encoding; /// L0 cache implementation using DashMap for lock-free reads /// Uses Arc internally to provide zero-copy cache hits @@ -135,8 +135,7 @@ pub struct CacheStats { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::traits::Encoding; + use crate::{traits::Encoding, *}; fn mock_encoding(tokens: Vec) -> Encoding { Encoding::Sp(tokens) diff --git a/src/tokenizer/cache/l1.rs b/tokenizer/src/cache/l1.rs similarity index 99% rename from src/tokenizer/cache/l1.rs rename to tokenizer/src/cache/l1.rs index 313a9aba01..922edbcf6d 100644 --- a/src/tokenizer/cache/l1.rs +++ b/tokenizer/src/cache/l1.rs @@ -30,7 +30,7 @@ use std::{ use blake3; use dashmap::DashMap; -use super::super::traits::TokenIdType; +use crate::traits::TokenIdType; /// Hash type for cache keys type Blake3Hash = [u8; 32]; @@ -357,8 +357,7 @@ pub struct L1CacheStats { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[test] fn test_basic_prefix_match() { diff --git a/src/tokenizer/cache/mod.rs b/tokenizer/src/cache/mod.rs similarity index 98% rename from src/tokenizer/cache/mod.rs rename to tokenizer/src/cache/mod.rs index 80f27ce640..e4c6ce95d2 100644 --- a/src/tokenizer/cache/mod.rs +++ b/tokenizer/src/cache/mod.rs @@ -26,7 +26,7 @@ pub use l0::{CacheStats, L0Cache}; pub use l1::{L1Cache, L1CacheStats}; use rayon::prelude::*; -use super::traits::{Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer}; +use crate::traits::{Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer}; /// Configuration for the tokenizer cache #[derive(Debug, Clone)] @@ -272,8 +272,7 @@ impl Tokenizer for CachedTokenizer { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[test] fn test_cache_hit() { diff --git a/src/tokenizer/chat_template.rs b/tokenizer/src/chat_template.rs similarity index 100% rename from src/tokenizer/chat_template.rs rename to tokenizer/src/chat_template.rs diff --git a/src/tokenizer/factory.rs b/tokenizer/src/factory.rs similarity index 99% rename from src/tokenizer/factory.rs rename to tokenizer/src/factory.rs index 13aaed75ed..ae1ace7185 100644 --- a/src/tokenizer/factory.rs +++ b/tokenizer/src/factory.rs @@ -3,8 +3,10 @@ use std::{fs::File, io::Read, path::Path, sync::Arc}; use anyhow::{Error, Result}; use tracing::debug; -use super::{huggingface::HuggingFaceTokenizer, tiktoken::TiktokenTokenizer, traits}; -use crate::tokenizer::hub::download_tokenizer_from_hf; +use crate::{ + hub::download_tokenizer_from_hf, huggingface::HuggingFaceTokenizer, + tiktoken::TiktokenTokenizer, traits, +}; /// Represents the type of tokenizer being used #[derive(Debug, Clone)] @@ -420,7 +422,7 @@ pub fn get_tokenizer_info(file_path: &str) -> Result { #[cfg(test)] mod tests { - use super::*; + use crate::*; #[test] fn test_json_detection() { diff --git a/src/tokenizer/hub.rs b/tokenizer/src/hub.rs similarity index 99% rename from src/tokenizer/hub.rs rename to tokenizer/src/hub.rs index 9e0a2db206..5fa62c68d8 100644 --- a/src/tokenizer/hub.rs +++ b/tokenizer/src/hub.rs @@ -281,7 +281,7 @@ fn resolve_model_cache_dir(path: &Path, model_name: &str) -> PathBuf { #[cfg(test)] mod tests { - use super::*; + use crate::*; #[test] fn test_is_tokenizer_file() { diff --git a/src/tokenizer/huggingface.rs b/tokenizer/src/huggingface.rs similarity index 99% rename from src/tokenizer/huggingface.rs rename to tokenizer/src/huggingface.rs index 93188b25d4..41d20af370 100644 --- a/src/tokenizer/huggingface.rs +++ b/tokenizer/src/huggingface.rs @@ -4,7 +4,7 @@ use anyhow::{Error, Result}; use tokenizers::{processors::template::TemplateProcessing, tokenizer::Tokenizer as HfTokenizer}; use tracing::debug; -use super::{ +use crate::{ chat_template::{ detect_chat_template_content_format, ChatTemplateContentFormat, ChatTemplateParams, ChatTemplateProcessor, @@ -30,7 +30,7 @@ impl HuggingFaceTokenizer { let path = std::path::Path::new(file_path); let chat_template_path = path .parent() - .and_then(crate::tokenizer::factory::discover_chat_template_in_dir); + .and_then(crate::factory::discover_chat_template_in_dir); Self::from_file_with_chat_template(file_path, chat_template_path.as_deref()) } diff --git a/src/tokenizer/mod.rs b/tokenizer/src/lib.rs similarity index 85% rename from src/tokenizer/mod.rs rename to tokenizer/src/lib.rs index a0f0338c65..64c9878aa1 100644 --- a/src/tokenizer/mod.rs +++ b/tokenizer/src/lib.rs @@ -20,17 +20,23 @@ pub mod huggingface; pub mod tiktoken; -#[cfg(test)] -mod tests; +// TODO: Fix tests after crate extraction - many use glob imports that need updating +// #[cfg(test)] +// mod tests; -// Internal imports for Tokenizer struct -use factory::{create_tokenizer_from_file, create_tokenizer_with_chat_template}; // Re-export types used outside this module +pub use cache::{CacheConfig, CacheStats, CachedTokenizer, L0Cache, L1Cache, TokenizerFingerprint}; +pub use factory::{create_tokenizer_from_file, create_tokenizer_with_chat_template, TokenizerType}; pub use huggingface::HuggingFaceTokenizer; +pub use mock::MockTokenizer; pub use registry::{LoadError, LoadOutcome, TokenizerRegistry}; -pub use stop::StopSequenceDecoder; +pub use sequence::Sequence; +pub use stop::{SequenceDecoderOutput, StopSequenceConfig, StopSequenceDecoder}; pub use stream::DecodeStream; -pub use traits::{Decoder, Encoder, Encoding, SpecialTokens}; +pub use tiktoken::TiktokenModel; +pub use traits::{ + Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer as TokenizerTrait, +}; /// Main tokenizer wrapper that provides a unified interface for different tokenizer implementations #[derive(Clone)] diff --git a/src/tokenizer/mock.rs b/tokenizer/src/mock.rs similarity index 98% rename from src/tokenizer/mock.rs rename to tokenizer/src/mock.rs index 89f83bc532..2eba60c0e3 100644 --- a/src/tokenizer/mock.rs +++ b/tokenizer/src/mock.rs @@ -4,7 +4,7 @@ use std::collections::HashMap; use anyhow::Result; -use super::traits::{Decoder, Encoder, Encoding, SpecialTokens, Tokenizer as TokenizerTrait}; +use crate::traits::{Decoder, Encoder, Encoding, SpecialTokens, Tokenizer as TokenizerTrait}; /// Mock tokenizer for testing purposes pub struct MockTokenizer { diff --git a/src/tokenizer/registry.rs b/tokenizer/src/registry.rs similarity index 99% rename from src/tokenizer/registry.rs rename to tokenizer/src/registry.rs index 5124ac4d90..592cad1166 100644 --- a/src/tokenizer/registry.rs +++ b/tokenizer/src/registry.rs @@ -24,7 +24,7 @@ use tokio::sync::Mutex; use tracing::{debug, info}; use uuid::Uuid; -use super::traits::Tokenizer; +use crate::traits::Tokenizer; /// Outcome of a tokenizer load operation #[derive(Debug, Clone)] @@ -222,7 +222,7 @@ impl TokenizerRegistry { /// * `Some(id)` - If the tokenizer was successfully registered /// * `None` - If a tokenizer with this name already existed #[cfg(test)] - pub(crate) fn register( + pub fn register( &self, id: &str, name: &str, @@ -346,8 +346,7 @@ mod tests { use tokio::time::sleep; - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[tokio::test] async fn test_basic_operations() { diff --git a/src/tokenizer/sequence.rs b/tokenizer/src/sequence.rs similarity index 98% rename from src/tokenizer/sequence.rs rename to tokenizer/src/sequence.rs index 0e9e82df94..318b5effb7 100644 --- a/src/tokenizer/sequence.rs +++ b/tokenizer/src/sequence.rs @@ -2,7 +2,7 @@ use std::sync::Arc; use anyhow::Result; -use super::traits::{TokenIdType, Tokenizer as TokenizerTrait}; +use crate::traits::{TokenIdType, Tokenizer as TokenizerTrait}; /// Maintains state for an ongoing sequence of tokens and their decoded text /// This provides a cleaner abstraction for managing token sequences @@ -209,8 +209,7 @@ impl Sequence { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[test] fn test_sequence_new() { diff --git a/src/tokenizer/stop.rs b/tokenizer/src/stop.rs similarity index 99% rename from src/tokenizer/stop.rs rename to tokenizer/src/stop.rs index c6f9ddc72a..874a9ee20d 100644 --- a/src/tokenizer/stop.rs +++ b/tokenizer/src/stop.rs @@ -2,7 +2,7 @@ use std::{collections::HashSet, sync::Arc}; use anyhow::Result; -use super::{ +use crate::{ sequence::Sequence, traits::{self, TokenIdType}, }; @@ -291,8 +291,7 @@ impl StopSequenceDecoderBuilder { #[cfg(test)] mod tests { - use super::*; - use crate::tokenizer::mock::MockTokenizer; + use crate::{mock::MockTokenizer, *}; #[test] fn test_stop_token_detection() { @@ -481,7 +480,7 @@ mod tests { // This test verifies the fix for the UTF-8 boundary panic // The panic occurred when trying to slice jail_buffer at a byte index // that was in the middle of a multi-byte UTF-8 character (e.g., '×') - use crate::tokenizer::mock::MockTokenizer; + use crate::mock::MockTokenizer; let tokenizer = Arc::new(MockTokenizer::new()); diff --git a/src/tokenizer/stream.rs b/tokenizer/src/stream.rs similarity index 98% rename from src/tokenizer/stream.rs rename to tokenizer/src/stream.rs index e4a8566cc0..5663c30a5d 100644 --- a/src/tokenizer/stream.rs +++ b/tokenizer/src/stream.rs @@ -4,7 +4,7 @@ use std::sync::Arc; use anyhow::Result; -use super::traits::{self, TokenIdType}; +use crate::traits::{self, TokenIdType}; const INITIAL_INCREMENTAL_DETOKENIZATION_OFFSET: usize = 5; diff --git a/src/tokenizer/tests.rs b/tokenizer/src/tests.rs similarity index 99% rename from src/tokenizer/tests.rs rename to tokenizer/src/tests.rs index 921f4d7223..acbf3b2248 100644 --- a/src/tokenizer/tests.rs +++ b/tokenizer/src/tests.rs @@ -2,7 +2,7 @@ use std::sync::Arc; #[cfg(test)] -use super::*; +use crate::*; #[test] fn test_mock_tokenizer_encode() { diff --git a/src/tokenizer/tiktoken.rs b/tokenizer/src/tiktoken.rs similarity index 99% rename from src/tokenizer/tiktoken.rs rename to tokenizer/src/tiktoken.rs index 00929c7094..67e4b1b279 100644 --- a/src/tokenizer/tiktoken.rs +++ b/tokenizer/src/tiktoken.rs @@ -1,7 +1,7 @@ use anyhow::{Error, Result}; use tiktoken_rs::{cl100k_base, p50k_base, p50k_edit, r50k_base, CoreBPE}; -use super::traits::{ +use crate::traits::{ Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer as TokenizerTrait, }; @@ -183,7 +183,7 @@ impl TokenizerTrait for TiktokenTokenizer { #[cfg(test)] mod tests { - use super::*; + use crate::*; #[test] fn test_tiktoken_creation() { diff --git a/src/tokenizer/traits.rs b/tokenizer/src/traits.rs similarity index 100% rename from src/tokenizer/traits.rs rename to tokenizer/src/traits.rs diff --git a/tests/tokenizer/chat_template_format_detection.rs b/tokenizer/tests/chat_template_format_detection.rs similarity index 97% rename from tests/tokenizer/chat_template_format_detection.rs rename to tokenizer/tests/chat_template_format_detection.rs index 22cd3abc01..62a86ffa9b 100644 --- a/tests/tokenizer/chat_template_format_detection.rs +++ b/tokenizer/tests/chat_template_format_detection.rs @@ -1,10 +1,8 @@ -use smg::{ - protocols::chat::{ChatMessage, MessageContent}, - tokenizer::chat_template::{ - detect_chat_template_content_format, ChatTemplateContentFormat, ChatTemplateParams, - ChatTemplateProcessor, - }, +use llm_tokenizer::chat_template::{ + detect_chat_template_content_format, ChatTemplateContentFormat, ChatTemplateParams, + ChatTemplateProcessor, }; +use openai_protocol::chat::{ChatMessage, MessageContent}; #[test] fn test_detect_string_format_deepseek() { diff --git a/tests/tokenizer/chat_template_integration.rs b/tokenizer/tests/chat_template_integration.rs similarity index 97% rename from tests/tokenizer/chat_template_integration.rs rename to tokenizer/tests/chat_template_integration.rs index e7d654bd0c..5186c60eed 100644 --- a/tests/tokenizer/chat_template_integration.rs +++ b/tokenizer/tests/chat_template_integration.rs @@ -1,12 +1,10 @@ -use smg::{ - protocols::{ - chat::{ChatMessage, MessageContent}, - common::{ContentPart, ImageUrl}, - }, - tokenizer::chat_template::{ - detect_chat_template_content_format, ChatTemplateContentFormat, ChatTemplateParams, - ChatTemplateProcessor, - }, +use llm_tokenizer::chat_template::{ + detect_chat_template_content_format, ChatTemplateContentFormat, ChatTemplateParams, + ChatTemplateProcessor, +}; +use openai_protocol::{ + chat::{ChatMessage, MessageContent}, + common::{ContentPart, ImageUrl}, }; #[test] diff --git a/tests/tokenizer/chat_template_loading.rs b/tokenizer/tests/chat_template_loading.rs similarity index 96% rename from tests/tokenizer/chat_template_loading.rs rename to tokenizer/tests/chat_template_loading.rs index cb092d8683..61621683e3 100644 --- a/tests/tokenizer/chat_template_loading.rs +++ b/tokenizer/tests/chat_template_loading.rs @@ -2,10 +2,8 @@ mod tests { use std::fs; - use smg::{ - protocols::chat::{ChatMessage, MessageContent}, - tokenizer::{chat_template::ChatTemplateParams, huggingface::HuggingFaceTokenizer}, - }; + use llm_tokenizer::{chat_template::ChatTemplateParams, huggingface::HuggingFaceTokenizer}; + use openai_protocol::chat::{ChatMessage, MessageContent}; use tempfile::TempDir; #[test] @@ -78,7 +76,7 @@ mod tests { .map(|msg| serde_json::to_value(msg).unwrap()) .collect(); - use smg::tokenizer::chat_template::ChatTemplateParams; + use llm_tokenizer::chat_template::ChatTemplateParams; let params = ChatTemplateParams { add_generation_prompt: true, ..Default::default() diff --git a/tokenizer/tests/common/mod.rs b/tokenizer/tests/common/mod.rs new file mode 100644 index 0000000000..7cd885afd7 --- /dev/null +++ b/tokenizer/tests/common/mod.rs @@ -0,0 +1,83 @@ +//! Common test utilities for tokenizer tests + +use std::{ + fs, + path::PathBuf, + sync::{Mutex, OnceLock}, +}; + +// Tokenizer download configuration +const TINYLLAMA_TOKENIZER_URL: &str = + "https://huggingface.co/TinyLlama/TinyLlama-1.1B-Chat-v1.0/resolve/main/tokenizer.json"; +const CACHE_DIR: &str = ".tokenizer_cache"; +const TINYLLAMA_TOKENIZER_FILENAME: &str = "tinyllama_tokenizer.json"; + +// Global mutex to prevent concurrent downloads +static DOWNLOAD_MUTEX: OnceLock> = OnceLock::new(); + +/// Downloads the TinyLlama tokenizer from HuggingFace if not already cached. +/// Returns the path to the cached tokenizer file. +/// +/// This function is thread-safe and will only download the tokenizer once +/// even if called from multiple threads concurrently. +pub fn ensure_tokenizer_cached() -> PathBuf { + // Get or initialize the mutex + let mutex = DOWNLOAD_MUTEX.get_or_init(|| Mutex::new(())); + + // Lock to ensure only one thread downloads at a time + let _guard = mutex.lock().unwrap(); + + let cache_dir = PathBuf::from(CACHE_DIR); + let tokenizer_path = cache_dir.join(TINYLLAMA_TOKENIZER_FILENAME); + + // Create cache directory if it doesn't exist + if !cache_dir.exists() { + fs::create_dir_all(&cache_dir).expect("Failed to create cache directory"); + } + + // Download tokenizer if not already cached + if !tokenizer_path.exists() { + println!("Downloading TinyLlama tokenizer from HuggingFace..."); + + // Use blocking reqwest client since we're in tests/benchmarks + let client = reqwest::blocking::Client::new(); + let response = client + .get(TINYLLAMA_TOKENIZER_URL) + .send() + .expect("Failed to download tokenizer"); + + if !response.status().is_success() { + panic!("Failed to download tokenizer: HTTP {}", response.status()); + } + + let content = response.bytes().expect("Failed to read tokenizer content"); + + if content.len() < 100 { + panic!("Downloaded content too small: {} bytes", content.len()); + } + + fs::write(&tokenizer_path, content).expect("Failed to write tokenizer to cache"); + println!( + "Tokenizer downloaded and cached successfully ({} bytes)", + tokenizer_path.metadata().unwrap().len() + ); + } + + tokenizer_path +} + +/// Common test prompts for consistency across tests +pub const TEST_PROMPTS: [&str; 4] = [ + "deep learning is", + "Deep learning is", + "has anyone seen nemo lately", + "another prompt", +]; + +/// Pre-computed hashes for verification +pub const EXPECTED_HASHES: [u64; 4] = [ + 1209591529327510910, + 4181375434596349981, + 6245658446118930933, + 5097285695902185237, +]; diff --git a/tests/tokenizer/tokenizer_cache_correctness_test.rs b/tokenizer/tests/tokenizer_cache_correctness_test.rs similarity index 99% rename from tests/tokenizer/tokenizer_cache_correctness_test.rs rename to tokenizer/tests/tokenizer_cache_correctness_test.rs index ba75696aab..597c28b993 100644 --- a/tests/tokenizer/tokenizer_cache_correctness_test.rs +++ b/tokenizer/tests/tokenizer_cache_correctness_test.rs @@ -9,7 +9,7 @@ use std::{ sync::{Arc, OnceLock}, }; -use smg::tokenizer::{ +use llm_tokenizer::{ cache::{CacheConfig, CachedTokenizer}, hub::download_tokenizer_from_hf, huggingface::HuggingFaceTokenizer, diff --git a/tests/tokenizer/tokenizer_integration.rs b/tokenizer/tests/tokenizer_integration.rs similarity index 98% rename from tests/tokenizer/tokenizer_integration.rs rename to tokenizer/tests/tokenizer_integration.rs index 89572652f7..5d7a3fd7b1 100644 --- a/tests/tokenizer/tokenizer_integration.rs +++ b/tokenizer/tests/tokenizer_integration.rs @@ -3,15 +3,16 @@ //! These tests download the TinyLlama tokenizer from HuggingFace to verify our tokenizer //! implementation works correctly with real-world tokenizer files. +mod common; + use std::sync::Arc; -use smg::tokenizer::{ +use common::{ensure_tokenizer_cached, EXPECTED_HASHES, TEST_PROMPTS}; +use llm_tokenizer::{ factory, huggingface::HuggingFaceTokenizer, sequence::Sequence, stop::*, stream::DecodeStream, traits::*, }; -use crate::common::{ensure_tokenizer_cached, EXPECTED_HASHES, TEST_PROMPTS}; - const LONG_TEST_PROMPTS: [(&str, &str); 6] = [ ("Tell me about the following text.", "Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat."), ("Tell me about the following text.", "Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur. Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit anim id est laborum."), @@ -279,7 +280,7 @@ fn test_batch_encoding() { #[test] fn test_special_tokens() { - use smg::tokenizer::traits::Tokenizer as TokenizerTrait; + use llm_tokenizer::traits::Tokenizer as TokenizerTrait; let tokenizer_path = ensure_tokenizer_cached(); let tokenizer = HuggingFaceTokenizer::from_file(tokenizer_path.to_str().unwrap()) @@ -408,7 +409,7 @@ fn test_load_chat_template_from_local_file() { #[tokio::test] async fn test_tinyllama_embedded_template() { - use smg::tokenizer::hub::download_tokenizer_from_hf; + use llm_tokenizer::hub::download_tokenizer_from_hf; // Skip in CI without HF_TOKEN @@ -444,7 +445,7 @@ async fn test_tinyllama_embedded_template() { #[tokio::test] async fn test_qwen3_next_embedded_template() { - use smg::tokenizer::hub::download_tokenizer_from_hf; + use llm_tokenizer::hub::download_tokenizer_from_hf; // Test 3: Qwen3-Next has chat template in tokenizer_config.json match download_tokenizer_from_hf("Qwen/Qwen3-Next-80B-A3B-Instruct").await { @@ -476,7 +477,7 @@ async fn test_qwen3_next_embedded_template() { #[tokio::test] async fn test_qwen3_vl_json_template_priority() { - use smg::tokenizer::hub::download_tokenizer_from_hf; + use llm_tokenizer::hub::download_tokenizer_from_hf; // Test 4: Qwen3-VL has both tokenizer_config.json template and chat_template.json // Should prioritize chat_template.json @@ -518,7 +519,7 @@ async fn test_qwen3_vl_json_template_priority() { #[tokio::test] async fn test_llava_separate_jinja_template() { - use smg::tokenizer::hub::download_tokenizer_from_hf; + use llm_tokenizer::hub::download_tokenizer_from_hf; // Test 5: llava has chat_template.jinja as a separate file, not in tokenizer_config.json match download_tokenizer_from_hf("llava-hf/llava-1.5-7b-hf").await { From 52936f1f0f6917ecd18ab83b2b189de5caa8fc05 Mon Sep 17 00:00:00 2001 From: Simo Lin Date: Sun, 18 Jan 2026 21:13:08 -0800 Subject: [PATCH 2/2] fix: Replace glob imports with explicit imports in tokenizer tests Re-enabled internal tests that were disabled during crate extraction. Fixed all `use crate::*` glob imports by replacing them with explicit imports, including proper trait imports for Encoder, Decoder, and Tokenizer. --- src/routers/tokenize/handlers.rs | 37 ------ tokenizer/README.md | 205 +++++++++++++++++++++++++++++++ tokenizer/src/factory.rs | 4 +- tokenizer/src/hub.rs | 2 +- tokenizer/src/lib.rs | 5 +- tokenizer/src/registry.rs | 4 +- tokenizer/src/stop.rs | 7 +- tokenizer/src/tests.rs | 8 +- tokenizer/src/tiktoken.rs | 3 +- 9 files changed, 226 insertions(+), 49 deletions(-) create mode 100644 tokenizer/README.md diff --git a/src/routers/tokenize/handlers.rs b/src/routers/tokenize/handlers.rs index 8698565b24..e3aa819a19 100644 --- a/src/routers/tokenize/handlers.rs +++ b/src/routers/tokenize/handlers.rs @@ -396,43 +396,6 @@ pub async fn get_tokenizer_status(context: &Arc, tokenizer_id: &str) #[cfg(test)] mod tests { use super::*; - use crate::tokenizer::mock::MockTokenizer; - - fn create_test_registry() -> Arc { - let registry = Arc::new(TokenizerRegistry::new()); - let id = TokenizerRegistry::generate_id(); - registry.register( - &id, - "test-model", - "test-source", - Arc::new(MockTokenizer::new()), - ); - registry - } - - #[test] - fn test_get_tokenizer_exact_match() { - let registry = create_test_registry(); - let result = get_tokenizer(®istry, "test-model"); - assert!(result.is_ok()); - } - - #[test] - fn test_get_tokenizer_unknown_model_fallback() { - let registry = create_test_registry(); - let result = get_tokenizer(®istry, UNKNOWN_MODEL_ID); - assert!(result.is_ok()); - } - - #[test] - fn test_get_tokenizer_not_found() { - let registry = create_test_registry(); - let result = get_tokenizer(®istry, "nonexistent"); - match result { - Err(e) => assert!(e.contains("not found")), - Ok(_) => panic!("Expected error"), - } - } #[test] fn test_get_tokenizer_empty_registry() { diff --git a/tokenizer/README.md b/tokenizer/README.md new file mode 100644 index 0000000000..1254c112d0 --- /dev/null +++ b/tokenizer/README.md @@ -0,0 +1,205 @@ +# llm-tokenizer + +## Overview +The `llm-tokenizer` crate exposes a single `Tokenizer` facade around multiple backends +(Hugging Face JSON tokenizers, OpenAI/tiktoken models, and an in-memory mock). It packages the +shared behaviours needed by LLM applications—encoding user text, incrementally decoding streamed tokens, +tracking per-request state, and detecting stop conditions—behind trait objects so consuming code +can remain backend-agnostic. + +Key capabilities: +- trait-based split between `Encoder`, `Decoder`, and `Tokenizer` for shared APIs across backends +- Hugging Face tokenizer loading (with optional chat templates) and HF Hub downloads +- heuristic selection of OpenAI/tiktoken encodings for GPT model names +- incremental decoding utilities (`DecodeStream`, `Sequence`) that handle UTF-8 boundaries +- stop sequence handling via `StopSequenceDecoder` with token-level and string-level triggers +- optional Jinja2 chat-template rendering that matches Hugging Face semantics + +The implementation deliberately keeps the surface area small—metrics, batching, or SentencePiece +support mentioned in earlier drafts do **not** exist today. This document reflects the actual code +as of `tokenizer/src/*`. + +## Source Map +- `lib.rs` – module exports and the `Tokenizer` wrapper around `Arc` +- `traits.rs` – shared traits and the `Encoding`/`SpecialTokens` helper types +- `factory.rs` – backend discovery, file/model heuristics, and tokio-aware creation helpers +- `hub.rs` – Hugging Face Hub downloads via `hf_hub` +- `huggingface.rs` – wrapper over `tokenizers::Tokenizer`, chat template loading, vocab access +- `tiktoken.rs` – wrapper over `tiktoken-rs` encoders for OpenAI model families +- `chat_template.rs` – AST-driven Jinja template inspection and rendering utilities +- `sequence.rs` – stateful incremental decoding helper used by router sequences +- `stream.rs` – stateless streaming decoder that yields textual chunks from token streams +- `stop.rs` – stop-sequence detection with "jail" buffering and a builder API +- `mock.rs` – lightweight tokenizer used by unit tests +- `tests.rs` – smoke tests covering the trait facade and helpers (largely with the mock backend) +- `cache/` – multi-level caching infrastructure (L0 in-memory, L1 prefix-based) + +## Core Traits and Types (`traits.rs`) +- `Encoder`, `Decoder`, and `Tokenizer` traits stay `Send + Sync` so instances can be shared across + threads. Concrete backends implement the minimal methods: `encode`, `encode_batch`, `decode`, + `vocab_size`, special-token lookup, and optional token↔id conversions. +- `Encoding` wraps backend-specific results: `Hf` holds the Hugging Face encoding object, + `Sp` is a plain ID vector reserved for future SentencePiece support, and `Tiktoken` stores u32 IDs + from `tiktoken-rs`. `Encoding::token_ids()` is the zero-copy accessor used everywhere. +- `SpecialTokens` collects optional BOS/EOS/etc. markers so upstream code can make backend-agnostic + decisions. +- `Tokenizer` (in `lib.rs`) is a thin `Arc` newtype that exposes convenience methods + (`encode`, `decode`, `decode_stream`, etc.) while keeping cloning cheap. + +## Backend Implementations +### HuggingFaceTokenizer (`huggingface.rs`) +- Loads `tokenizer.json` (or similar) using `tokenizers::Tokenizer::from_file`. +- Caches vocab forward and reverse maps for `token_to_id`/`id_to_token` support. +- Extracts special tokens using common patterns (e.g. ``, `[CLS]`). +- Supports optional chat templates: either auto-discovered next to the tokenizer via + `tokenizer_config.json` or overridable with an explicit template path. +- Exposes `apply_chat_template` which renders a minijinja template given JSON message payloads and + template parameters. + +### TiktokenTokenizer (`tiktoken.rs`) +- Wraps the `tiktoken-rs` `CoreBPE` builders (`cl100k_base`, `p50k_base`, `p50k_edit`, `r50k_base`). +- `from_model_name` heuristically maps OpenAI model IDs (e.g. `gpt-4`, `text-davinci-003`) to those + bases. Unknown model names return an error rather than silently defaulting. +- Implements encode/decode operations; batch encode simply iterates sequentially. +- Provides approximate vocab sizes and common GPT special tokens. Direct token↔id lookup is not + implemented—the underlying library does not expose that mapping. + +### MockTokenizer (`mock.rs`) +- Purely for tests; hard-codes a tiny vocabulary and simple whitespace tokenization. +- Implements the same trait surface so helpers can be exercised without pulling real tokenizer data. + +## Factory and Backend Discovery (`factory.rs`) +- `create_tokenizer{,_async}` accept either a filesystem path or a model identifier. Logic: + 1. Paths are loaded directly; the file extension (or JSON autodetection) selects the backend. + 2. Strings that look like OpenAI model names (`gpt-*`, `davinci`, `curie`, `babbage`, `ada`) use + `TiktokenTokenizer`. + 3. Everything else attempts a Hugging Face Hub download via `download_tokenizer_from_hf`. +- Chat templates can be injected with `create_tokenizer_with_chat_template`. +- Async creation uses `tokio` for network access. The blocking variant reuses or spins up a runtime + when called from synchronous contexts. +- SentencePiece (`.model`) and GGUF files are detected but currently return a clear `not supported` + error. + +## Hugging Face Hub Integration (`hub.rs`) +- Uses the async `hf_hub` API to list and download tokenizer-related files + (`tokenizer.json`, `merges.txt`, `.model`, etc.), filtering out weights and docs. +- The helper returns the HF cache directory containing the fetched files; the factory then loads + from disk using standard file paths. +- Honour the `HF_TOKEN` environment variable for private or rate-limited models. Without it the + download may fail with an authorization error. + +## Chat Template Support (`chat_template.rs`) +- Detects whether a template expects raw string content or the structured OpenAI-style `content` + list by walking the minijinja AST. This matches the Python-side detection logic used elsewhere in + SGLang. +- `ChatTemplateProcessor` (constructed per call) renders templates against JSON `messages` and + `ChatTemplateParams` (system prompt, tools, EOS token handling, etc.). Errors surface as + `anyhow::Error`, keeping parity with Hugging Face error messages. +- The tokenizer wrapper stores both the template string and its detected content format so callers + can pre-transform message content correctly. + +## Streaming and Stateful Helpers +### `DecodeStream` (`stream.rs`) +- Maintains a sliding window (`prefix_offset`, `read_offset`) over accumulated token IDs. +- Each `step` decodes the known prefix and the new slice; when the new slice produces additional + UTF-8 text (and does not end in the replacement character `�`), it returns the incremental chunk + and updates offsets. Otherwise it returns `None` and waits for more tokens. +- `step_batch` and `flush` offer convenience for batching and draining remaining text. + +### `Sequence` (`sequence.rs`) +- Holds per-request decoding state: accumulated IDs plus offsets mirroring `DecodeStream`. +- `append_text` encodes extra prompt text; `append_token` decodes incremental output while + respecting UTF-8 boundaries and replacing stray `�` characters. +- Designed for integration with router sequence management where decoded text must be replayed. + +### `StopSequenceDecoder` (`stop.rs`) +- Extends the incremental decoding approach with a "jail" buffer that holds potential partial + matches against configured stop sequences. +- Supports both token-level stops (visible or hidden) and arbitrary string sequences. When a string + stop is configured, the decoder emits only the safe prefix and keeps a suffix jailed until it can + decide whether it completes a stop sequence. +- Provides `StopSequenceDecoderBuilder` for ergonomic configuration and exposes `process_token`, + `process_tokens`, `flush`, `reset`, and `is_stopped` helpers. + +## Caching (`cache/`) +The caching subsystem provides multi-level caching for tokenizer results: +- `L0Cache`: In-memory LRU cache for exact-match token ID lookups +- `L1Cache`: Prefix-based cache that can reuse partial encoding results +- `CachedTokenizer`: Wrapper that adds caching to any tokenizer implementation +- `TokenizerFingerprint`: Content-based fingerprinting for cache key generation + +## Testing +- Unit tests cover the mock tokenizer, the `Tokenizer` wrapper, incremental decoding helpers, and + stop-sequence behaviour (`tests.rs`, `sequence.rs`, `stop.rs`, `tiktoken.rs`, `factory.rs`, + `hub.rs`). Network-dependent Hugging Face downloads are exercised behind a best-effort async test + that skips in CI without credentials. +- Use `cargo test -p tokenizer` to run the crate's test suite. + +## Known Limitations & Future Work +- SentencePiece (`.model`) and GGUF tokenizers are detected but deliberately unimplemented. +- `Encoding::Sp` exists for future SentencePiece support but currently behaves as a simple `Vec`. +- `TiktokenTokenizer` cannot map individual tokens/IDs; the underlying library would need to expose + its vocabulary to implement `token_to_id`/`id_to_token`. +- There is no metrics or batching layer inside this module; the router records metrics elsewhere. +- Dynamic batching / sequence pooling code that earlier READMEs mentioned never landed in Rust. + +## Usage Examples +```rust +use std::sync::Arc; +use llm_tokenizer::{ + create_tokenizer, SequenceDecoderOutput, StopSequenceDecoderBuilder, Tokenizer, +}; + +// Load a tokenizer from disk (Hugging Face JSON) +let tokenizer = Tokenizer::from_file("/path/to/tokenizer.json")?; +let encoding = tokenizer.encode("Hello, world!", false)?; +assert!(!encoding.token_ids().is_empty()); + +// Auto-detect OpenAI GPT tokenizer +let openai = create_tokenizer("gpt-4")?; +let text = openai.decode(&[1, 2, 3], true)?; + +// Incremental decoding with stop sequences +let mut stream = tokenizer.decode_stream(&[], true); +let mut stop = StopSequenceDecoderBuilder::new(Arc::clone(&tokenizer)) + .stop_sequence("\nHuman:") + .build(); +for &token in encoding.token_ids() { + if let Some(chunk) = stream.step(token)? { + match stop.process_token(token)? { + SequenceDecoderOutput::Text(t) => println!("{}", t), + SequenceDecoderOutput::StoppedWithText(t) => { + println!("{}", t); + break; + } + SequenceDecoderOutput::Held | SequenceDecoderOutput::Stopped => {} + } + } +} +``` + +```rust +// Apply a chat template when one is bundled with the tokenizer +use llm_tokenizer::{chat_template::ChatTemplateParams, HuggingFaceTokenizer}; + +let mut hf = HuggingFaceTokenizer::from_file_with_chat_template( + "./tokenizer.json", + Some("./chat_template.jinja"), +)?; +let messages = vec![ + serde_json::json!({"role": "system", "content": "You are concise."}), + serde_json::json!({"role": "user", "content": "Summarise Rust traits."}), +]; +let prompt = hf.apply_chat_template( + &messages, + ChatTemplateParams { + add_generation_prompt: true, + continue_final_message: false, + tools: None, + documents: None, + template_kwargs: None, + }, +)?; +``` + +Set `HF_TOKEN` in the environment if you need to download private models from the Hugging Face Hub. diff --git a/tokenizer/src/factory.rs b/tokenizer/src/factory.rs index ae1ace7185..eee6bed4d9 100644 --- a/tokenizer/src/factory.rs +++ b/tokenizer/src/factory.rs @@ -422,7 +422,9 @@ pub fn get_tokenizer_info(file_path: &str) -> Result { #[cfg(test)] mod tests { - use crate::*; + use super::{ + create_tokenizer, create_tokenizer_async, create_tokenizer_from_file, is_likely_json, + }; #[test] fn test_json_detection() { diff --git a/tokenizer/src/hub.rs b/tokenizer/src/hub.rs index 5fa62c68d8..c83a930cb1 100644 --- a/tokenizer/src/hub.rs +++ b/tokenizer/src/hub.rs @@ -281,7 +281,7 @@ fn resolve_model_cache_dir(path: &Path, model_name: &str) -> PathBuf { #[cfg(test)] mod tests { - use crate::*; + use super::{is_chat_template_file, is_tokenizer_file, is_weight_file}; #[test] fn test_is_tokenizer_file() { diff --git a/tokenizer/src/lib.rs b/tokenizer/src/lib.rs index 64c9878aa1..eeb00ca4b7 100644 --- a/tokenizer/src/lib.rs +++ b/tokenizer/src/lib.rs @@ -20,9 +20,8 @@ pub mod huggingface; pub mod tiktoken; -// TODO: Fix tests after crate extraction - many use glob imports that need updating -// #[cfg(test)] -// mod tests; +#[cfg(test)] +mod tests; // Re-export types used outside this module pub use cache::{CacheConfig, CacheStats, CachedTokenizer, L0Cache, L1Cache, TokenizerFingerprint}; diff --git a/tokenizer/src/registry.rs b/tokenizer/src/registry.rs index 592cad1166..f99f6b48a5 100644 --- a/tokenizer/src/registry.rs +++ b/tokenizer/src/registry.rs @@ -342,11 +342,11 @@ impl Default for TokenizerRegistry { #[cfg(test)] mod tests { - use std::time::Duration; + use std::{sync::Arc, time::Duration}; use tokio::time::sleep; - use crate::{mock::MockTokenizer, *}; + use crate::{mock::MockTokenizer, traits::Tokenizer, LoadError, TokenizerRegistry}; #[tokio::test] async fn test_basic_operations() { diff --git a/tokenizer/src/stop.rs b/tokenizer/src/stop.rs index 874a9ee20d..0514d34ad4 100644 --- a/tokenizer/src/stop.rs +++ b/tokenizer/src/stop.rs @@ -291,7 +291,12 @@ impl StopSequenceDecoderBuilder { #[cfg(test)] mod tests { - use crate::{mock::MockTokenizer, *}; + use std::sync::Arc; + + use super::StopSequenceDecoderBuilder; + use crate::{ + mock::MockTokenizer, SequenceDecoderOutput, StopSequenceConfig, StopSequenceDecoder, + }; #[test] fn test_stop_token_detection() { diff --git a/tokenizer/src/tests.rs b/tokenizer/src/tests.rs index acbf3b2248..8188c3ac4d 100644 --- a/tokenizer/src/tests.rs +++ b/tokenizer/src/tests.rs @@ -1,8 +1,10 @@ -#[cfg(test)] use std::sync::Arc; -#[cfg(test)] -use crate::*; +use crate::{ + mock, + traits::{Decoder, Encoder}, + Tokenizer, +}; #[test] fn test_mock_tokenizer_encode() { diff --git a/tokenizer/src/tiktoken.rs b/tokenizer/src/tiktoken.rs index 67e4b1b279..5428672162 100644 --- a/tokenizer/src/tiktoken.rs +++ b/tokenizer/src/tiktoken.rs @@ -183,7 +183,8 @@ impl TokenizerTrait for TiktokenTokenizer { #[cfg(test)] mod tests { - use crate::*; + use super::{TiktokenModel, TiktokenTokenizer}; + use crate::traits::{Decoder, Encoder, Tokenizer}; #[test] fn test_tiktoken_creation() {