Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion Cargo.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
[workspace]
members = ["protocols", "reasoning-parser", "tool-parser", "workflow"]
members = ["protocols", "reasoning-parser", "tool-parser", "workflow", "tokenizer"]
exclude = ["bindings/python"]
resolver = "2"

Expand Down Expand Up @@ -123,6 +123,7 @@ openai-protocol = { path = "protocols", features = ["axum"] }
reasoning-parser = { path = "reasoning-parser" }
tool-parser = { path = "tool-parser" }
wfaas = { path = "workflow", package = "workflow" }
llm-tokenizer = { path = "tokenizer", package = "tokenizer" }

# gRPC and Protobuf dependencies
tonic = { version = "0.14.2", features = ["gzip", "transport"] }
Expand Down
2 changes: 1 addition & 1 deletion src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@ pub use reasoning_parser;
pub mod routers;
pub mod server;
pub mod service_discovery;
pub mod tokenizer;
pub use llm_tokenizer as tokenizer;
pub use tool_parser;
pub mod version;
pub mod wasm;
Expand Down
37 changes: 0 additions & 37 deletions src/routers/tokenize/handlers.rs
Original file line number Diff line number Diff line change
Expand Up @@ -396,43 +396,6 @@ pub async fn get_tokenizer_status(context: &Arc<AppContext>, tokenizer_id: &str)
#[cfg(test)]
mod tests {
use super::*;
use crate::tokenizer::mock::MockTokenizer;

fn create_test_registry() -> Arc<TokenizerRegistry> {
let registry = Arc::new(TokenizerRegistry::new());
let id = TokenizerRegistry::generate_id();
registry.register(
&id,
"test-model",
"test-source",
Arc::new(MockTokenizer::new()),
);
registry
}

#[test]
fn test_get_tokenizer_exact_match() {
let registry = create_test_registry();
let result = get_tokenizer(&registry, "test-model");
assert!(result.is_ok());
}

#[test]
fn test_get_tokenizer_unknown_model_fallback() {
let registry = create_test_registry();
let result = get_tokenizer(&registry, UNKNOWN_MODEL_ID);
assert!(result.is_ok());
}

#[test]
fn test_get_tokenizer_not_found() {
let registry = create_test_registry();
let result = get_tokenizer(&registry, "nonexistent");
match result {
Err(e) => assert!(e.contains("not found")),
Ok(_) => panic!("Expected error"),
}
}

#[test]
fn test_get_tokenizer_empty_registry() {
Expand Down
7 changes: 0 additions & 7 deletions tests/tokenizer/mod.rs

This file was deleted.

6 changes: 0 additions & 6 deletions tests/tokenizer_tests.rs

This file was deleted.

37 changes: 37 additions & 0 deletions tokenizer/Cargo.toml
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
[package]
name = "tokenizer"
version = "0.1.0"
edition = "2021"
description = "LLM tokenizer library with caching and chat template support"
license = "Apache-2.0"
repository = "https://github.com/lightseekorg/smg"
keywords = ["tokenizer", "llm", "huggingface", "tiktoken", "chat-template"]
categories = ["text-processing", "parsing"]

[lib]
name = "llm_tokenizer"

[dependencies]
anyhow = "1.0"
blake3 = "1.5"
bytemuck = { version = "1.21", features = ["derive"] }
dashmap = "6.1.0"
hf-hub = { version = "0.4.3", features = ["tokio"] }
lru = "0.16.2"
minijinja = { version = "2.0", features = ["unstable_machinery", "json", "builtins"] }
minijinja-contrib = { version = "2.0", features = ["pycompat"] }
parking_lot = "0.12.4"
rayon = "1.10"
serde = { version = "1.0", features = ["derive"] }
serde_json = "1.0"
thiserror = "2.0.12"
tiktoken-rs = "0.7.0"
tokenizers = "0.22.0"
tokio = { version = "1.42.0", features = ["sync", "rt-multi-thread", "macros"] }
tracing = "0.1"
uuid = { version = "1.10", features = ["v4", "serde"] }

[dev-dependencies]
openai-protocol.workspace = true
reqwest = { version = "0.12.8", features = ["blocking"] }
tempfile = "3.8"
56 changes: 32 additions & 24 deletions src/tokenizer/README.md → tokenizer/README.md
Original file line number Diff line number Diff line change
@@ -1,11 +1,11 @@
# Tokenizer Module
# llm-tokenizer

## Overview
The `sgl-model-gateway` tokenizer subsystem exposes a single `Tokenizer` facade around multiple backends
(Hugging Face JSON tokenizers, OpenAI/tiktoken models, and an in-memory mock). It packages the
shared behaviours needed by the router–encoding user text, incrementally decoding streamed tokens,
tracking per-request state, and detecting stop conditions—behind trait objects so the rest of the
router can remain backend-agnostic.
The `llm-tokenizer` crate exposes a single `Tokenizer` facade around multiple backends
(Hugging Face JSON tokenizers, OpenAI/tiktoken models, and an in-memory mock). It packages the
shared behaviours needed by LLM applications—encoding user text, incrementally decoding streamed tokens,
tracking per-request state, and detecting stop conditions—behind trait objects so consuming code
can remain backend-agnostic.

Key capabilities:
- trait-based split between `Encoder`, `Decoder`, and `Tokenizer` for shared APIs across backends
Expand All @@ -16,11 +16,11 @@ Key capabilities:
- optional Jinja2 chat-template rendering that matches Hugging Face semantics

The implementation deliberately keeps the surface area small—metrics, batching, or SentencePiece
support mentioned in earlier drafts do **not** exist today. This document reflects the actual code
as of `sgl-model-gateway/src/tokenizer/*`.
support mentioned in earlier drafts do **not** exist today. This document reflects the actual code
as of `tokenizer/src/*`.

## Source Map
- `mod.rs` – module exports and the `Tokenizer` wrapper around `Arc<dyn Tokenizer>`
- `lib.rs` – module exports and the `Tokenizer` wrapper around `Arc<dyn Tokenizer>`
- `traits.rs` – shared traits and the `Encoding`/`SpecialTokens` helper types
- `factory.rs` – backend discovery, file/model heuristics, and tokio-aware creation helpers
- `hub.rs` – Hugging Face Hub downloads via `hf_hub`
Expand All @@ -32,17 +32,18 @@ as of `sgl-model-gateway/src/tokenizer/*`.
- `stop.rs` – stop-sequence detection with "jail" buffering and a builder API
- `mock.rs` – lightweight tokenizer used by unit tests
- `tests.rs` – smoke tests covering the trait facade and helpers (largely with the mock backend)
- `cache/` – multi-level caching infrastructure (L0 in-memory, L1 prefix-based)

## Core Traits and Types (`traits.rs`)
- `Encoder`, `Decoder`, and `Tokenizer` traits stay `Send + Sync` so instances can be shared across
threads. Concrete backends implement the minimal methods: `encode`, `encode_batch`, `decode`,
threads. Concrete backends implement the minimal methods: `encode`, `encode_batch`, `decode`,
`vocab_size`, special-token lookup, and optional token↔id conversions.
- `Encoding` wraps backend-specific results: `Hf` holds the Hugging Face encoding object,
`Sp` is a plain ID vector reserved for future SentencePiece support, and `Tiktoken` stores u32 IDs
from `tiktoken-rs`. `Encoding::token_ids()` is the zero-copy accessor used everywhere.
from `tiktoken-rs`. `Encoding::token_ids()` is the zero-copy accessor used everywhere.
- `SpecialTokens` collects optional BOS/EOS/etc. markers so upstream code can make backend-agnostic
decisions.
- `Tokenizer` (in `mod.rs`) is a thin `Arc<dyn Tokenizer>` newtype that exposes convenience methods
- `Tokenizer` (in `lib.rs`) is a thin `Arc<dyn Tokenizer>` newtype that exposes convenience methods
(`encode`, `decode`, `decode_stream`, etc.) while keeping cloning cheap.

## Backend Implementations
Expand All @@ -60,15 +61,15 @@ as of `sgl-model-gateway/src/tokenizer/*`.
- `from_model_name` heuristically maps OpenAI model IDs (e.g. `gpt-4`, `text-davinci-003`) to those
bases. Unknown model names return an error rather than silently defaulting.
- Implements encode/decode operations; batch encode simply iterates sequentially.
- Provides approximate vocab sizes and common GPT special tokens. Direct token↔id lookup is not
- Provides approximate vocab sizes and common GPT special tokens. Direct token↔id lookup is not
implemented—the underlying library does not expose that mapping.

### MockTokenizer (`mock.rs`)
- Purely for tests; hard-codes a tiny vocabulary and simple whitespace tokenization.
- Implements the same trait surface so helpers can be exercised without pulling real tokenizer data.

## Factory and Backend Discovery (`factory.rs`)
- `create_tokenizer{,_async}` accept either a filesystem path or a model identifier. Logic:
- `create_tokenizer{,_async}` accept either a filesystem path or a model identifier. Logic:
1. Paths are loaded directly; the file extension (or JSON autodetection) selects the backend.
2. Strings that look like OpenAI model names (`gpt-*`, `davinci`, `curie`, `babbage`, `ada`) use
`TiktokenTokenizer`.
Expand All @@ -84,15 +85,15 @@ as of `sgl-model-gateway/src/tokenizer/*`.
(`tokenizer.json`, `merges.txt`, `.model`, etc.), filtering out weights and docs.
- The helper returns the HF cache directory containing the fetched files; the factory then loads
from disk using standard file paths.
- Honour the `HF_TOKEN` environment variable for private or rate-limited models. Without it the
- Honour the `HF_TOKEN` environment variable for private or rate-limited models. Without it the
download may fail with an authorization error.

## Chat Template Support (`chat_template.rs`)
- Detects whether a template expects raw string content or the structured OpenAI-style `content`
list by walking the minijinja AST. This matches the Python-side detection logic used elsewhere in
list by walking the minijinja AST. This matches the Python-side detection logic used elsewhere in
SGLang.
- `ChatTemplateProcessor` (constructed per call) renders templates against JSON `messages` and
`ChatTemplateParams` (system prompt, tools, EOS token handling, etc.). Errors surface as
`ChatTemplateParams` (system prompt, tools, EOS token handling, etc.). Errors surface as
`anyhow::Error`, keeping parity with Hugging Face error messages.
- The tokenizer wrapper stores both the template string and its detected content format so callers
can pre-transform message content correctly.
Expand All @@ -102,7 +103,7 @@ as of `sgl-model-gateway/src/tokenizer/*`.
- Maintains a sliding window (`prefix_offset`, `read_offset`) over accumulated token IDs.
- Each `step` decodes the known prefix and the new slice; when the new slice produces additional
UTF-8 text (and does not end in the replacement character `�`), it returns the incremental chunk
and updates offsets. Otherwise it returns `None` and waits for more tokens.
and updates offsets. Otherwise it returns `None` and waits for more tokens.
- `step_batch` and `flush` offer convenience for batching and draining remaining text.

### `Sequence` (`sequence.rs`)
Expand All @@ -114,18 +115,25 @@ as of `sgl-model-gateway/src/tokenizer/*`.
### `StopSequenceDecoder` (`stop.rs`)
- Extends the incremental decoding approach with a "jail" buffer that holds potential partial
matches against configured stop sequences.
- Supports both token-level stops (visible or hidden) and arbitrary string sequences. When a string
- Supports both token-level stops (visible or hidden) and arbitrary string sequences. When a string
stop is configured, the decoder emits only the safe prefix and keeps a suffix jailed until it can
decide whether it completes a stop sequence.
- Provides `StopSequenceDecoderBuilder` for ergonomic configuration and exposes `process_token`,
`process_tokens`, `flush`, `reset`, and `is_stopped` helpers.

## Caching (`cache/`)
The caching subsystem provides multi-level caching for tokenizer results:
- `L0Cache`: In-memory LRU cache for exact-match token ID lookups
- `L1Cache`: Prefix-based cache that can reuse partial encoding results
- `CachedTokenizer`: Wrapper that adds caching to any tokenizer implementation
- `TokenizerFingerprint`: Content-based fingerprinting for cache key generation

## Testing
- Unit tests cover the mock tokenizer, the `Tokenizer` wrapper, incremental decoding helpers, and
stop-sequence behaviour (`tests.rs`, `sequence.rs`, `stop.rs`, `tiktoken.rs`, `factory.rs`,
`hub.rs`). Network-dependent Hugging Face downloads are exercised behind a best-effort async test
`hub.rs`). Network-dependent Hugging Face downloads are exercised behind a best-effort async test
that skips in CI without credentials.
- Use `cargo test -p sgl-model-gateway tokenizer` to run the module’s test suite.
- Use `cargo test -p tokenizer` to run the crate's test suite.

## Known Limitations & Future Work
- SentencePiece (`.model`) and GGUF tokenizers are detected but deliberately unimplemented.
Expand All @@ -138,13 +146,13 @@ as of `sgl-model-gateway/src/tokenizer/*`.
## Usage Examples
```rust
use std::sync::Arc;
use smg::tokenizer::{
use llm_tokenizer::{
create_tokenizer, SequenceDecoderOutput, StopSequenceDecoderBuilder, Tokenizer,
};

// Load a tokenizer from disk (Hugging Face JSON)
let tokenizer = Tokenizer::from_file("/path/to/tokenizer.json")?;
let encoding = tokenizer.encode("Hello, world!")?;
let encoding = tokenizer.encode("Hello, world!", false)?;
assert!(!encoding.token_ids().is_empty());

// Auto-detect OpenAI GPT tokenizer
Expand Down Expand Up @@ -172,7 +180,7 @@ for &token in encoding.token_ids() {

```rust
// Apply a chat template when one is bundled with the tokenizer
use smg::tokenizer::{chat_template::ChatTemplateParams, HuggingFaceTokenizer};
use llm_tokenizer::{chat_template::ChatTemplateParams, HuggingFaceTokenizer};

let mut hf = HuggingFaceTokenizer::from_file_with_chat_template(
"./tokenizer.json",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@ use std::{
hash::{Hash, Hasher},
};

use super::super::traits::Tokenizer;
use crate::traits::Tokenizer;

/// A fingerprint of a tokenizer's configuration
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
Expand Down Expand Up @@ -77,8 +77,7 @@ impl TokenizerFingerprint {

#[cfg(test)]
mod tests {
use super::*;
use crate::tokenizer::mock::MockTokenizer;
use crate::{mock::MockTokenizer, *};

#[test]
fn test_fingerprint_equality() {
Expand Down
5 changes: 2 additions & 3 deletions src/tokenizer/cache/l0.rs → tokenizer/src/cache/l0.rs
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ use std::sync::{

use dashmap::DashMap;

use super::super::traits::Encoding;
use crate::traits::Encoding;

/// L0 cache implementation using DashMap for lock-free reads
/// Uses Arc<Encoding> internally to provide zero-copy cache hits
Expand Down Expand Up @@ -135,8 +135,7 @@ pub struct CacheStats {

#[cfg(test)]
mod tests {
use super::*;
use crate::tokenizer::traits::Encoding;
use crate::{traits::Encoding, *};

fn mock_encoding(tokens: Vec<u32>) -> Encoding {
Encoding::Sp(tokens)
Expand Down
5 changes: 2 additions & 3 deletions src/tokenizer/cache/l1.rs → tokenizer/src/cache/l1.rs
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ use std::{
use blake3;
use dashmap::DashMap;

use super::super::traits::TokenIdType;
use crate::traits::TokenIdType;

/// Hash type for cache keys
type Blake3Hash = [u8; 32];
Expand Down Expand Up @@ -357,8 +357,7 @@ pub struct L1CacheStats {

#[cfg(test)]
mod tests {
use super::*;
use crate::tokenizer::mock::MockTokenizer;
use crate::{mock::MockTokenizer, *};

#[test]
fn test_basic_prefix_match() {
Expand Down
5 changes: 2 additions & 3 deletions src/tokenizer/cache/mod.rs → tokenizer/src/cache/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -26,7 +26,7 @@ pub use l0::{CacheStats, L0Cache};
pub use l1::{L1Cache, L1CacheStats};
use rayon::prelude::*;

use super::traits::{Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer};
use crate::traits::{Decoder, Encoder, Encoding, SpecialTokens, TokenIdType, Tokenizer};

/// Configuration for the tokenizer cache
#[derive(Debug, Clone)]
Expand Down Expand Up @@ -272,8 +272,7 @@ impl Tokenizer for CachedTokenizer {

#[cfg(test)]
mod tests {
use super::*;
use crate::tokenizer::mock::MockTokenizer;
use crate::{mock::MockTokenizer, *};

#[test]
fn test_cache_hit() {
Expand Down
File renamed without changes.
10 changes: 7 additions & 3 deletions src/tokenizer/factory.rs → tokenizer/src/factory.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3,8 +3,10 @@ use std::{fs::File, io::Read, path::Path, sync::Arc};
use anyhow::{Error, Result};
use tracing::debug;

use super::{huggingface::HuggingFaceTokenizer, tiktoken::TiktokenTokenizer, traits};
use crate::tokenizer::hub::download_tokenizer_from_hf;
use crate::{
hub::download_tokenizer_from_hf, huggingface::HuggingFaceTokenizer,
tiktoken::TiktokenTokenizer, traits,
};

/// Represents the type of tokenizer being used
#[derive(Debug, Clone)]
Expand Down Expand Up @@ -420,7 +422,9 @@ pub fn get_tokenizer_info(file_path: &str) -> Result<TokenizerType> {

#[cfg(test)]
mod tests {
use super::*;
use super::{
create_tokenizer, create_tokenizer_async, create_tokenizer_from_file, is_likely_json,
};

#[test]
fn test_json_detection() {
Expand Down
2 changes: 1 addition & 1 deletion src/tokenizer/hub.rs → tokenizer/src/hub.rs
Original file line number Diff line number Diff line change
Expand Up @@ -281,7 +281,7 @@ fn resolve_model_cache_dir(path: &Path, model_name: &str) -> PathBuf {

#[cfg(test)]
mod tests {
use super::*;
use super::{is_chat_template_file, is_tokenizer_file, is_weight_file};

#[test]
fn test_is_tokenizer_file() {
Expand Down
Loading
Loading