kitoken
Fast and versatile tokenizer for language models, compatible with SentencePiece, Tokenizers, Tiktoken and more. Supports BPE, Unigram and WordPiece tokenization in JavaScript, Python and Rust.
Language: rust
Author: GitHub Repos (@github-repos)
0 stars · 0 views
Viewing path: .taplo.toml
Files
- small_tokens_xlnet_base_cased.txt (txt)
- utf8_tokens_gemma3.txt (txt)
- tests (txt)
- models (txt)
- tiktoken (txt)
- tokenizers (txt)
- sentencepiece (txt)
- tekken (txt)
- packages (txt)
- javascript (txt)
- data (txt)
- examples (txt)
- src (txt)
- cli (txt)
- src (txt)
- python (txt)
- ISSUE_TEMPLATE (txt)
- src (txt)
- benches (txt)
- .github (github)
- workflows (txt)
- data (txt)
- tiktoken (txt)
- tokenizers (txt)
- sentencepiece (txt)
- tekken (txt)
- src (txt)
- encoder (txt)
- convert (txt)
- config (txt)
- feature_request.md (md)
- bug_report.md (md)
- FUNDING.yml (yml)
- sqlpilot.json (json)
- hw_doc_summarizer.json (json)
- utf8_output_mistral03.txt (txt)
- bench_encode_gpt2.rs (rs)
- bench_encode_cl100k.rs (rs)
- Cargo.toml (toml)
- encode_gemma3_time.svg (image)
- web.html (html)
- package.json (json)
- lib.rs (rs)
- README.md (md)
- README.md (md)
- test.js (js)
- main.rs (rs)
- README.md (md)
- pyproject.toml (toml)
- Cargo.toml (toml)
- util.rs (rs)
- README.md (md)
- question.md (md)
- encode_llama4_throughput.svg (image)
- bench_encode_llama2.rs (rs)
- encode_gemma3_throughput.svg (image)
- bench_encode_xlnet.rs (rs)
- .rustfmt.toml (toml)
- encode_llama4_time.svg (image)
- test.py (py)
- PULL_REQUEST_TEMPLATE.md (md)
- publish.yml (yml)
- publish-python.yml (yml)
- tests.yml (yml)
- publish-javascript.yml (yml)
- small_tokens_cl100k_base.txt (txt)
- utf8_tokens_llama4.txt (txt)
- utf8_tokens_gpt2.txt (txt)
- test_convert_sentencepiece.rs (rs)
- small_tokens_o200k_base.txt (txt)
- utf8_tokens_mistral2512.txt (txt)
- test_convert_tokenizers.rs (rs)
- mixed_input.txt (txt)
- utf8_tokens_cl100k_base.txt (txt)
- small_tokens_llama4.txt (txt)
- test_convert_tiktoken.rs (rs)
- utf8_tokens_o200k_base.txt (txt)
- utf8_tokens_kimi_k2.txt (txt)
- small_tokens_kimi_k2.txt (txt)
- utf8_tokens_p50k_base.txt (txt)
- small_tokens_p50k_base.txt (txt)
- util.rs (rs)
- mixed_output_bert_base_cased.txt (txt)
- utf8_output_gpt_neox.txt (txt)
- utf8_tokens_llama33.txt (txt)
- utf8_tokens_gpt_oss.txt (txt)
- mixed_output_mistral01.txt (txt)
- mixed_output_hw_doc_summarizer.txt (txt)
- small_tokens_llama33.txt (txt)
- utf8_output_mpt.txt (txt)
- small_tokens_mistral2512.txt (txt)
- small_tokens_llama4.txt (txt)
- small_tokens_gemma3.txt (txt)
- small_tokens_llama32.txt (txt)
- small_output_hw_doc_summarizer.txt (txt)
- small_tokens_mistral03.txt (txt)
- utf8_tokens_gte.txt (txt)
- utf8_tokens_r1.txt (txt)
- utf8_tokens_modernbert.txt (txt)
- small_tokens_r1.txt (txt)
- small_output_qwen35.txt (txt)
- utf8_output_mistral01.txt (txt)
- small_tokens_qwen35.txt (txt)
- utf8_tokens_gemma4.txt (txt)
- utf8_tokens_bert_base_cased.txt (txt)
- utf8_tokens_mistral01.txt (txt)
- small_tokens_gpt_oss.txt (txt)
- utf8_tokens_mistral35.txt (txt)
- utf8_output_modernbert.txt (txt)
- mixed_tokens_sqlpilot.txt (txt)
- small_tokens_hw_doc_summarizer.txt (txt)
- small_tokens_bert_base_cased.txt (txt)
- mixed_output_xlnet_base_cased.txt (txt)
- utf8_tokens_llama2.txt (txt)
- utf8_tokens_llama4.txt (txt)
- mixed_output_sqlpilot.txt (txt)
- small_output_mistral03.txt (txt)
- small_tokens_mistral35.txt (txt)
- mixed_tokens_hw_doc_summarizer.txt (txt)
- utf8_tokens_xlnet_base_cased.txt (txt)
- utf8_output_gemma3.txt (txt)
- small_output_bert_base_cased.txt (txt)
- small_output_llama2.txt (txt)
- mixed_output_mistral03.txt (txt)
- utf8_output_qwen35.txt (txt)
- utf8_output_hw_doc_summarizer.txt (txt)
- utf8_tokens_llama32.txt (txt)
- small_output_sqlpilot.txt (txt)
- utf8_tokens_xlnet_base_cased.txt (txt)
- mixed_output_gpt_neox.txt (txt)
- small_output_mistral01.txt (txt)
- small_tokens_xlnet_base_cased.txt (txt)
- small_tokens_llama2.txt (txt)
- small_tokens_llama2.txt (txt)
- small_tokens_mistral01.txt (txt)
- utf8_tokens_clip_base.txt (txt)
- utf8_tokens_glm46.txt (txt)
- wordpiece.rs (rs)
- utf8_output_llama2.txt (txt)
- small_tokens_gpt_neox.txt (txt)
- small_output_mpt.txt (txt)
- small_tokens_gpt2.txt (txt)
- utf8_output_bert_base_cased.txt (txt)
- utf8_tokens_mistral03.txt (txt)
- small_tokens_glm46.txt (txt)
- mixed_tokens_xlnet_base_cased.txt (txt)
- utf8_tokens_mpt.txt (txt)
- small_output_gte.txt (txt)
- bytepair.rs (rs)
- utf8_tokens_sqlpilot.txt (txt)
- mixed_output_qwen35.txt (txt)
- utf8_tokens_hw_doc_summarizer.txt (txt)
- small_output_modernbert.txt (txt)
- small_tokens_mistral2410.txt (txt)
- utf8_output_clip_base.txt (txt)
- unigram.rs (rs)
- mixed_output_modernbert.txt (txt)
- small_tokens_mpt.txt (txt)
- utf8_output_mistral01.txt (txt)
- mixed_output_nai-t5.txt (txt)
- utf8_tokens_llama2.txt (txt)
- small_tokens_mistral03.txt (txt)
- utf8_tokens_gemma.txt (txt)
- small_output_nai-t5.txt (txt)
- utf8_tokens_mistral01.txt (txt)
- small_tokens_gemma3.txt (txt)
- mixed_output_xlnet_base_cased.txt (txt)
- small_output_gpt_neox.txt (txt)
- utf8_output_nai-t5.txt (txt)
- utf8_output_mistral03.txt (txt)
- utf8_output_gemma.txt (txt)
- mixed_tokens_nai-t5.txt (txt)
- utf8_tokens_gemma3.txt (txt)
- small_tokens_nai-t5.txt (txt)
- utf8_tokens_mistral35.txt (txt)
- utf8_tokens_mistral03.txt (txt)
- utf8_output_xlnet_base_cased.txt (txt)
- mixed_tokens_xlnet_base_cased.txt (txt)
- utf8_tokens_mistral2512.txt (txt)
- lib.rs (rs)
- test_convert_tekken.rs (rs)
- utf8_output_llama2.txt (txt)
- utf8_tokens_mistral2410.txt (txt)
- small_tokens_nerdstash.txt (txt)
- small_tokens_gemma.txt (txt)
- .taplo.toml (toml)
- small_tokens_mistral2512.txt (txt)
- small_tokens_mistral01.txt (txt)
- small_tokens_mistral2410.txt (txt)
- utf8_output_nerdstash.txt (txt)
- small_tokens_mistral35.txt (txt)
- small_input.txt (txt)
- utf8_tokens_nerdstash.txt (txt)
- util.rs (rs)
- split.rs (rs)
- web.rs (rs)
- config.rs (rs)
- decoder.rs (rs)
- vocab.rs (rs)
- normalization.rs (rs)
- serialization.rs (rs)
- decoding.rs (rs)
- convert.rs (rs)
- charsmap.rs (rs)
- processing.rs (rs)
- tiktoken.rs (rs)
- tekken.rs (rs)
- encoder.rs (rs)
- regex.rs (rs)
- definition.rs (rs)
- sentencepiece.rs (rs)
- ATTRIBUTION.md (markdown)
- utf8_tokens_nai-t5.txt (txt)
- utf8_tokens_gpt_neox.txt (txt)
- mixed_tokens_llama4.txt (txt)
- utf8_tokens_qwen35.txt (txt)
- utf8_output_sqlpilot.txt (txt)
- small_tokens_sqlpilot.txt (txt)
- Cargo.toml (toml)
- mixed_tokens_bert_base_cased.txt (txt)
- utf8_output_xlnet_base_cased.txt (txt)
- utf8_output_gemma4.txt (txt)
- small_output_xlnet_base_cased.txt (txt)
- utf8_input.txt (txt)
- utf8_output_gemma3.txt (txt)
- LICENSE (txt)
- tokenizers.rs (rs)
- utf8_output_gte.txt (txt)
- small_output_xlnet_base_cased.txt (txt)
- mixed_tokens_llama4.txt (txt)
- mixed_output_mpt.txt (txt)
- small_tokens_gemma4.txt (txt)
- lib.rs (rs)
- mixed_output_clip_base.txt (txt)
- mixed_tokens_r1.txt (txt)
- small_tokens_modernbert.txt (txt)
- small_output_clip_base.txt (txt)
- small_tokens_gte.txt (txt)
- utf8_tokens_mistral2410.txt (txt)
- small_tokens_clip_base.txt (txt)
- Cargo.toml (toml)