Tokenizer for T5/Flan models, implemented from a Hugging Face tokenizer.json Unigram (SentencePiece) spec. It parses tokenizer.json, builds a pieces map and precompiled chars map, normalizes input, applies whitespace splitting and Metaspace pre-tokenization, then runs Viterbi segmentation to produce token ids ending with </s>.
using System;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Text;
using System.Text.Json;
namespace TextToAnimation.Editor.Inference.UniMate;
/// <summary>
/// SentencePiece-unigram tokenizer for flan-t5, read from the Hugging Face <c>tokenizer.json</c>: its precompiled
/// charsmap normalisation, whitespace split and Metaspace pre-tokenisation ("▁" word prefix),
/// Viterbi segmentation by piece log-probability, then <c></s></c> appended.
/// </summary>
public sealed class T5Tokenizer
{
public const int PadId = 0;
public const int EosId = 1;
public const int UnkId = 2;
const char Space = '▁';
readonly Dictionary<string, (int Id, double Score)> _pieces = new( StringComparer.Ordinal );
readonly int _maxPieceLength;
readonly double _unkScore;
public int VocabSize { get; }
public T5Tokenizer( string tokenizerJson )
{
using var doc = JsonDocument.Parse( tokenizerJson );
var model = doc.RootElement.GetProperty( "model" );
if ( model.GetProperty( "type" ).GetString() != "Unigram" )
throw new InvalidDataException( "tokenizer.json is not a Unigram tokenizer." );
var id = 0;
var min = double.MaxValue;
foreach ( var entry in model.GetProperty( "vocab" ).EnumerateArray() )
{
var piece = entry[0].GetString() ?? "";
var score = entry[1].GetDouble();
_pieces.TryAdd( piece, (id, score) );
min = Math.Min( min, score );
id++;
}
VocabSize = id;
_maxPieceLength = _pieces.Keys.Max( k => k.Length );
_unkScore = min - 10.0;
_charsMap = PrecompiledCharsMap.FromTokenizer( doc.RootElement );
}
readonly PrecompiledCharsMap _charsMap;
public static T5Tokenizer Load( string path ) => new( File.ReadAllText( path ) );
/// <summary>
/// Token ids for a text, ending with </s>, as transformers' T5Tokenizer gives them: the normalizer, then its
/// pre-tokenizers WhitespaceSplit (words between Unicode whitespace) and Metaspace (each word prefixed with "▁"
/// unless it starts with one, and split before every "▁"), then each piece segmented.
/// </summary>
public List<int> Encode( string text )
{
var ids = new List<int>();
var normalized = Normalize( text );
var start = -1;
for ( var i = 0; i <= normalized.Length; i++ )
{
if ( i < normalized.Length && !char.IsWhiteSpace( normalized[i] ) )
{
if ( start < 0 ) start = i;
continue;
}
if ( start >= 0 ) Metaspace( normalized[start..i], ids );
start = -1;
}
ids.Add( EosId );
return ids;
}
void Metaspace( string word, List<int> ids )
{
if ( word[0] != Space ) word = Space + word;
var from = 0;
for ( var i = 1; i <= word.Length; i++ )
{
if ( i < word.Length && word[i] != Space ) continue;
Segment( word[from..i], ids );
from = i;
}
}
/// <summary>tokenizer.json's normalizer: the SentencePiece precompiled charsmap (its space collapsing is moot once words are split on whitespace).</summary>
string Normalize( string text ) => _charsMap is null ? (text ?? "").Normalize( NormalizationForm.FormKC ) : _charsMap.Normalize( text ?? "" );
/// <summary>Viterbi: best-scoring segmentation of one word into vocabulary pieces.</summary>
void Segment( string word, List<int> ids )
{
var n = word.Length;
var best = new double[n + 1];
var from = new int[n + 1];
var piece = new int[n + 1];
for ( var i = 1; i <= n; i++ ) best[i] = double.NegativeInfinity;
for ( var end = 1; end <= n; end++ )
{
for ( var len = 1; len <= Math.Min( _maxPieceLength, end ); len++ )
{
var start = end - len;
if ( double.IsNegativeInfinity( best[start] ) ) continue;
if ( _pieces.TryGetValue( word.Substring( start, len ), out var p ) )
{
var score = best[start] + p.Score;
if ( score > best[end] ) { best[end] = score; from[end] = start; piece[end] = p.Id; }
}
}
if ( double.IsNegativeInfinity( best[end] ) )
{
// no piece ends here: consume one character as <unk>
best[end] = best[end - 1] + _unkScore;
from[end] = end - 1;
piece[end] = UnkId;
}
}
var result = new List<int>();
for ( var i = n; i > 0; i = from[i] ) result.Add( piece[i] );
result.Reverse();
// merge consecutive unknowns like SentencePiece
for ( var i = 0; i < result.Count; i++ )
if ( !(result[i] == UnkId && ids.Count > 0 && ids[^1] == UnkId && i > 0) ) ids.Add( result[i] );
}
}