32 references to GPT4
Microsoft.ML.Tokenizers.Tests (32)
TiktokenTests.cs (32)
46TestGPT4TokenizationEncoding(GPT4); 49Assert.True(GPT4 is TiktokenTokenizer); 50IReadOnlyDictionary<string, int>? specialTokens = (GPT4 as TiktokenTokenizer)!.SpecialTokens; 65Tokenizer tokenizer = TiktokenTokenizer.Create(tokenizerDataFileName, GPT4.PreTokenizer, null, specialTokens); 70tokenizer = TiktokenTokenizer.Create(stream, GPT4.PreTokenizer, null, specialTokens); 74tokenizer = await TiktokenTokenizer.CreateAsync(tokenizerDataFileName, GPT4.PreTokenizer, normalizer: null, specialTokens); 79tokenizer = await TiktokenTokenizer.CreateAsync(stream, GPT4.PreTokenizer, normalizer: null, specialTokens); 106yield return new object[] { GPT4, @"https://openaipublic.blob.core.windows.net/encodings/cl100k_base.tiktoken" }; 197IReadOnlyList<int> encoded = GPT4.EncodeToIds(text); 199Assert.Equal(text, GPT4.Decode(encoded)); 200TestDecodingWithSpan((GPT4 as TiktokenTokenizer)!, encoded.ToArray(), text); 202IReadOnlyList<EncodedToken> result = GPT4.EncodeToTokens(text, out string? normalizedText); 203int idsCount = GPT4.CountTokens(text); 240IReadOnlyList<int> encoded = GPT4.EncodeToIds(text); 242Assert.Equal(text, GPT4.Decode(encoded)); 243TestDecodingWithSpan((GPT4 as TiktokenTokenizer)!, encoded.ToArray(), text); 245IReadOnlyList<EncodedToken> result = GPT4.EncodeToTokens(text, out string? normalizedText); 250int idsCount = GPT4.CountTokens(text); 261IReadOnlyList<int> encoded = GPT4.EncodeToIds(text); 264IReadOnlyList<EncodedToken> result = GPT4.EncodeToTokens(text, out string? normalizedText); 265int idsCount = GPT4.CountTokens(text); 274IReadOnlyList<int> encoded = GPT4.EncodeToIds(text); 275int idsCount = GPT4.CountTokens(text); 277Assert.Equal(text, GPT4.Decode(encoded)); 278TestDecodingWithSpan((GPT4 as TiktokenTokenizer)!, encoded.ToArray(), text); 280IReadOnlyList<EncodedToken> result = GPT4.EncodeToTokens(text, out string? normalizedText); 612TestTokenizerEncodingForTokenizer(GPT4, text, expectedTokens, expectedOffsets, expectedIds); 733IReadOnlyList<EncodedToken> result = GPT4.EncodeToTokens(text, out _); 739Assert.Equal(expectedIds, GPT4.EncodeToIds(text)); 740Assert.Equal(expectedIds.Length, GPT4.CountTokens(text)); 744int length = GPT4.GetIndexByTokenCount(text, tokenCount, out _, out int count); 762int index = GPT4.GetIndexByTokenCountFromEnd(text, tokenCount, out _, out count);