mirror of
https://github.com/garrytan/gbrain.git
synced 2026-07-27 22:15:33 +00:00
Revert "fix(chunkers/code): tolerate tiktoken special tokens in estimateTokens (#2453)"
This reverts commit b7f70970c1.
This commit is contained in:
@@ -1235,25 +1235,7 @@ export function estimateTokens(text: string): number {
|
||||
tiktokenInitialized = true;
|
||||
}
|
||||
if (tiktokenEncoder) {
|
||||
try {
|
||||
return tiktokenEncoder.encode(text).length;
|
||||
} catch {
|
||||
// Code legitimately contains tiktoken special-token strings (e.g. CLIP/GPT
|
||||
// tokenizers embed the literal "<|endoftext|>"). The default encode() uses
|
||||
// disallowed_special='all' and THROWS on those, crashing reindex-code on
|
||||
// valid source files. For a token COUNT we don't need special-token
|
||||
// semantics: re-encode treating them as ordinary text (never throws),
|
||||
// heuristic only if even that fails.
|
||||
try {
|
||||
return (
|
||||
tiktokenEncoder as unknown as {
|
||||
encode: (s: string, allowed: string[], disallowed: string[]) => Uint32Array;
|
||||
}
|
||||
).encode(text, [], []).length;
|
||||
} catch {
|
||||
return Math.max(1, Math.ceil(text.length / 4));
|
||||
}
|
||||
}
|
||||
return tiktokenEncoder.encode(text).length;
|
||||
}
|
||||
return Math.max(1, Math.ceil(text.length / 4));
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user