Spaces:
Running
Running
File size: 1,505 Bytes
1a6d0d2 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 | import gptTokenizer from "gpt-tokenizer";
/**
* Trims page excerpts to fit a shared token budget.
*
* Pages are served shortest-first, each taking at most an equal share of what
* is left, so a single long article cannot crowd out the others and whatever
* short pages leave unused rolls over to the ones that need it.
*
* This is a leaf module that imports nothing but the tokenizer: the worker and
* the synchronous fallback both run this same function, and neither drags
* the PubSub state - which reads `localStorage` at import time, absent in a
* worker - into the other's context.
*/
export function allocatePageExcerpts(
contents: string[],
tokenBudget: number,
): string[] {
const excerpts = contents.map(() => "");
const pending = contents
.map((content, index) => ({
index,
tokens: content.length > 0 ? gptTokenizer.encode(content) : [],
}))
.filter(({ tokens }) => tokens.length > 0)
.sort((a, b) => a.tokens.length - b.tokens.length);
let remainingBudget = Math.max(0, tokenBudget);
let remainingPages = pending.length;
for (const { index, tokens } of pending) {
const taken = Math.min(
tokens.length,
Math.floor(remainingBudget / remainingPages),
);
if (taken > 0) {
excerpts[index] =
taken === tokens.length
? contents[index]
: `${gptTokenizer.decode(tokens.slice(0, taken)).trimEnd()}โฆ`;
}
remainingBudget -= taken;
remainingPages--;
}
return excerpts;
}
|