Repository navigation
🔎 feat: Per-Search Model-Output Budget for Web Search Highlights #242
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
5083179
875a081
e0b0cf9
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,242 @@ | ||
| import type * as t from './types'; | ||
| import { formatResultsForLLM, resolveMaxLLMOutputChars } from './format'; | ||
|
|
||
| const makeOrganic = ( | ||
| link: string, | ||
| highlights: t.Highlight[] | ||
| ): t.ProcessedOrganic => ({ | ||
| link, | ||
| title: `Title for ${link}`, | ||
| snippet: `Snippet for ${link}`, | ||
| highlights, | ||
| }); | ||
|
|
||
| const highlight = (text: string, score = 0.9): t.Highlight => ({ text, score }); | ||
|
|
||
| const reference = (url: string, originalIndex = 0): t.UsedReferences[number] => ({ | ||
| type: 'link', | ||
| originalIndex, | ||
| reference: { originalUrl: url, title: 'Ref', text: 'ref' }, | ||
| }); | ||
|
|
||
| const countHighlightBlocks = (output: string): number => | ||
| (output.match(/### Highlight \d+/g) ?? []).length; | ||
|
|
||
| const OMISSION_MARKER = 'omitted to fit the context budget'; | ||
|
|
||
| describe('resolveMaxLLMOutputChars', () => { | ||
| const originalEnv = process.env.SEARCH_MAX_LLM_OUTPUT_CHARS; | ||
|
|
||
| afterEach(() => { | ||
| if (originalEnv == null) { | ||
| delete process.env.SEARCH_MAX_LLM_OUTPUT_CHARS; | ||
| } else { | ||
| process.env.SEARCH_MAX_LLM_OUTPUT_CHARS = originalEnv; | ||
| } | ||
| }); | ||
|
|
||
| test('falls back to the 50,000 char default when nothing is configured', () => { | ||
| delete process.env.SEARCH_MAX_LLM_OUTPUT_CHARS; | ||
| expect(resolveMaxLLMOutputChars()).toBe(50000); | ||
| expect(resolveMaxLLMOutputChars(0)).toBe(50000); | ||
| expect(resolveMaxLLMOutputChars(-100)).toBe(50000); | ||
| }); | ||
|
|
||
| test('honors the SEARCH_MAX_LLM_OUTPUT_CHARS env var', () => { | ||
| process.env.SEARCH_MAX_LLM_OUTPUT_CHARS = '777'; | ||
| expect(resolveMaxLLMOutputChars()).toBe(777); | ||
| expect(resolveMaxLLMOutputChars(0)).toBe(777); | ||
| }); | ||
|
|
||
| test('an explicit positive config value wins over env and default', () => { | ||
| process.env.SEARCH_MAX_LLM_OUTPUT_CHARS = '777'; | ||
| expect(resolveMaxLLMOutputChars(1234)).toBe(1234); | ||
| }); | ||
|
|
||
| test('ignores a non-numeric env var', () => { | ||
| process.env.SEARCH_MAX_LLM_OUTPUT_CHARS = 'not-a-number'; | ||
| expect(resolveMaxLLMOutputChars()).toBe(50000); | ||
| }); | ||
| }); | ||
|
|
||
| describe('formatResultsForLLM highlight budget', () => { | ||
| test('keeps whole highlights in relevance order until the budget is hit', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [ | ||
| makeOrganic('https://a.com', [highlight('A'.repeat(100))]), | ||
| makeOrganic('https://b.com', [highlight('B'.repeat(100))]), | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 100); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).not.toContain('B'.repeat(100)); | ||
| expect(countHighlightBlocks(output)).toBe(1); | ||
| expect(output).toContain('_[1 additional highlight omitted to fit the context budget'); | ||
| }); | ||
|
|
||
| test('truncates the boundary highlight when meaningful room remains', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [highlight('A'.repeat(1000))])], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 500); | ||
|
|
||
| expect(output).toContain('…[truncated]'); | ||
| expect(output).toContain('A'.repeat(500)); | ||
| expect(output).not.toContain('A'.repeat(501)); | ||
| expect(output).toContain('_[1 additional highlight omitted to fit the context budget'); | ||
| }); | ||
|
|
||
| test('drops the boundary highlight entirely when too little room remains', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [ | ||
| makeOrganic('https://a.com', [highlight('A'.repeat(100))]), | ||
| makeOrganic('https://b.com', [highlight('B'.repeat(100))]), | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 150); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).not.toContain('…[truncated]'); | ||
| expect(output).not.toContain('B'); | ||
| expect(countHighlightBlocks(output)).toBe(1); | ||
| }); | ||
|
|
||
| test('always keeps snippets, titles, and URLs even when all highlights are dropped', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [highlight('A'.repeat(100))])], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 10); | ||
|
|
||
| expect(output).toContain('URL: https://a.com'); | ||
| expect(output).toContain('Summary: Snippet for https://a.com'); | ||
| expect(output).toContain('"Title for https://a.com"'); | ||
| expect(countHighlightBlocks(output)).toBe(0); | ||
| expect(output).toContain('_[1 additional highlight omitted to fit the context budget'); | ||
| }); | ||
|
|
||
| test('emits no omission marker when every highlight fits the budget', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [ | ||
| makeOrganic('https://a.com', [highlight('A'.repeat(100))]), | ||
| makeOrganic('https://b.com', [highlight('B'.repeat(100))]), | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 50000); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).toContain('B'.repeat(100)); | ||
| expect(countHighlightBlocks(output)).toBe(2); | ||
| expect(output).not.toContain(OMISSION_MARKER); | ||
| }); | ||
|
|
||
| test('drops references with no surviving marker when truncating', () => { | ||
| const withRefs = highlight('A'.repeat(1000)); | ||
| withRefs.references = [reference('https://cited.example')]; | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [withRefs])], | ||
| }; | ||
|
|
||
| const { output, references } = formatResultsForLLM(0, results, 500); | ||
|
|
||
| expect(output).toContain('…[truncated]'); | ||
| expect(output).not.toContain('Core References'); | ||
| expect(output).not.toContain('https://cited.example'); | ||
| expect(references).toHaveLength(0); | ||
| }); | ||
|
|
||
| test('keeps references whose marker survives truncation and drops the rest', () => { | ||
| const withRefs = highlight(`(link#1) ${'A'.repeat(1000)} (link#2)`); | ||
| withRefs.references = [ | ||
| reference('https://one.example', 0), | ||
| reference('https://two.example', 1), | ||
| ]; | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [withRefs])], | ||
| }; | ||
|
|
||
| const { output, references } = formatResultsForLLM(0, results, 500); | ||
|
|
||
| expect(output).toContain('…[truncated]'); | ||
| expect(output).toContain('https://one.example'); | ||
| expect(output).not.toContain('https://two.example'); | ||
| expect(references).toHaveLength(1); | ||
| expect(references[0].link).toBe('https://one.example'); | ||
| }); | ||
|
|
||
| test('stops at the boundary highlight — no lower-ranked highlight slips in', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [ | ||
| makeOrganic('https://a.com', [ | ||
| highlight('A'.repeat(100), 0.9), | ||
| highlight('B'.repeat(300), 0.8), | ||
| highlight('C'.repeat(10), 0.7), | ||
| ]), | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 150); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).not.toContain('B'.repeat(300)); | ||
| expect(output).not.toContain('C'.repeat(10)); | ||
| expect(output).not.toContain('…[truncated]'); | ||
| expect(countHighlightBlocks(output)).toBe(1); | ||
| }); | ||
|
|
||
| test('keeps references on a whole highlight that fits the budget', () => { | ||
| const withRefs = highlight('A'.repeat(100)); | ||
| withRefs.references = [reference('https://cited.example')]; | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [withRefs])], | ||
| }; | ||
|
|
||
| const { output, references } = formatResultsForLLM(0, results, 50000); | ||
|
|
||
| expect(output).toContain('Core References'); | ||
| expect(references).toHaveLength(1); | ||
| expect(references[0].link).toBe('https://cited.example'); | ||
| }); | ||
|
|
||
| test('skips blank highlights instead of charging them against the budget', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [ | ||
| makeOrganic('https://a.com', [ | ||
| highlight(' \n\t '), | ||
| highlight('A'.repeat(100)), | ||
| ]), | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 100); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).not.toContain('…[truncated]'); | ||
| expect(countHighlightBlocks(output)).toBe(1); | ||
| expect(output).not.toContain(OMISSION_MARKER); | ||
| }); | ||
|
|
||
| test('spends the budget across organic results before news results', () => { | ||
| const results: t.SearchResultData = { | ||
| organic: [makeOrganic('https://a.com', [highlight('A'.repeat(100))])], | ||
| topStories: [ | ||
| { | ||
| link: 'https://news.com', | ||
| title: 'Story', | ||
| highlights: [highlight('N'.repeat(100))], | ||
| }, | ||
| ], | ||
| }; | ||
|
|
||
| const { output } = formatResultsForLLM(0, results, 100); | ||
|
|
||
| expect(output).toContain('A'.repeat(100)); | ||
| expect(output).not.toContain('N'.repeat(100)); | ||
| expect(output).toContain('_[1 additional highlight omitted to fit the context budget'); | ||
| }); | ||
| }); |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,6 +1,113 @@ | ||
| import type * as t from './types'; | ||
| import { getDomainName, fileExtRegex } from './utils'; | ||
|
|
||
| /** Default per-search budget for model-facing highlight content (chars). Hosts | ||
| * that know the context window (e.g. LibreChat) pass a window-relative value; | ||
| * this fixed fallback keeps standalone consumers bounded instead of dumping the | ||
| * full reranked content of every source into the prompt. */ | ||
| const DEFAULT_MAX_LLM_OUTPUT_CHARS = 50000; | ||
|
|
||
| /** Minimum room (chars) worth filling with a truncated boundary highlight; below | ||
| * this we drop it whole rather than emit a useless sliver. */ | ||
| const MIN_PARTIAL_HIGHLIGHT_CHARS = 200; | ||
|
|
||
| /** Resolves the per-search highlight budget from config, the | ||
| * `SEARCH_MAX_LLM_OUTPUT_CHARS` env var, or the default (50,000 chars). */ | ||
| export function resolveMaxLLMOutputChars(maxOutputChars?: number): number { | ||
| if (maxOutputChars != null && maxOutputChars > 0) { | ||
| return maxOutputChars; | ||
| } | ||
| const envValue = Number(process.env.SEARCH_MAX_LLM_OUTPUT_CHARS); | ||
| if (Number.isFinite(envValue) && envValue > 0) { | ||
| return envValue; | ||
| } | ||
| return DEFAULT_MAX_LLM_OUTPUT_CHARS; | ||
| } | ||
|
|
||
| /** Inline citation markers embedded in highlight text, e.g. `(link#2 "Title")`. | ||
| * Mirrors the matcher in `highlights.ts` so truncation can tell which citations | ||
| * survive in a sliced prefix. */ | ||
| const REFERENCE_MARKER_REGEX = /\((link|image|video)#(\d+)(?:\s+"[^"]*")?\)/g; | ||
|
|
||
| /** Builds the set of `type#originalIndex` keys whose complete citation marker | ||
| * appears in `text`, so references can be filtered to those still visible. */ | ||
| function visibleReferenceKeys(text: string): Set<string> { | ||
| const keys = new Set<string>(); | ||
| if (!text.includes('#')) { | ||
| return keys; | ||
| } | ||
| const regex = new RegExp(REFERENCE_MARKER_REGEX); | ||
| let match: RegExpExecArray | null; | ||
| while ((match = regex.exec(text)) !== null) { | ||
| keys.add(`${match[1]}#${parseInt(match[2], 10) - 1}`); | ||
| } | ||
| return keys; | ||
| } | ||
|
|
||
| /** Truncates a highlight to `maxLen` chars of (already-trimmed) text, keeping | ||
| * only the references whose markers survive in the kept prefix — markers in the | ||
| * cut tail would otherwise emit Core References for citations the model can no | ||
| * longer see, while a blanket drop would lose still-visible ones. */ | ||
| function truncateHighlight(highlight: t.Highlight, text: string, maxLen: number): t.Highlight { | ||
| const prefix = text.slice(0, maxLen); | ||
| const truncated: t.Highlight = { score: highlight.score, text: `${prefix}\n…[truncated]` }; | ||
| if (highlight.references != null && highlight.references.length > 0) { | ||
| const keys = visibleReferenceKeys(prefix); | ||
| const visible = highlight.references.filter((ref) => keys.has(`${ref.type}#${ref.originalIndex}`)); | ||
| if (visible.length > 0) { | ||
| truncated.references = visible; | ||
| } | ||
| } | ||
| return truncated; | ||
| } | ||
|
|
||
| /** Bounds the highlight chunks — the dominant, unbounded part of search output — | ||
| * to `maxChars`, walking sources in relevance order (organic first, then news; | ||
| * highlights in their reranked order). Whole highlights are kept until the | ||
| * budget is hit, the boundary one is truncated if meaningful room remains, and | ||
| * every later highlight is dropped (relevance-ordered prefix). Blank highlights | ||
| * are skipped (never rendered, so never charged); a truncated highlight keeps | ||
| * only references whose markers survive in the kept prefix. Snippets/titles/URLs | ||
| * are left untouched (small, high-signal) and per-source `content` stays in the | ||
| * `WEB_SEARCH` artifact for citations. Mutates `results` in place; returns how | ||
| * many highlights were dropped or truncated (0 when everything fit). */ | ||
| function trimHighlightsToBudget(results: t.SearchResultData, maxChars: number): number { | ||
| let used = 0; | ||
| let trimmed = 0; | ||
| const sections: (t.ValidSource[] | undefined)[] = [results.organic, results.topStories]; | ||
| for (const sources of sections) { | ||
| if (sources == null) { | ||
| continue; | ||
| } | ||
| for (const source of sources) { | ||
| const highlights = source.highlights; | ||
| if (highlights == null || highlights.length === 0) { | ||
| continue; | ||
| } | ||
| const kept: t.Highlight[] = []; | ||
| for (const highlight of highlights) { | ||
| const text = highlight.text.trim(); | ||
| if (text.length === 0) { | ||
| continue; | ||
| } | ||
| if (used + text.length <= maxChars) { | ||
| kept.push(highlight); | ||
| used += text.length; | ||
| continue; | ||
| } | ||
| const remaining = maxChars - used; | ||
| if (remaining >= MIN_PARTIAL_HIGHLIGHT_CHARS) { | ||
| kept.push(truncateHighlight(highlight, text, remaining)); | ||
| } | ||
| used = maxChars; | ||
| trimmed++; | ||
| } | ||
| source.highlights = kept; | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more.
When a search exceeds Useful? React with 👍 / 👎.
Collaborator
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Looked into this and I don't think it's a regression. Highlights are never carried in the For Happy to revisit if the goal is to surface full highlights in the artifact UI, but that would be changing the pre-existing post-format |
||
| } | ||
| } | ||
| return trimmed; | ||
| } | ||
|
|
||
| function addHighlightSection(): string[] { | ||
| return ['\n## Highlights', '']; | ||
| } | ||
|
|
@@ -112,8 +219,15 @@ function formatSource( | |
|
|
||
| export function formatResultsForLLM( | ||
| turn: number, | ||
| results: t.SearchResultData | ||
| results: t.SearchResultData, | ||
| maxOutputChars?: number | ||
| ): { output: string; references: t.ResultReference[] } { | ||
| /** Bound highlight content to the per-search budget before formatting */ | ||
| const trimmedHighlights = trimHighlightsToBudget( | ||
| results, | ||
| resolveMaxLLMOutputChars(maxOutputChars) | ||
| ); | ||
|
|
||
| /** Array to collect all output lines */ | ||
| const outputLines: string[] = []; | ||
|
|
||
|
|
@@ -243,8 +357,11 @@ export function formatResultsForLLM( | |
| outputLines.push(paaLines.join('')); | ||
| } | ||
|
|
||
| return { | ||
| output: outputLines.join('\n').trim(), | ||
| references, | ||
| }; | ||
| let output = outputLines.join('\n').trim(); | ||
| if (trimmedHighlights > 0) { | ||
| output += `\n\n_[${trimmedHighlights} additional highlight${ | ||
| trimmedHighlights === 1 ? '' : 's' | ||
| } omitted to fit the context budget; the cited sources contain the full content.]_`; | ||
| } | ||
| return { output, references }; | ||
| } | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
When a high-relevance highlight exceeds the remaining budget with less than
MIN_PARTIAL_HIGHLIGHT_CHARSleft, this branch drops it but leavesusedbelowmaxCharsand continues iterating, so a later lower-ranked short highlight can still be emitted. This breaks the relevance-ordered budget policy for searches with one long boundary highlight followed by short ones; mark the budget exhausted or stop scanning once the boundary highlight is dropped.Useful? React with 👍 / 👎.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
Fixed in e0b0cf9.
usedis now set tomaxCharswhenever the boundary highlight can't be kept whole — in the drop branch too, not just on truncation — so once relevance order forces a drop, every later (lower-ranked) highlight is dropped as well. Added a test: a long boundary highlight with <MIN room followed by a short one now correctly omits both.