Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "@librechat/agents",
"version": "3.9.0",
"version": "3.9.1",
"reova": {
"enabled": true,
"endpoint": "https://telemetry.reo.dev/data"
Expand Down
14 changes: 10 additions & 4 deletions src/graphs/Graph.ts
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,7 @@ import {
convertInjectedMessages,
coalesceAdjacentUserTurns,
appendPredecessorHandoffCue,
appendInstructionlessHandoffCue,
stampSyntheticProviderMessage,
} from '@/messages';
import {
Expand Down Expand Up @@ -3721,11 +3722,16 @@ export class StandardGraph extends Graph<t.BaseGraphState, t.GraphNode> {
* Applied HERE for the primary so the cue is part of the MEASURED
* payload — the pre-invoke projection and overflow guard run on this
* stage's output, and a post-measure append could push a just-fits
* prompt over budget unreported (#346 round 2). The attemptInvoke
* funnel re-keys per SERVING provider: it strips this cue for a
* tolerant fallback and adds it for a Claude fallback behind a
* tolerant primary.
* prompt over budget unreported (#346 round 2). Instructionless tool
* handoffs are grounded for every transport. Only the direct-edge
* predecessor cue is re-keyed per SERVING provider: tolerant
* fallbacks strip it, while Claude fallbacks add it.
*/
const beforeHandoffCue = transformed;
transformed = trackProviderMessageOrigins(
beforeHandoffCue,
appendInstructionlessHandoffCue(beforeHandoffCue, callConfig)
);
if (isAnthropicLike(provider, clientOptions as { model?: string })) {
const before = transformed;
transformed = trackProviderMessageOrigins(
Expand Down
33 changes: 23 additions & 10 deletions src/graphs/MultiAgentGraph.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,12 +28,13 @@ import {
setProviderMessageProvenance,
stampSyntheticProviderMessage,
} from '@/messages/provenance';
import { serializeToolContentBounded } from '@/utils/toolContent';
import { Constants, MULTI_AGENT_GRAPH_RUN_NAME } from '@/common';
import {
calculateMaxToolResultChars,
HARD_MAX_TOOL_RESULT_CHARS,
} from '@/utils/truncation';
import { withInstructionlessHandoffCue } from '@/messages/handoffCue';
import { serializeToolContentBounded } from '@/utils/toolContent';
import { Constants, MULTI_AGENT_GRAPH_RUN_NAME } from '@/common';
import { StandardGraph } from './Graph';

/** Pattern to extract instructions from transfer ToolMessage content */
Expand Down Expand Up @@ -1354,12 +1355,6 @@ export class MultiAgentGraph extends StandardGraph {
this.memberRecursionLimit == null
? config
: { ...config, recursionLimit: this.memberRecursionLimit };
const memberConfig = withActiveAgentMetadata(
recursionLimitedConfig,
agentId,
agentContext?.name
);

/**
* Check if this agent is receiving a handoff.
* If so, filter out the transfer messages and inject instructions as preamble.
Expand All @@ -1371,6 +1366,19 @@ export class MultiAgentGraph extends StandardGraph {
agentId
);

const memberConfig = withInstructionlessHandoffCue(
withActiveAgentMetadata(
recursionLimitedConfig,
agentId,
agentContext?.name
),
handoffContext != null &&
(handoffContext.instructions == null ||
handoffContext.instructions === '')
? handoffContext.filteredMessages.at(-1)
: undefined
);

if (
handoffContext?.sourceAgentName != null &&
handoffContext.sourceAgentName !== ''
Expand Down Expand Up @@ -1582,7 +1590,9 @@ export class MultiAgentGraph extends StandardGraph {
}

const startingNodes =
summarizeOnlyAgentId != null ? [summarizeOnlyAgentId] : this.startingNodes;
summarizeOnlyAgentId != null
? [summarizeOnlyAgentId]
: this.startingNodes;
for (const startNode of startingNodes) {
// eslint-disable-next-line @typescript-eslint/ban-ts-comment
/** @ts-ignore */
Expand Down Expand Up @@ -1634,7 +1644,10 @@ export class MultiAgentGraph extends StandardGraph {
this.resolveMaxRoutingPromptChars(destination);

if (typeof prompt === 'function') {
const resolvedPrompt = await prompt(state.messages, this.startIndex);
const resolvedPrompt = await prompt(
state.messages,
this.startIndex
);
promptText =
resolvedPrompt == null
? undefined
Expand Down
59 changes: 51 additions & 8 deletions src/llm/invoke.handoffCue.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,15 +7,20 @@
* cue stripped; and the serving model id must be read through the wrapper
* stack, or a wrapped Bedrock-Nova model would default to Claude.
*/
import { RunnableBinding } from '@langchain/core/runnables';
import { AIMessage, HumanMessage } from '@langchain/core/messages';
import type { BaseMessage } from '@langchain/core/messages';
import type { CallbackManagerForLLMRun } from '@langchain/core/callbacks/manager';
import type { ChatGenerationChunk } from '@langchain/core/outputs';
import { RunnableBinding } from '@langchain/core/runnables';
import { Providers } from '@/common';
import { PREDECESSOR_HANDOFF_CUE } from '@/messages/handoffCue';
import type { BaseMessage } from '@langchain/core/messages';
import type { InvokeContext } from './invoke';
import {
PREDECESSOR_HANDOFF_CUE,
INSTRUCTIONLESS_HANDOFF_CUE,
withInstructionlessHandoffCue,
} from '@/messages/handoffCue';
import { FakeChatModel } from '@/llm/fake';
import { attemptInvoke, type InvokeContext } from './invoke';
import { attemptInvoke } from './invoke';
import { Providers } from '@/common';

class CapturingChatModel extends FakeChatModel {
readonly invocations: BaseMessage[][] = [];
Expand Down Expand Up @@ -70,6 +75,46 @@ async function sentBy(
}

describe('attemptInvoke handoff-cue funnel', () => {
it('grounds repeated gateway attempts without mutating their replay source', async () => {
const model = new CapturingChatModel('analytics-gateway');
const messages = [new HumanMessage('go'), runTail];
const config = withInstructionlessHandoffCue(undefined, runTail);
const stream = model._streamResponseChunks.bind(model);
let failed = false;
jest
.spyOn(model, '_streamResponseChunks')
.mockImplementation(async function* (payload, options, runManager) {
if (!failed) {
failed = true;
model.invocations.push(payload);
throw new Error('Transient gateway failure');
}
yield* stream(payload, options, runManager);
});
const invoke = (): ReturnType<typeof attemptInvoke> =>
attemptInvoke(
{
model,
messages,
provider: Providers.OPENAI,
onChunk: async () => undefined,
},
config
);
await expect(invoke()).rejects.toThrow('Transient gateway failure');
await invoke();
expect(model.invocations).toHaveLength(2);
for (const request of model.invocations) {
expect(request.at(-1)?.content).toBe(INSTRUCTIONLESS_HANDOFF_CUE);
expect(
request.filter(
(message) => message.content === INSTRUCTIONLESS_HANDOFF_CUE
)
).toHaveLength(1);
}
expect(messages).toEqual([new HumanMessage('go'), runTail]);
});

it('applies the cue when the serving provider is Anthropic', async () => {
const sent = await sentBy(Providers.ANTHROPIC);
expect(sent.at(-1)?.content).toBe(PREDECESSOR_HANDOFF_CUE);
Expand All @@ -96,9 +141,7 @@ describe('attemptInvoke handoff-cue funnel', () => {
messages: [new HumanMessage('go'), runTail, bakedCue()],
});
expect(sent.at(-1)?.getType()).toBe('ai');
expect(sent.some((m) => m.content === PREDECESSOR_HANDOFF_CUE)).toBe(
false
);
expect(sent.some((m) => m.content === PREDECESSOR_HANDOFF_CUE)).toBe(false);
});

it('is idempotent when the primary already baked the cue', async () => {
Expand Down
69 changes: 62 additions & 7 deletions src/llm/prepareProviderRequest.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,17 +6,21 @@ import {
assertPreparedProviderRequestFor,
prepareProviderRequest,
} from '@/llm/prepareProviderRequest';
import { ToolOutputReferenceRegistry } from '@/tools/toolOutputReferences';
import { _convertMessagesToOpenAIParams } from '@/llm/openai/utils';
import { attemptInvoke } from '@/llm/invoke';
import { Providers } from '@/common';
import { foldToolBlocksForToollessAgent } from '@/messages/format';
import { createToolHistoryPreparation } from '@/messages/toolHistoryProjection';
import { createContextPressureMeter } from './contextPressureMeter';
import {
INSTRUCTIONLESS_HANDOFF_CUE,
withInstructionlessHandoffCue,
} from '@/messages/handoffCue';
import {
getProviderMessageProvenance,
getProviderSourceMessageIds,
} from '@/messages/provenance';
import { createToolHistoryPreparation } from '@/messages/toolHistoryProjection';
import { ToolOutputReferenceRegistry } from '@/tools/toolOutputReferences';
import { _convertMessagesToOpenAIParams } from '@/llm/openai/utils';
import { createContextPressureMeter } from './contextPressureMeter';
import { foldToolBlocksForToollessAgent } from '@/messages/format';
import { attemptInvoke } from '@/llm/invoke';
import { Providers } from '@/common';

type StubModel = {
model?: string;
Expand Down Expand Up @@ -44,6 +48,57 @@ function createCapturingModel(): CapturingModel {
}

describe('prepareProviderRequest', () => {
it.each([Providers.OPENAI, Providers.ANTHROPIC, Providers.BEDROCK])(
'measures the instructionless handoff cue for %s without modifying its source',
(provider) => {
const { model } = createCapturingModel();
const tail = new AIMessage({
content: 'Transferring',
id: 'handoff-tail',
});
const messages = [new HumanMessage('Analyze this'), tail];
const config = withInstructionlessHandoffCue(undefined, tail);
const measured = jest.fn((prepared: BaseMessage[]) => ({
fits: prepared.length <= messages.length,
}));
const request = prepareProviderRequest({
model: model as t.ChatModel,
messages,
provider,
config,
measure: measured,
});
expect(request.messages.at(-1)?.content).toBe(
INSTRUCTIONLESS_HANDOFF_CUE
);
expect(measured).toHaveBeenCalledWith(request.messages);
expect(request.measurement?.fits).toBe(false);
expect(messages).toHaveLength(2);
expect(tail.additional_kwargs).toEqual({});

const fallback = prepareProviderRequest({
model: model as t.ChatModel,
messages: request.messages,
provider: Providers.OPENAI,
config,
});
expect(
fallback.messages.filter(
(message) => message.content === INSTRUCTIONLESS_HANDOFF_CUE
)
).toHaveLength(1);
const freshFallback = prepareProviderRequest({
model: model as t.ChatModel,
messages,
provider: Providers.OPENAI,
config,
});
expect(freshFallback.messages.at(-1)?.content).toBe(
INSTRUCTIONLESS_HANDOFF_CUE
);
}
);

it('honors an explicit Chat override over Responses model defaults', () => {
const { model } = createCapturingModel();
model._useResponsesApi = (options?: unknown): boolean =>
Expand Down
10 changes: 6 additions & 4 deletions src/llm/prepareProviderRequest.ts
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,8 @@ import type {
} from '@langchain/core/messages';
import type { RunnableConfig } from '@langchain/core/runnables';
import type { ToolOutputReferenceRegistry } from '@/tools/toolOutputReferences';
import type * as t from '@/types';
import type { ToolHistoryPreparation } from '@/messages/toolHistoryProjection';
import type * as t from '@/types';
import {
projectCacheControlledToolOutputsToText,
projectComputerCallOutputsToText,
Expand All @@ -21,19 +21,20 @@ import {
import {
coalesceAdjacentUserTurns,
appendPredecessorHandoffCue,
appendInstructionlessHandoffCue,
removePredecessorHandoffCue,
} from '@/messages';
import {
stripAnthropicCacheControl,
stripBedrockCacheControl,
cloneMessage,
} from '@/messages/cache';
import { createToolHistoryPreparation } from '@/messages/toolHistoryProjection';
import { isAnthropicLike, isGoogleLike, isOpenAILike } from '@/utils/llm';
import { annotateMessagesForLLM } from '@/tools/toolOutputReferences';
import { providerRequiresStrictAlternation } from '@/llm/providers';
import { getProviderFamily } from '@/llm/providerRegistry';
import { Providers } from '@/common';
import { createToolHistoryPreparation } from '@/messages/toolHistoryProjection';

const preparedProviderRequestBrand = Symbol('PreparedProviderRequest');
const OMITTED_ATTACHMENT_TEXT =
Expand Down Expand Up @@ -555,16 +556,17 @@ export function prepareProviderRequest({
const annotated = annotateMessagesForLLM(projected, registry, runId);
const isRunProduced = context?.isRunProducedMessage;
const modelId = resolveServingModelId(model);
const handoffGrounded = appendInstructionlessHandoffCue(annotated, config);
const cued = isAnthropicLike(provider, {
model: modelId,
})
? appendPredecessorHandoffCue(
annotated,
handoffGrounded,
isRunProduced == null
? undefined
: (message): boolean => isRunProduced.call(context, message)
)
: removePredecessorHandoffCue(annotated);
: removePredecessorHandoffCue(handoffGrounded);
const preparedMessages = providerRequiresStrictAlternation(provider)
? coalesceAdjacentUserTurns(cued)
: cued;
Expand Down
Loading
Loading