Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions docs/examples.md
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,23 @@ text-only again, remove `image` from `modalities.input` and remove or set
declare image input support. Restart MCode after editing configuration and select
the configured model with `/provider`, or use `exec --model` for a single run.

Images stay in the conversation history, so a long session can carry more images
than a provider accepts in one request. MCode sends only the 20 most recent images
per request and replaces older ones with a short text note naming the original
file; the stored history keeps every image. If your provider allows fewer images
per request, set the limit on the model (Mistral's API is limited to 8
automatically):

```yaml
custom_provider:
my-vision-provider:
models:
my-vision-model:
capabilities:
support_image: true
max_images_per_request: 8
```

After signing in to MiniMax, try a task that explicitly requires search:

> Use web_search to find the official Node.js test runner documentation. Summarize how to run tests and include the source URL. If the tool is unavailable, say so.
Expand Down
16 changes: 16 additions & 0 deletions packages/agent-core/src/pi-turn-runner/agent.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ import { Agent } from '@earendil-works/pi-agent-core';
import type { UserMessage } from '@earendil-works/pi-ai';
import type { TurnTerminationReason } from '../event-bridge/types.js';
import { projectAgentMessagesForModel } from './outbound-message-normalizer.js';
import { resolveMaxImagesPerRequest } from './request-image-limit.js';
import { failureReason } from './terminal.js';
import type { turnState } from './turn.js';
import type { UserMessageInput } from './types.js';
Expand Down Expand Up @@ -73,6 +74,21 @@ export function newAgent(turn: turnState): Agent {
'[pi-turn-runner] kept provider-bound images whose dimensions could not be determined',
);
}
if (outbound.omittedImageCount > 0) {
// History keeps every image; only this request copy drops the oldest
// ones so it stays within the model's per-request image limit (#425).
turn.logger.info(
{
event: 'outbound_images_limited',
session_id: turn.input.sessionId,
turn_id: turn.input.turnId,
provider: turn.llm.model.provider,
omitted_image_count: outbound.omittedImageCount,
max_images_per_request: resolveMaxImagesPerRequest(turn.llm.model),
},
'[pi-turn-runner] replaced older provider-bound images with placeholders',
);
}
return outbound.messages;
},
streamFn: turn.streamFn,
Expand Down
8 changes: 8 additions & 0 deletions packages/agent-core/src/pi-turn-runner/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,14 @@ export {
projectAgentMessagesForModel,
removeOrphanToolResults,
} from './outbound-message-normalizer.js';
export {
DEFAULT_MAX_IMAGES_PER_REQUEST,
limitRequestImages,
knownMaxImagesPerRequestForBaseUrl,
normalizeMaxImagesPerRequest,
resolveMaxImagesPerRequest,
} from './request-image-limit.js';
export type { ModelRequestImageLimit, RequestImageLimitResult } from './request-image-limit.js';
export { toPiUserMessage } from './agent.js';
export {
createToolContextHistogramBucketsByName,
Expand Down
12 changes: 9 additions & 3 deletions packages/agent-core/src/pi-turn-runner/llm-retry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@ import type {
SimpleStreamOptions,
} from '@earendil-works/pi-ai';
import {
isLLMDeterministicRequestRejection,
normalizeLLMError,
toLLMMetricErrorKind,
toLLMProtocolClassification,
Expand Down Expand Up @@ -579,12 +580,17 @@ function failureResult(input: {
explicitAbort: input.final?.stopReason === 'aborted',
});
// BYOK retries every pre-output failure because custom gateways report errors
// inconsistently, but a model safety refusal is deterministic: retrying the
// same request only repeats (and may re-bill) the decline.
// inconsistently, but some failures are deterministic: a model safety refusal,
// or a request the provider rejects as invalid (too many images, or an HTTP
// 400 `invalid_request_error`). Retrying those only re-sends the same request,
// repeats (and may re-bill) the rejection and delays the actionable error by
// the whole backoff window. The deterministic check excludes anything that
// also looks transient (timeout, network, 408/429/5xx).
const decision =
input.retryAllErrors &&
!normalized.facts.explicitAbort &&
!normalized.facts.signals.has('refusal')
!normalized.facts.signals.has('refusal') &&
!isLLMDeterministicRequestRejection(normalized)
? { retryable: true, reason: 'network' as const }
: toLLMRetryDecision(normalized);
if (!decision.retryable) {
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ import type {
import { convertToLlm } from '@earendil-works/pi-coding-agent/messages';

import { imageDimensions } from './image-dimensions.js';
import { limitRequestImages, resolveMaxImagesPerRequest } from './request-image-limit.js';

const PRIOR_THINKING_OPEN = '<|prior-thinking|>';
const PRIOR_THINKING_CLOSE = '<|/prior-thinking|>';
Expand Down Expand Up @@ -52,6 +53,11 @@ export function projectAgentMessagesForModel(
* cluster is the signal to teach `imageDimensions` another format.
*/
undeterminedImageMimeTypes: Record<string, number>;
/**
* Older image blocks replaced by a placeholder so the request stays within
* the model's per-request image limit (see `request-image-limit.ts`).
*/
omittedImageCount: number;
} {
const providerVisible = messages.filter((message) => !isHostOnlyMessage(message));
const compatible = removeOrphanToolResults(
Expand All @@ -65,9 +71,13 @@ export function projectAgentMessagesForModel(
const projected = normalized
.map(projectVideoBlocks)
.map((message) => projectUndersizedImageBlocks(message, undetermined));
// Last, so the ceiling counts exactly the images that would be sent: video
// frames projected to images count, undersized placeholders do not.
const limited = limitRequestImages(projected, resolveMaxImagesPerRequest(targetModel));
return {
removedCount: compatible.removedCount,
messages: projected,
messages: limited.messages,
omittedImageCount: limited.omittedCount,
undeterminedImageMimeTypes: Object.fromEntries(undetermined),
};
}
Expand Down
241 changes: 241 additions & 0 deletions packages/agent-core/src/pi-turn-runner/request-image-limit.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,241 @@
import type { Api, Message, Model } from '@earendil-works/pi-ai';

/**
* Default ceiling on images in one provider request when the model declares
* none (`capabilities.max_images_per_request`).
*
* Inline images live in persisted history and every request re-sends the whole
* provider-visible history, so without a request-level ceiling a long session
* eventually exceeds the provider's per-request image limit. Every following
* request then fails the same way and the session cannot recover (#425: a
* gateway rejected `Too many images in request: 31 > 30`).
*
* 20 stays below the smallest general-purpose limit seen in practice — the
* gateway in #425 rejects more than 30 — with headroom for gateways that count
* images differently, and far below Anthropic (100 per request) and OpenAI
* (500). It still keeps the attachments of the last five turns (at most four
* inline images per user message) in view. Providers with a smaller limit set
* `max_images_per_request` on the model; Mistral's API (8) is recognized by
* host, see `knownMaxImagesPerRequestForBaseUrl`.
*/
export const DEFAULT_MAX_IMAGES_PER_REQUEST = 20;

/**
* Optional request limit carried on the resolved model object, so every
* request builder that receives the model (agent turns, compaction
* checkpoints, footprint measurement) applies the same ceiling.
*/
export interface ModelRequestImageLimit {
readonly maxImagesPerRequest?: number;
}

/** Normalize a configured image ceiling; anything but a positive integer is ignored. */
export function normalizeMaxImagesPerRequest(value: unknown): number | undefined {
const parsed =
typeof value === 'number'
? value
: typeof value === 'string' && value.trim()
? Number(value)
: Number.NaN;
return Number.isSafeInteger(parsed) && parsed > 0 ? parsed : undefined;
}

/**
* Documented per-request image limits for first-party API hosts whose limit is
* below the default. Matched on the exact hostname so gateways that merely
* proxy these models keep the default unless they declare their own.
*/
const KNOWN_HOST_IMAGE_LIMITS: ReadonlyMap<string, number> = new Map([
// Mistral Vision FAQ (docs.mistral.ai/capabilities/vision): "The maximum number
// images per request via API is 8."
['api.mistral.ai', 8],
]);

/** Known image ceiling for a first-party API base URL, if it is below the default. */
export function knownMaxImagesPerRequestForBaseUrl(baseUrl: string | undefined): number | undefined {
if (!baseUrl) return undefined;
try {
return KNOWN_HOST_IMAGE_LIMITS.get(new URL(baseUrl).hostname.toLowerCase());
} catch {
return undefined;
}
}

/** Image ceiling for requests to `model`: its declared limit, else the default. */
export function resolveMaxImagesPerRequest(model: Model<Api> | undefined): number {
const declared = normalizeMaxImagesPerRequest(
(model as (Model<Api> & ModelRequestImageLimit) | undefined)?.maxImagesPerRequest,
);
return declared ?? DEFAULT_MAX_IMAGES_PER_REQUEST;
}

export interface RequestImageLimitResult {
readonly messages: Message[];
/** Number of image blocks replaced by a placeholder in this request copy. */
readonly omittedCount: number;
}

/**
* Keep only the newest `maxImages` image blocks across the whole request
* (user messages and tool results) and replace older ones with a short text
* placeholder naming the original file when it is known.
*
* Operates on the temporary provider request copy only; canonical history is
* never rewritten, so raising the limit (or switching to a model with a higher
* one) brings the images back.
*/
export function limitRequestImages(
messages: readonly Message[],
maxImages: number,
): RequestImageLimitResult {
const limit = Math.max(0, Math.floor(maxImages));
let total = 0;
for (const message of messages) total += imageBlockCount(message);
if (total <= limit) return { messages: [...messages], omittedCount: 0 };

let toOmit = total - limit;
const toolCallPaths = collectToolCallPaths(messages);
const projected = messages.map((message) => {
if (toOmit === 0) return message;
const count = imageBlockCount(message);
if (count === 0) return message;
const paths = imagePathsForMessage(message, count, toolCallPaths);
let imageIndex = 0;
const content = (message.content as Array<{ type?: unknown }>).map((block) => {
if (!isImageBlock(block)) return block;
const path = paths[imageIndex];
imageIndex += 1;
if (toOmit === 0) return block;
toOmit -= 1;
return { type: 'text' as const, text: omittedImageText(limit, path) };
});
return { ...message, content } as Message;
});
return { messages: projected, omittedCount: total - limit };
}

function omittedImageText(limit: number, path: string | undefined): string {
const scope = `this request keeps only the ${limit} most recent images`;
return path
? `[Earlier image omitted: ${scope}. Original file: ${path}]`
: `[Earlier image omitted: ${scope}.]`;
}

function isImageBlock(block: unknown): boolean {
return Boolean(block) && typeof block === 'object' && (block as { type?: unknown }).type === 'image';
}

function imageBlockCount(message: Message): number {
if (!Array.isArray(message.content)) return 0;
let count = 0;
for (const block of message.content as unknown[]) if (isImageBlock(block)) count += 1;
return count;
}

const ATTACHMENT_TAG_OPEN = '<attachment';
const PATH_ARGUMENT_KEYS = ['path', 'file_path', 'filePath'] as const;

function imagePathsForMessage(
message: Message,
imageCount: number,
toolCallPaths: ReadonlyMap<string, string>,
): Array<string | undefined> {
if (message.role === 'toolResult') {
const path = toolCallPaths.get(message.toolCallId);
return imageCount === 1 && path ? [path] : [];
}
if (message.role !== 'user' || !Array.isArray(message.content)) return [];
// Attachment reminders list every attachment of the message in order; inline
// images are the ones sent as image blocks. Only trust the mapping when the
// counts agree, otherwise fall back to a placeholder without a path.
const inlineImagePaths: string[] = [];
for (const block of message.content) {
if (block.type !== 'text') continue;
for (const attributes of attachmentTagAttributes(block.text)) {
const path = attributes.get('path');
if (attributes.get('inline') !== 'true' || !path) continue;
const kind = attributes.get('kind');
const mime = attributes.get('mime') ?? '';
if (kind === 'image' || (kind === undefined && mime.startsWith('image/'))) {
inlineImagePaths.push(path);
}
}
}
return inlineImagePaths.length === imageCount ? inlineImagePaths : [];
}

function collectToolCallPaths(messages: readonly Message[]): Map<string, string> {
const paths = new Map<string, string>();
for (const message of messages) {
if (message.role !== 'assistant' || !Array.isArray(message.content)) continue;
for (const block of message.content) {
if (block.type !== 'toolCall' || !block.arguments || typeof block.arguments !== 'object') {
continue;
}
for (const key of PATH_ARGUMENT_KEYS) {
const value: unknown = block.arguments[key];
if (typeof value === 'string' && value.trim()) {
paths.set(block.id, value.trim());
break;
}
}
}
}
return paths;
}

/**
* Attributes of each `<attachment …>` tag in `text`. A linear scan rather than
* a regular expression, because message text is user-controlled and nested
* quantifiers over it can backtrack polynomially.
*/
function attachmentTagAttributes(text: string): Array<Map<string, string>> {
const tags: Array<Map<string, string>> = [];
let from = 0;
for (;;) {
const start = text.indexOf(ATTACHMENT_TAG_OPEN, from);
if (start < 0) break;
const end = text.indexOf('>', start);
// No later tag can be closed either.
if (end < 0) break;
const next = text.charAt(start + ATTACHMENT_TAG_OPEN.length);
if (next === '>' || next === '/' || /\s/u.test(next)) {
tags.push(parseTagAttributes(text.slice(start + ATTACHMENT_TAG_OPEN.length, end)));
}
from = end + 1;
}
return tags;
}

function parseTagAttributes(source: string): Map<string, string> {
const attributes = new Map<string, string>();
let position = 0;
while (position < source.length) {
const equals = source.indexOf('="', position);
if (equals < 0) break;
const close = source.indexOf('"', equals + 2);
if (close < 0) break;
let nameStart = equals;
while (nameStart > position && isAttributeNameChar(source.charCodeAt(nameStart - 1))) {
nameStart -= 1;
}
if (nameStart < equals) {
attributes.set(source.slice(nameStart, equals), unescapeAttribute(source.slice(equals + 2, close)));
}
position = close + 1;
}
return attributes;
}

/** `[a-z_]`, matching the attribute names the attachment reminder writes. */
function isAttributeNameChar(code: number): boolean {
return (code >= 97 && code <= 122) || code === 95;
}

function unescapeAttribute(value: string): string {
return value
.replaceAll('&quot;', '"')
.replaceAll('&lt;', '<')
.replaceAll('&gt;', '>')
.replaceAll('&amp;', '&');
}
Loading
Loading