2026-01-04 22:37:09 -05:00
|
|
|
// Claude helper functions for translator
|
|
|
|
|
import { DEFAULT_THINKING_CLAUDE_SIGNATURE } from "../../config/defaultThinkingSignature.js";
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
import { ROLE, CLAUDE_BLOCK } from "../schema/index.js";
|
|
|
|
|
import { adjustMaxTokens } from "./maxTokens.js";
|
2026-03-03 02:46:05 -05:00
|
|
|
import { applyCloaking } from "../../utils/claudeCloaking.js";
|
2026-06-15 00:38:43 -04:00
|
|
|
import { resolveSessionId } from "../../utils/sessionManager.js";
|
2026-06-20 04:08:11 -04:00
|
|
|
import { isValidClaudeSignature } from "../../utils/claudeSignature.js";
|
2026-06-14 02:15:48 -04:00
|
|
|
import { PROVIDERS } from "../../providers/index.js";
|
2026-06-16 12:32:28 -04:00
|
|
|
import { getCapabilitiesForModel } from "../../providers/capabilities.js";
|
|
|
|
|
import { DEFAULT_MAX_TOKENS } from "../../config/runtimeConfig.js";
|
2026-01-04 22:37:09 -05:00
|
|
|
|
2026-08-14 05:08:30 -04:00
|
|
|
const CACHE_CONTROL_5M = { type: "ephemeral" };
|
|
|
|
|
const CACHE_CONTROL_1H = { type: "ephemeral", ttl: "1h" };
|
|
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Check if message has valid non-empty content
|
|
|
|
|
export function hasValidContent(msg) {
|
|
|
|
|
if (typeof msg.content === "string" && msg.content.trim()) return true;
|
|
|
|
|
if (Array.isArray(msg.content)) {
|
|
|
|
|
return msg.content.some(block =>
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
(block.type === CLAUDE_BLOCK.TEXT && block.text?.trim()) ||
|
|
|
|
|
block.type === CLAUDE_BLOCK.TOOL_USE ||
|
2026-08-05 02:19:55 -04:00
|
|
|
block.type === CLAUDE_BLOCK.TOOL_RESULT ||
|
|
|
|
|
block.type === CLAUDE_BLOCK.IMAGE ||
|
|
|
|
|
block.type === CLAUDE_BLOCK.DOCUMENT
|
2026-01-04 22:37:09 -05:00
|
|
|
);
|
|
|
|
|
}
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Fix tool_use/tool_result ordering for Claude API
|
|
|
|
|
// 1. Assistant message with tool_use: remove text AFTER tool_use (Claude doesn't allow)
|
|
|
|
|
// 2. Merge consecutive same-role messages
|
|
|
|
|
export function fixToolUseOrdering(messages) {
|
|
|
|
|
if (messages.length <= 1) return messages;
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Pass 1: Fix assistant messages with tool_use - remove text after tool_use
|
|
|
|
|
for (const msg of messages) {
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
if (msg.role === ROLE.ASSISTANT && Array.isArray(msg.content)) {
|
|
|
|
|
const hasToolUse = msg.content.some(b => b.type === CLAUDE_BLOCK.TOOL_USE);
|
2026-01-04 22:37:09 -05:00
|
|
|
if (hasToolUse) {
|
|
|
|
|
// Keep only: thinking blocks + tool_use blocks (remove text blocks after tool_use)
|
|
|
|
|
const newContent = [];
|
|
|
|
|
let foundToolUse = false;
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
for (const block of msg.content) {
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
if (block.type === CLAUDE_BLOCK.TOOL_USE) {
|
2026-01-04 22:37:09 -05:00
|
|
|
foundToolUse = true;
|
|
|
|
|
newContent.push(block);
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
} else if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
|
2026-01-04 22:37:09 -05:00
|
|
|
newContent.push(block);
|
|
|
|
|
} else if (!foundToolUse) {
|
|
|
|
|
// Keep text blocks BEFORE tool_use
|
|
|
|
|
newContent.push(block);
|
|
|
|
|
}
|
|
|
|
|
// Skip text blocks AFTER tool_use
|
|
|
|
|
}
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
msg.content = newContent;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Pass 2: Merge consecutive same-role messages
|
|
|
|
|
const merged = [];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
for (const msg of messages) {
|
|
|
|
|
const last = merged[merged.length - 1];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
if (last && last.role === msg.role) {
|
|
|
|
|
// Merge content arrays
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
const lastContent = Array.isArray(last.content) ? last.content : [{ type: CLAUDE_BLOCK.TEXT, text: last.content }];
|
|
|
|
|
const msgContent = Array.isArray(msg.content) ? msg.content : [{ type: CLAUDE_BLOCK.TEXT, text: msg.content }];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Put tool_result first, then other content
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
const toolResults = [...lastContent.filter(b => b.type === CLAUDE_BLOCK.TOOL_RESULT), ...msgContent.filter(b => b.type === CLAUDE_BLOCK.TOOL_RESULT)];
|
|
|
|
|
const otherContent = [...lastContent.filter(b => b.type !== CLAUDE_BLOCK.TOOL_RESULT), ...msgContent.filter(b => b.type !== CLAUDE_BLOCK.TOOL_RESULT)];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
last.content = [...toolResults, ...otherContent];
|
|
|
|
|
} else {
|
|
|
|
|
// Ensure content is array
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
const content = Array.isArray(msg.content) ? msg.content : [{ type: CLAUDE_BLOCK.TEXT, text: msg.content }];
|
2026-01-04 22:37:09 -05:00
|
|
|
merged.push({ role: msg.role, content: [...content] });
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
return merged;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-20 05:29:15 -04:00
|
|
|
// Models that reject thinking.type "adaptive" + output_config.effort (Opus 4.5+/Sonnet 4.6+ only)
|
2026-06-08 04:37:01 -04:00
|
|
|
const ADAPTIVE_THINKING_UNSUPPORTED = /haiku/i;
|
|
|
|
|
|
2026-06-25 23:05:01 -04:00
|
|
|
function handlesThinkingBlocks(provider) {
|
|
|
|
|
return provider === "claude" || provider?.startsWith("anthropic-compatible") || provider === "deepseek";
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
function buildThinkingPlaceholder(provider) {
|
|
|
|
|
const block = {
|
|
|
|
|
type: CLAUDE_BLOCK.THINKING,
|
|
|
|
|
thinking: ".",
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
// DeepSeek's Anthropic-compatible endpoint requires a thinking block in
|
|
|
|
|
// thinking mode, but it does not need Anthropic's signed-thinking fallback.
|
|
|
|
|
if (provider !== "deepseek") {
|
|
|
|
|
block.signature = DEFAULT_THINKING_CLAUDE_SIGNATURE;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return block;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-08 04:37:01 -04:00
|
|
|
// Normalize a native Claude passthrough body to match Anthropic Messages API spec.
|
|
|
|
|
// Newer Cowork/Claude Code clients emit beta-only shapes that OAuth endpoints reject:
|
|
|
|
|
// 1. thinking.type "adaptive" → unsupported on Haiku
|
2026-06-20 05:29:15 -04:00
|
|
|
// 2. output_config.effort → unsupported on Haiku
|
|
|
|
|
// 3. role "system" messages (mid-conversation-system beta) → only top-level system is allowed
|
2026-06-08 04:37:01 -04:00
|
|
|
export function normalizeClaudePassthrough(body, model = "") {
|
|
|
|
|
if (!body || typeof body !== "object") return body;
|
|
|
|
|
|
|
|
|
|
// 1. Downgrade adaptive thinking for models that don't support it
|
|
|
|
|
if (body.thinking?.type === "adaptive" && ADAPTIVE_THINKING_UNSUPPORTED.test(model)) {
|
|
|
|
|
body.thinking = { type: "enabled", budget_tokens: 10000 };
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-20 05:29:15 -04:00
|
|
|
// 2. Strip effort param for models that don't support it (keep other output_config fields)
|
|
|
|
|
if (ADAPTIVE_THINKING_UNSUPPORTED.test(model) && body.output_config?.effort != null) {
|
|
|
|
|
delete body.output_config.effort;
|
|
|
|
|
if (Object.keys(body.output_config).length === 0) delete body.output_config;
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-14 05:08:30 -04:00
|
|
|
// 2. Fold mid-conversation system messages into the neighbouring turn.
|
|
|
|
|
// Hoisting them into body.system would insert volatile content (token counters,
|
|
|
|
|
// reminders) ahead of the whole conversation and invalidate the prefix cache on
|
|
|
|
|
// every request. Folding in place keeps the cached prefix stable.
|
2026-06-08 04:37:01 -04:00
|
|
|
if (Array.isArray(body.messages)) {
|
|
|
|
|
const messages = [];
|
|
|
|
|
for (const msg of body.messages) {
|
2026-08-14 05:08:30 -04:00
|
|
|
if (msg.role !== ROLE.SYSTEM) {
|
|
|
|
|
messages.push(msg);
|
2026-06-08 04:37:01 -04:00
|
|
|
continue;
|
|
|
|
|
}
|
2026-08-14 05:08:30 -04:00
|
|
|
const text = typeof msg.content === "string"
|
|
|
|
|
? msg.content
|
|
|
|
|
: Array.isArray(msg.content)
|
|
|
|
|
? msg.content.map(b => (typeof b === "string" ? b : b?.text || "")).join("\n")
|
|
|
|
|
: "";
|
|
|
|
|
if (!text.trim()) continue;
|
2026-06-08 04:37:01 -04:00
|
|
|
|
2026-08-14 05:08:30 -04:00
|
|
|
// Copy-on-write: the caller's body is reused across account-fallback
|
|
|
|
|
// attempts, so folding must never mutate the original message.
|
|
|
|
|
const block = { type: CLAUDE_BLOCK.TEXT, text };
|
|
|
|
|
const prev = messages[messages.length - 1];
|
|
|
|
|
if (prev?.role === ROLE.USER) {
|
|
|
|
|
const content = typeof prev.content === "string"
|
|
|
|
|
? [{ type: CLAUDE_BLOCK.TEXT, text: prev.content }]
|
|
|
|
|
: Array.isArray(prev.content) ? [...prev.content] : [];
|
|
|
|
|
messages[messages.length - 1] = { ...prev, content: [...content, block] };
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
messages.push({ role: ROLE.USER, content: [block] });
|
2026-06-08 04:37:01 -04:00
|
|
|
}
|
2026-08-14 05:08:30 -04:00
|
|
|
body.messages = messages;
|
2026-06-08 04:37:01 -04:00
|
|
|
}
|
|
|
|
|
|
2026-07-03 04:06:19 -04:00
|
|
|
// 3. Drop thinking blocks whose signature is not Claude's (combo mixes models,
|
|
|
|
|
// so foreign signatures leak into history and Anthropic rejects them).
|
|
|
|
|
const thinkingEnabled = body.thinking?.type === "enabled";
|
|
|
|
|
if (Array.isArray(body.messages)) {
|
|
|
|
|
for (const msg of body.messages) {
|
|
|
|
|
if (msg.role !== ROLE.ASSISTANT || !Array.isArray(msg.content)) continue;
|
|
|
|
|
let hasToolUse = false;
|
|
|
|
|
let hasKeptThinking = false;
|
|
|
|
|
const kept = [];
|
|
|
|
|
for (const block of msg.content) {
|
|
|
|
|
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) {
|
|
|
|
|
if (isValidClaudeSignature(block.signature)) {
|
|
|
|
|
hasKeptThinking = true;
|
|
|
|
|
kept.push(block);
|
|
|
|
|
}
|
|
|
|
|
continue;
|
|
|
|
|
}
|
|
|
|
|
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
|
|
|
|
kept.push(block);
|
|
|
|
|
}
|
|
|
|
|
msg.content = kept;
|
|
|
|
|
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
|
|
|
|
|
msg.content.unshift(buildThinkingPlaceholder("claude"));
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-08 04:37:01 -04:00
|
|
|
return body;
|
|
|
|
|
}
|
|
|
|
|
|
2026-08-14 05:08:30 -04:00
|
|
|
// Put a 5m breakpoint on the last cache-eligible block of a message.
|
|
|
|
|
// thinking/redacted_thinking blocks do not accept cache_control.
|
|
|
|
|
function markLastCacheableBlock(msg) {
|
|
|
|
|
if (!Array.isArray(msg?.content)) return false;
|
|
|
|
|
for (let i = msg.content.length - 1; i >= 0; i--) {
|
|
|
|
|
const block = msg.content[i];
|
|
|
|
|
if (typeof block !== "object" || block === null) continue;
|
|
|
|
|
if (block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING) continue;
|
|
|
|
|
block.cache_control = { ...CACHE_CONTROL_5M };
|
|
|
|
|
return true;
|
|
|
|
|
}
|
|
|
|
|
return false;
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Re-anchor cache breakpoints on a Claude passthrough body (same policy as
|
|
|
|
|
// prepareClaudeRequest): last tool + last system block at 1h, last assistant at 5m.
|
|
|
|
|
// The client's own markers point at pre-normalization offsets, so they are dropped.
|
|
|
|
|
// Must run LAST, after every step that can reshape system/tools/messages
|
|
|
|
|
// (normalize, tool dedupe, token savers) — otherwise the anchor drifts off the tail.
|
|
|
|
|
export function anchorClaudeCache(body) {
|
|
|
|
|
if (!body || typeof body !== "object") return body;
|
|
|
|
|
|
|
|
|
|
if (Array.isArray(body.system)) {
|
|
|
|
|
const last = body.system.length - 1;
|
|
|
|
|
body.system.forEach((block, i) => {
|
|
|
|
|
if (typeof block !== "object" || block === null) return;
|
|
|
|
|
if (i === last) block.cache_control = { ...CACHE_CONTROL_1H };
|
|
|
|
|
else delete block.cache_control;
|
|
|
|
|
});
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (Array.isArray(body.tools)) {
|
|
|
|
|
const last = body.tools.length - 1;
|
|
|
|
|
body.tools.forEach((tool, i) => {
|
|
|
|
|
if (i === last) tool.cache_control = { ...CACHE_CONTROL_1H };
|
|
|
|
|
else delete tool.cache_control;
|
|
|
|
|
});
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if (Array.isArray(body.messages)) {
|
|
|
|
|
let anchored = null;
|
|
|
|
|
for (let i = body.messages.length - 1; i >= 0; i--) {
|
|
|
|
|
const msg = body.messages[i];
|
|
|
|
|
if (!Array.isArray(msg.content)) continue;
|
|
|
|
|
for (const block of msg.content) delete block.cache_control;
|
|
|
|
|
|
|
|
|
|
// Prefer the last assistant turn: it ends a completed exchange, so the
|
|
|
|
|
// prefix up to it stays byte-stable across the following requests.
|
|
|
|
|
if (anchored || msg.role !== ROLE.ASSISTANT) continue;
|
|
|
|
|
anchored = markLastCacheableBlock(msg);
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// First turn of a conversation has no assistant yet — anchor the final
|
|
|
|
|
// message instead, so the opening prompt is cached rather than paid twice.
|
|
|
|
|
if (!anchored) {
|
|
|
|
|
for (let i = body.messages.length - 1; i >= 0 && !anchored; i--) {
|
|
|
|
|
anchored = markLastCacheableBlock(body.messages[i]);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return body;
|
|
|
|
|
}
|
|
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Prepare request for Claude format endpoints
|
|
|
|
|
// - Cleanup cache_control
|
|
|
|
|
// - Filter empty messages
|
|
|
|
|
// - Add thinking block for Anthropic endpoint (provider === "claude")
|
|
|
|
|
// - Fix tool_use/tool_result ordering
|
2026-03-03 02:46:05 -05:00
|
|
|
// - Apply cloaking (billing header + fake user ID) for OAuth tokens
|
2026-06-16 12:32:28 -04:00
|
|
|
export function prepareClaudeRequest(body, provider = null, apiKey = null, connectionId = null, rawHeaders = null, sessionId = null) {
|
2026-06-14 02:15:48 -04:00
|
|
|
// quirk: MiniMax's Claude-compatible endpoint rejects Anthropic's output_config (400 invalid params)
|
|
|
|
|
if (PROVIDERS[provider]?.quirks?.dropOutputConfig) {
|
2026-05-01 05:16:01 -04:00
|
|
|
delete body.output_config;
|
|
|
|
|
}
|
|
|
|
|
|
2026-07-05 06:32:25 -04:00
|
|
|
// Clamp max_tokens to the model's real output ceiling. Models whose caps
|
|
|
|
|
// declare a higher maxOutput (e.g. Opus 4.8 / Sonnet 4.6 = 128000) are allowed
|
|
|
|
|
// up to it, so max-effort thinking gets full budget; others fall back to the
|
|
|
|
|
// conservative 64000 default.
|
2026-06-16 12:32:28 -04:00
|
|
|
if (body.max_tokens) {
|
2026-07-05 06:32:25 -04:00
|
|
|
const ceiling = getCapabilitiesForModel(provider, body.model).maxOutput || DEFAULT_MAX_TOKENS;
|
2026-06-16 12:32:28 -04:00
|
|
|
if (body.max_tokens > ceiling) body.max_tokens = ceiling;
|
2026-07-05 06:32:25 -04:00
|
|
|
|
|
|
|
|
// Reconcile against thinking budget. applyThinking (thinkingUnified.js) runs
|
|
|
|
|
// AFTER adjustMaxTokens capped max_tokens, and the claude-budget format maps
|
|
|
|
|
// max effort → budget_tokens 128000 — larger than the clamped max_tokens.
|
|
|
|
|
// Anthropic requires max_tokens strictly greater than budget_tokens (else 400).
|
|
|
|
|
// Prefer raising max_tokens to preserve the requested thinking depth; if the
|
|
|
|
|
// budget alone meets/exceeds the ceiling, cap output and shrink the budget so
|
|
|
|
|
// some tokens remain for the answer.
|
|
|
|
|
if (body.thinking?.type === "enabled" && body.thinking.budget_tokens && body.thinking.budget_tokens >= body.max_tokens) {
|
|
|
|
|
body.max_tokens = Math.min(body.thinking.budget_tokens + 1024, ceiling);
|
|
|
|
|
if (body.thinking.budget_tokens >= body.max_tokens) {
|
|
|
|
|
body.thinking.budget_tokens = Math.max(1024, body.max_tokens - 1024);
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-06-16 12:32:28 -04:00
|
|
|
}
|
|
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// 1. System: remove all cache_control, add only to last block with ttl 1h
|
|
|
|
|
if (body.system && Array.isArray(body.system)) {
|
|
|
|
|
body.system = body.system.map((block, i) => {
|
|
|
|
|
const { cache_control, ...rest } = block;
|
|
|
|
|
if (i === body.system.length - 1) {
|
|
|
|
|
return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } };
|
|
|
|
|
}
|
|
|
|
|
return rest;
|
|
|
|
|
});
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// 2. Messages: process in optimized passes
|
|
|
|
|
if (body.messages && Array.isArray(body.messages)) {
|
|
|
|
|
const len = body.messages.length;
|
|
|
|
|
let filtered = [];
|
|
|
|
|
|
|
|
|
|
// Pass 1: remove cache_control + filter empty messages
|
|
|
|
|
for (let i = 0; i < len; i++) {
|
|
|
|
|
const msg = body.messages[i];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
// Remove cache_control from content blocks
|
|
|
|
|
if (Array.isArray(msg.content)) {
|
|
|
|
|
for (const block of msg.content) {
|
|
|
|
|
delete block.cache_control;
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Keep final assistant even if empty, otherwise check valid content
|
|
|
|
|
const isFinalAssistant = i === len - 1 && msg.role === "assistant";
|
|
|
|
|
if (isFinalAssistant || hasValidContent(msg)) {
|
|
|
|
|
filtered.push(msg);
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Pass 1.5: Fix tool_use/tool_result ordering
|
|
|
|
|
// Each tool_use must have tool_result in the NEXT message (not same message with other content)
|
|
|
|
|
filtered = fixToolUseOrdering(filtered);
|
|
|
|
|
|
|
|
|
|
body.messages = filtered;
|
|
|
|
|
|
|
|
|
|
// Check if thinking is enabled AND last message is from user
|
|
|
|
|
const lastMessage = filtered[filtered.length - 1];
|
|
|
|
|
const lastMessageIsUser = lastMessage?.role === "user";
|
|
|
|
|
const thinkingEnabled = body.thinking?.type === "enabled" && lastMessageIsUser;
|
|
|
|
|
|
|
|
|
|
// Pass 2 (reverse): add cache_control to last assistant + handle thinking for Anthropic
|
|
|
|
|
let lastAssistantProcessed = false;
|
|
|
|
|
for (let i = filtered.length - 1; i >= 0; i--) {
|
|
|
|
|
const msg = filtered[i];
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
if (msg.role === "assistant" && Array.isArray(msg.content)) {
|
2026-03-25 05:57:26 -04:00
|
|
|
// Add cache_control to last non-thinking block of first (from end) assistant with content
|
|
|
|
|
// thinking/redacted_thinking blocks do not support cache_control
|
2026-01-04 22:37:09 -05:00
|
|
|
if (!lastAssistantProcessed && msg.content.length > 0) {
|
2026-03-25 05:57:26 -04:00
|
|
|
for (let j = msg.content.length - 1; j >= 0; j--) {
|
|
|
|
|
const block = msg.content[j];
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
if (block.type !== CLAUDE_BLOCK.THINKING && block.type !== CLAUDE_BLOCK.REDACTED_THINKING) {
|
2026-03-25 05:57:26 -04:00
|
|
|
block.cache_control = { type: "ephemeral" };
|
|
|
|
|
break;
|
|
|
|
|
}
|
|
|
|
|
}
|
2026-01-04 22:37:09 -05:00
|
|
|
lastAssistantProcessed = true;
|
|
|
|
|
}
|
|
|
|
|
|
2026-06-25 23:05:01 -04:00
|
|
|
// Handle thinking blocks for Anthropic-compatible endpoints.
|
|
|
|
|
if (handlesThinkingBlocks(provider)) {
|
2026-01-04 22:37:09 -05:00
|
|
|
let hasToolUse = false;
|
2026-06-25 23:05:01 -04:00
|
|
|
let hasKeptThinking = false;
|
2026-03-25 05:57:26 -04:00
|
|
|
|
2026-06-20 04:08:11 -04:00
|
|
|
// Claude native: preserve valid signatures, drop invalid blocks.
|
|
|
|
|
// anthropic-compatible: replace with default (safe fallback for lenient upstreams).
|
2026-06-25 23:05:01 -04:00
|
|
|
// DeepSeek: keep existing thinking as-is; add an unsigned placeholder only if missing.
|
2026-06-20 04:08:11 -04:00
|
|
|
const isClaudeNative = provider === "claude";
|
2026-06-25 23:05:01 -04:00
|
|
|
const isDeepSeek = provider === "deepseek";
|
2026-06-20 04:08:11 -04:00
|
|
|
const kept = [];
|
2026-01-04 22:37:09 -05:00
|
|
|
for (const block of msg.content) {
|
2026-06-20 04:08:11 -04:00
|
|
|
const isThinking = block.type === CLAUDE_BLOCK.THINKING || block.type === CLAUDE_BLOCK.REDACTED_THINKING;
|
|
|
|
|
if (isThinking) {
|
|
|
|
|
if (isClaudeNative) {
|
2026-06-25 23:05:01 -04:00
|
|
|
if (isValidClaudeSignature(block.signature)) {
|
|
|
|
|
hasKeptThinking = true;
|
|
|
|
|
kept.push(block);
|
|
|
|
|
}
|
|
|
|
|
} else if (isDeepSeek) {
|
|
|
|
|
hasKeptThinking = true;
|
|
|
|
|
kept.push(block);
|
2026-06-20 04:08:11 -04:00
|
|
|
} else {
|
|
|
|
|
block.signature = DEFAULT_THINKING_CLAUDE_SIGNATURE;
|
2026-06-25 23:05:01 -04:00
|
|
|
hasKeptThinking = true;
|
2026-06-20 04:08:11 -04:00
|
|
|
kept.push(block);
|
|
|
|
|
}
|
|
|
|
|
continue;
|
2026-01-04 22:37:09 -05:00
|
|
|
}
|
refactor(open-sse): translator DRY + schema enums, bug fixes, dead code cleanup
- Bug B1-B7: media UI m.kind||m.type, serviceKinds, gemini mediaPriority, schema kind, models/info lookup by kind
- Dead code D1-D6: safeParseJSON, drop PROVIDER_ENDPOINTS, orphan fetcher, GITHUB_CONFIG derive, getProviderConfig internal, legacy kiro file
- Translator concerns: toOpenAIUsage, toOpenAIFinish (gemini/kiro/ollama + fix kiro tool finish), thinking effort maps
- Reorg helpers/ → concerns/ (logic) + formats/ (per-format) + schema/ (pure enums: roles/blocks/finishReasons/defaults)
- Wire ~280 hardcoded role/block/finish/default literals to schema enums across 20+ files
- collapseTextParts + extractTextContent dedup
- Normalize translator fn names to openaiToXRequest / xToOpenAIResponse
- Golden tests lock behavior; 0 regression (byte-for-byte providers/alias, 26=26 known fails)
Co-authored-by: Cursor <cursoragent@cursor.com>
2026-06-14 07:49:38 -04:00
|
|
|
if (block.type === CLAUDE_BLOCK.TOOL_USE) hasToolUse = true;
|
2026-06-20 04:08:11 -04:00
|
|
|
kept.push(block);
|
2026-01-04 22:37:09 -05:00
|
|
|
}
|
2026-06-20 04:08:11 -04:00
|
|
|
msg.content = kept;
|
2026-01-04 22:37:09 -05:00
|
|
|
|
|
|
|
|
// Add thinking block if thinking enabled + has tool_use but no thinking
|
2026-06-25 23:05:01 -04:00
|
|
|
if (thinkingEnabled && !hasKeptThinking && hasToolUse) {
|
|
|
|
|
msg.content.unshift(buildThinkingPlaceholder(provider));
|
2026-01-04 22:37:09 -05:00
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-02-10 07:18:40 -05:00
|
|
|
// 3. Tools: filter built-in tools for non-Anthropic providers, then handle cache_control
|
2026-01-04 22:37:09 -05:00
|
|
|
if (body.tools && Array.isArray(body.tools)) {
|
2026-06-21 05:26:17 -04:00
|
|
|
// Strip built-in tools (e.g. web_search_20250305) and normalize to Anthropic-native shape
|
|
|
|
|
// (drop `type` field, fold `function.{name,description,parameters}`) for non-Anthropic providers
|
2026-02-10 07:18:40 -05:00
|
|
|
if (provider !== "claude") {
|
2026-06-21 05:26:17 -04:00
|
|
|
body.tools = body.tools
|
|
|
|
|
.filter(tool => !tool.type || tool.type === "function")
|
|
|
|
|
.map(tool => {
|
|
|
|
|
if (tool.function) {
|
|
|
|
|
return {
|
|
|
|
|
name: tool.function.name,
|
|
|
|
|
description: tool.function.description,
|
|
|
|
|
input_schema: tool.function.parameters,
|
|
|
|
|
};
|
|
|
|
|
}
|
|
|
|
|
const { type, ...rest } = tool;
|
|
|
|
|
return rest;
|
|
|
|
|
});
|
2026-02-10 07:18:40 -05:00
|
|
|
}
|
|
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
body.tools = body.tools.map((tool, i) => {
|
|
|
|
|
const { cache_control, ...rest } = tool;
|
|
|
|
|
if (i === body.tools.length - 1) {
|
|
|
|
|
return { ...rest, cache_control: { type: "ephemeral", ttl: "1h" } };
|
|
|
|
|
}
|
|
|
|
|
return rest;
|
|
|
|
|
});
|
2026-02-10 07:18:40 -05:00
|
|
|
|
|
|
|
|
// Remove tools array and tool_choice if empty after filtering
|
|
|
|
|
if (body.tools.length === 0) {
|
|
|
|
|
delete body.tools;
|
|
|
|
|
delete body.tool_choice;
|
|
|
|
|
}
|
2026-01-04 22:37:09 -05:00
|
|
|
}
|
|
|
|
|
|
2026-03-03 02:46:05 -05:00
|
|
|
// Apply cloaking for OAuth tokens (billing header + fake user ID)
|
2026-04-04 20:46:26 -04:00
|
|
|
// session_id in user_id must match X-Claude-Code-Session-Id for fingerprint consistency
|
2026-03-25 05:57:26 -04:00
|
|
|
if ((provider === "claude" || provider?.startsWith("anthropic-compatible")) && apiKey) {
|
2026-06-16 12:32:28 -04:00
|
|
|
const sid = sessionId || resolveSessionId({ headers: rawHeaders, body, connectionId, scope: "claude" });
|
|
|
|
|
body = applyCloaking(body, apiKey, sid);
|
2026-03-03 02:46:05 -05:00
|
|
|
}
|
|
|
|
|
|
2026-01-04 22:37:09 -05:00
|
|
|
return body;
|
|
|
|
|
}
|