2026-06-19 23:09:50 -04:00
import { detectFormat , getTargetFormat , resolveTransport } from "../services/provider.js" ;
2026-02-26 23:15:12 -05:00
import { translateRequest } from "../translator/index.js" ;
2026-08-05 00:39:07 -04:00
import { applyThinking , extractThinking , stripThinkingSuffix } from "../translator/concerns/thinkingUnified.js" ;
2026-01-04 22:37:09 -05:00
import { FORMATS } from "../translator/formats.js" ;
2026-08-14 05:08:30 -04:00
import { normalizeClaudePassthrough , anchorClaudeCache } from "../translator/formats/claude.js" ;
2026-02-26 23:15:12 -05:00
import { createStreamController } from "../utils/streamHandler.js" ;
2026-01-11 09:45:01 -05:00
import { refreshWithRetry } from "../services/tokenRefresh.js" ;
2026-01-04 22:37:09 -05:00
import { createRequestLogger } from "../utils/requestLogger.js" ;
2026-06-06 00:31:52 -04:00
import { getModelTargetFormat , getModelStrip , getModelUpstreamId , getModelType , PROVIDER _ID _TO _ALIAS } from "../config/providerModels.js" ;
2026-06-13 11:05:01 -04:00
import { PROVIDERS } from "../config/providers.js" ;
2026-01-04 22:37:09 -05:00
import { createErrorResult , parseUpstreamError , formatProviderError } from "../utils/error.js" ;
2026-07-16 00:18:27 -04:00
import { HTTP _STATUS , TOKEN _SAVER _HEADER } from "../config/runtimeConfig.js" ;
2026-01-04 22:37:09 -05:00
import { handleBypassRequest } from "../utils/bypassHandler.js" ;
2026-02-26 23:15:12 -05:00
import { trackPendingRequest , appendRequestLog , saveRequestDetail } from "@/lib/usageDb.js" ;
2026-01-11 09:45:01 -05:00
import { getExecutor } from "../executors/index.js" ;
2026-07-16 04:28:25 -04:00
import { supportsGrokCliReasoningEffort } from "../config/grokCli.js" ;
2026-02-26 23:15:12 -05:00
import { buildRequestDetail , extractRequestConfig } from "./chatCore/requestDetail.js" ;
import { handleForcedSSEToJson } from "./chatCore/sseToJsonHandler.js" ;
import { handleNonStreamingResponse } from "./chatCore/nonStreamingHandler.js" ;
import { handleStreamingResponse , buildOnStreamComplete } from "./chatCore/streamingHandler.js" ;
2026-04-04 12:47:39 -04:00
import { detectClientTool , isNativePassthrough } from "../utils/clientDetector.js" ;
2026-05-11 22:19:50 -04:00
import { dedupeTools } from "../utils/toolDeduper.js" ;
2026-04-30 07:00:38 -04:00
import { injectCaveman } from "../rtk/caveman.js" ;
2026-06-19 23:09:50 -04:00
import { injectPonytail } from "../rtk/ponytail.js" ;
2026-04-30 07:00:38 -04:00
import { compressMessages , formatRtkLog } from "../rtk/index.js" ;
2026-06-26 00:08:33 -04:00
import { compressWithHeadroom , formatHeadroomLog , formatHeadroomSizeLog , isHeadroomPhantomSavings } from "../rtk/headroom.js" ;
2026-07-10 07:01:20 -04:00
import { compressWithPxpipe } from "../rtk/pxpipe.js" ;
2026-06-15 07:18:04 -04:00
import { getCapabilitiesForModel } from "../providers/capabilities.js" ;
import { stripUnsupportedModalities } from "../translator/concerns/modality.js" ;
import { prefetchRemoteImages } from "../translator/concerns/prefetch.js" ;
2026-07-10 07:01:20 -04:00
import { resolveSessionId } from "../utils/sessionManager.js" ;
2026-02-08 04:45:31 -05:00
2026-01-04 22:37:09 -05:00
/ * *
* Core chat handler - shared between SSE and Worker
* @ param { object } options . body - Request body
* @ param { object } options . modelInfo - { provider , model }
* @ param { object } options . credentials - Provider credentials
2026-02-26 23:15:12 -05:00
* @ param { string } options . sourceFormatOverride - Override detected source format ( e . g . "openai-responses" )
2026-01-04 22:37:09 -05:00
* /
2026-08-05 02:23:03 -04:00
/ * *
* Remove translator - internal continuity fields from the outbound upstream
* body . The Responses → Chat request translator stashes reasoning
* ` encrypted_content ` on assistant messages so a later openai → responses
* round - trip can restore the store = false continuity blob ; that stash must
* never reach an upstream provider . Chat - native proxies reject the unknown
* assistant - message field and answer every turn with a literal "400" body
* ( observed with multi - turn Codex sessions via OpenAI - compatible nodes ) .
* /
export function stripContinuityFields ( body ) {
if ( ! body || ! Array . isArray ( body . messages ) ) return body ;
for ( const msg of body . messages ) {
if ( msg && typeof msg === "object" ) {
delete msg . encrypted _content ;
delete msg . reasoning _encrypted _content ;
}
}
return body ;
}
2026-07-10 05:10:16 -04:00
export async function handleChatCore ( { body , modelInfo , credentials , log , onCredentialsRefreshed , onRequestSuccess , onDisconnect , clientRawRequest , connectionId , userAgent , apiKey , ccFilterNaming , rtkEnabled , headroomEnabled , headroomUrl , headroomCompressUserMessages , cavemanEnabled , cavemanLevel , ponytailEnabled , ponytailLevel , pxpipeEnabled , pxpipeMinChars , pxpipeTimeoutMs , pxpipeTransform , onPxpipeEvent , sourceFormatOverride , providerThinking } ) {
2026-01-04 22:37:09 -05:00
const { provider , model } = modelInfo ;
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
const requestStartTime = Date . now ( ) ;
2026-07-10 07:01:20 -04:00
// Stable per-session color so all lines of one CLI conversation share a tag
const sessionSeed = ( ( ) => {
try {
return resolveSessionId ( { headers : clientRawRequest ? . headers , body , connectionId , scope : provider } ) ;
} catch {
return connectionId || "" ;
}
} ) ( ) ;
const reqTag = log ? . tagForSession ? log . tagForSession ( sessionSeed ) : ( log ? . nextTag ? log . nextTag ( ) : "" ) ;
2026-01-04 22:37:09 -05:00
2026-02-26 23:15:12 -05:00
const sourceFormat = sourceFormatOverride || detectFormat ( body ) ;
2026-01-04 22:37:09 -05:00
2026-03-12 22:41:40 -04:00
// Check for bypass patterns (warmup, skip, cc naming)
const bypassResponse = handleBypassRequest ( body , model , userAgent , ccFilterNaming ) ;
2026-02-26 23:15:12 -05:00
if ( bypassResponse ) return bypassResponse ;
2026-01-04 22:37:09 -05:00
const alias = PROVIDER _ID _TO _ALIAS [ provider ] || provider ;
const modelTargetFormat = getModelTargetFormat ( alias , model ) ;
2026-06-19 23:09:50 -04:00
// Multi-endpoint providers: pick transport matching sourceFormat → zero translation
const runtimeTransport = resolveTransport ( provider , sourceFormat ) ;
2026-08-05 02:23:03 -04:00
const targetFormat = modelTargetFormat || runtimeTransport ? . format || getTargetFormat ( provider , credentials ) ;
2026-06-19 23:09:50 -04:00
if ( runtimeTransport && credentials ) credentials . runtimeTransport = runtimeTransport ;
2026-04-06 23:06:55 -04:00
const stripList = getModelStrip ( alias , model ) ;
2026-06-05 23:36:03 -04:00
const upstreamModel = getModelUpstreamId ( alias , model ) ;
2026-02-01 21:17:15 -05:00
2026-04-13 01:04:57 -04:00
// Inject provider-level thinking config override (only if client hasn't set)
// on/off → extended type (body.thinking), none/low/medium/high → effort type (body.reasoning_effort)
if ( providerThinking ? . mode && providerThinking . mode !== "auto" ) {
const mode = providerThinking . mode ;
if ( mode === "on" && ! body . thinking ) {
console . log ( "Injecting provider-level thinking config override: on" ) ;
body = { ... body , thinking : { type : "enabled" , budget _tokens : 10000 } } ;
} else if ( mode === "off" && ! body . thinking ) {
body = { ... body , thinking : { type : "disabled" } } ;
} else if ( ! body . reasoning _effort ) {
body = { ... body , reasoning _effort : mode } ;
}
}
2026-02-22 09:44:11 -05:00
const clientRequestedStreaming = body . stream === true || sourceFormat === FORMATS . ANTIGRAVITY || sourceFormat === FORMATS . GEMINI || sourceFormat === FORMATS . GEMINI _CLI ;
2026-06-13 11:05:01 -04:00
const providerRequiresStreaming = PROVIDERS [ provider ] ? . forceStream === true ;
2026-03-12 23:00:47 -04:00
let stream = providerRequiresStreaming ? true : ( body . stream !== false ) ;
2026-06-21 06:50:21 -04:00
// Image generation models require non-streaming (Google v1internal:generateContent)
const modelType = getModelType ( alias , model ) ;
const isImageGenModel = modelType === "imageGen" || /image|imagen|image-generation/i . test ( model ) ;
if ( isImageGenModel && ( provider === "antigravity" || provider === "gemini-cli" ) ) {
stream = false ;
}
2026-05-13 11:40:42 -04:00
// DeepSeek-TUI: interactive TUI panel sends stream:true and needs SSE.
// Non-interactive mode (-p flag) sends without stream and can't parse SSE.
// Only force non-streaming when client didn't explicitly request it.
const detectedTool = detectClientTool ( clientRawRequest ? . headers || { } , body ) ;
if ( detectedTool === "deepseek-tui" && body . stream !== true ) stream = false ;
2026-03-12 23:00:47 -04:00
// Check client Accept header preference for non-streaming requests
// This fixes AI SDK compatibility where clients send Accept: application/json
const acceptHeader = clientRawRequest ? . headers ? . accept || "" ;
const clientPrefersJson = acceptHeader . includes ( "application/json" ) ;
const clientPrefersSSE = acceptHeader . includes ( "text/event-stream" ) ;
2026-06-25 23:33:29 -04:00
if ( clientPrefersJson && ! clientPrefersSSE && body . stream !== true && ! providerRequiresStreaming ) {
2026-03-12 23:00:47 -04:00
stream = false ;
}
2026-01-04 22:37:09 -05:00
const reqLogger = await createRequestLogger ( sourceFormat , targetFormat , model ) ;
2026-02-26 23:15:12 -05:00
if ( clientRawRequest ) reqLogger . logClientRawRequest ( clientRawRequest . endpoint , clientRawRequest . body , clientRawRequest . headers ) ;
2026-01-04 22:37:09 -05:00
reqLogger . logRawRequest ( body ) ;
log ? . debug ? . ( "FORMAT" , ` ${ sourceFormat } → ${ targetFormat } | stream= ${ stream } ` ) ;
2026-04-04 12:47:39 -04:00
// Native passthrough: CLI tool and provider are the same ecosystem
// Skip all translation/normalization — only model and Bearer are swapped
const clientTool = detectClientTool ( clientRawRequest ? . headers || { } , body ) ;
const passthrough = isNativePassthrough ( clientTool , provider ) ;
2026-06-15 00:38:43 -04:00
// Expose raw client headers to translators/executors for session-id resolution
if ( credentials ) credentials . rawHeaders = clientRawRequest ? . headers || { } ;
2026-06-15 07:18:04 -04:00
// Auto-strip media blocks the model can't read (vision/audio/pdf) before translation.
if ( ! passthrough ) {
const caps = getCapabilitiesForModel ( provider , model ) ;
if ( stripUnsupportedModalities ( body , sourceFormat , caps ) ) {
log ? . debug ? . ( "MODALITY" , ` stripped unsupported media for ${ provider } / ${ model } ` ) ;
}
// Convert remote image URLs to base64 for targets that can't fetch URLs.
try {
const n = await prefetchRemoteImages ( body , sourceFormat , targetFormat , { signal : undefined } ) ;
if ( n > 0 ) log ? . debug ? . ( "MODALITY" , ` prefetched ${ n } remote image(s) for ${ targetFormat } ` ) ;
} catch ( e ) { log ? . warn ? . ( "MODALITY" , ` image prefetch failed: ${ e . message } ` ) ; }
}
2026-04-04 12:47:39 -04:00
let translatedBody ;
let toolNameMap ;
2026-08-05 02:23:03 -04:00
let customToolNames ;
2026-04-04 12:47:39 -04:00
if ( passthrough ) {
log ? . debug ? . ( "PASSTHROUGH" , ` ${ clientTool } → ${ provider } | native lossless ` ) ;
# v0.5.20 (2026-07-07)
## Features
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
- **RTK**: add JS-native git-log filter (#2423)
- **Caveman**: add targeted upstream-aligned style rules (#2424)
- **i18n**: add Farsi (fa) language support (#2385)
## Fixes
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
- **count_tokens**: count structured Anthropic blocks (#2419)
- **Volcengine-ark**: clamp GLM-5 max_tokens to model output ceiling (#2428)
- **Kimi**: normalize reasoning_effort to backend enum (#2427)
- **Claude**: reconcile max_tokens vs thinking budget and lift per-model ceiling (#2381)
- **Kiro**: deliver system prompt natively, add Opus 4.5/4.7/4.8, tolerate dash version ids (#2366)
- **Headroom**: proxy dashboard through app (#2372)
- **MITM**: recover from stale lock file on server start
2026-07-07 05:29:11 -04:00
translatedBody = { ... body , model : stripThinkingSuffix ( upstreamModel ) } ;
2026-08-05 00:39:07 -04:00
if ( provider === "codex" ) {
const suffixThinking = { } ;
applyThinking ( sourceFormat , upstreamModel , suffixThinking , provider ) ;
if ( suffixThinking . reasoning _effort ) {
const reasoning = translatedBody . reasoning ;
translatedBody . reasoning = {
... ( reasoning && typeof reasoning === "object" && ! Array . isArray ( reasoning ) ? reasoning : { } ) ,
effort : suffixThinking . reasoning _effort ,
} ;
delete translatedBody . reasoning _effort ;
}
}
2026-06-08 04:37:01 -04:00
// Normalize newer Cowork/CC beta shapes (adaptive thinking, mid-conversation system) the API rejects
# v0.5.20 (2026-07-07)
## Features
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
- **RTK**: add JS-native git-log filter (#2423)
- **Caveman**: add targeted upstream-aligned style rules (#2424)
- **i18n**: add Farsi (fa) language support (#2385)
## Fixes
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
- **count_tokens**: count structured Anthropic blocks (#2419)
- **Volcengine-ark**: clamp GLM-5 max_tokens to model output ceiling (#2428)
- **Kimi**: normalize reasoning_effort to backend enum (#2427)
- **Claude**: reconcile max_tokens vs thinking budget and lift per-model ceiling (#2381)
- **Kiro**: deliver system prompt natively, add Opus 4.5/4.7/4.8, tolerate dash version ids (#2366)
- **Headroom**: proxy dashboard through app (#2372)
- **MITM**: recover from stale lock file on server start
2026-07-07 05:29:11 -04:00
if ( clientTool === "claude" ) normalizeClaudePassthrough ( translatedBody , translatedBody . model ) ;
2026-04-04 12:47:39 -04:00
} else {
2026-06-05 23:36:03 -04:00
translatedBody = translateRequest ( sourceFormat , targetFormat , upstreamModel , body , stream , credentials , provider , reqLogger , stripList , connectionId , clientTool ) ;
2026-04-04 12:47:39 -04:00
if ( ! translatedBody ) {
trackPendingRequest ( model , provider , connectionId , false , true ) ;
return createErrorResult ( HTTP _STATUS . BAD _REQUEST , ` Failed to translate request for ${ sourceFormat } → ${ targetFormat } ` ) ;
}
toolNameMap = translatedBody . _toolNameMap ;
delete translatedBody . _toolNameMap ;
2026-08-05 02:23:03 -04:00
customToolNames = translatedBody . _customToolNames ;
delete translatedBody . _customToolNames ;
# v0.5.20 (2026-07-07)
## Features
- **Thinking**: per-model thinking level picker on provider page — appends `(level)` suffix to copied model names for forced reasoning effort across all formats (openai, claude, gemini, deepseek, kimi, qwen, zai, minimax, hunyuan, step)
- **RTK**: add JS-native git-log filter (#2423)
- **Caveman**: add targeted upstream-aligned style rules (#2424)
- **i18n**: add Farsi (fa) language support (#2385)
## Fixes
- **Thinking**: strip `(level)` suffix from upstream `body.model` so providers no longer reject requests
- **Translator**: preserve developer instructions in openai-responses conversion (#2434)
- **count_tokens**: count structured Anthropic blocks (#2419)
- **Volcengine-ark**: clamp GLM-5 max_tokens to model output ceiling (#2428)
- **Kimi**: normalize reasoning_effort to backend enum (#2427)
- **Claude**: reconcile max_tokens vs thinking budget and lift per-model ceiling (#2381)
- **Kiro**: deliver system prompt natively, add Opus 4.5/4.7/4.8, tolerate dash version ids (#2366)
- **Headroom**: proxy dashboard through app (#2372)
- **MITM**: recover from stale lock file on server start
2026-07-07 05:29:11 -04:00
translatedBody . model = stripThinkingSuffix ( upstreamModel ) ;
2026-08-05 02:23:03 -04:00
stripContinuityFields ( translatedBody ) ;
2026-03-13 22:37:29 -04:00
}
2026-01-04 22:37:09 -05:00
2026-05-11 22:19:50 -04:00
// Dedupe duplicate built-in tools when equivalent MCP tools are present (Claude clients only).
if ( clientTool === "claude" && Array . isArray ( translatedBody . tools ) ) {
const { tools : deduped , stripped } = dedupeTools ( translatedBody . tools ) ;
if ( stripped . length > 0 ) {
translatedBody . tools = deduped ;
log ? . debug ? . ( "TOOLDEDUP" , ` stripped ${ stripped . length } : ${ stripped . slice ( 0 , 3 ) . join ( ", " ) } ${ stripped . length > 3 ? "..." : "" } ` ) ;
}
}
2026-04-30 07:00:38 -04:00
// Token savers: applied at the final body just before dispatch
// Covers both passthrough (source shape) and translated (target shape) flows
const finalFormat = passthrough ? sourceFormat : targetFormat ;
2026-07-10 07:01:20 -04:00
// Request line: one correlated summary (fmt + thinking + counts + account)
if ( log ? . line ) {
const clientModel = clientRawRequest ? . body ? . model || ` ${ provider } / ${ model } ` ;
const msgN = translatedBody . messages ? . length || translatedBody . input ? . length || translatedBody . contents ? . length || body . messages ? . length || body . input ? . length || 0 ;
const toolN = translatedBody . tools ? . length || body . tools ? . length || 0 ;
const fmtStr = passthrough ? ` FMT: ${ sourceFormat } (passthrough) ` : ` FMT: ${ sourceFormat } → ${ targetFormat } ` ;
2026-07-16 04:28:25 -04:00
const showThinking = provider !== "grok-cli" || supportsGrokCliReasoningEffort ( model ) ;
const think = showThinking ? log . fmtThink ? . ( extractThinking ( translatedBody ) ) : null ;
2026-07-10 07:01:20 -04:00
const acc = credentials ? . connectionName || credentials ? . connectionId ? . slice ( 0 , 8 ) || "-" ;
const parts = [
` POST ${ clientModel } → ${ provider } / ${ model } ` ,
fmtStr ,
stream ? "STREAM" : "JSON" ,
` ${ msgN } MSG ` ,
] ;
if ( toolN ) parts . push ( ` ${ toolN } TOOL ` ) ;
if ( think ) parts . push ( ` THINK: ${ think } ` ) ;
parts . push ( ` ACC: ${ acc } ` ) ;
log . line ( reqTag , "▶" , parts . join ( " · " ) ) ;
}
2026-06-06 00:31:52 -04:00
// TTS models don't support tool messages/function calling
if ( getModelType ( alias , model ) === "tts" && translatedBody . messages ) {
translatedBody . messages = translatedBody . messages . filter ( msg => msg . role !== "tool" ) ;
delete translatedBody . tools ;
}
2026-07-16 00:18:27 -04:00
// Per-request opt-out: client can bypass all token savers via header
const tokenSaverEnabled = clientRawRequest ? . headers ? . [ TOKEN _SAVER _HEADER ] ? . toLowerCase ( ) !== "off" ;
2026-04-30 07:00:38 -04:00
// RTK: compress tool_result content
2026-07-16 00:18:27 -04:00
const rtkStats = compressMessages ( translatedBody , tokenSaverEnabled && rtkEnabled ) ;
2026-04-30 07:00:38 -04:00
const rtkLine = formatRtkLog ( rtkStats ) ;
if ( rtkLine ) console . log ( rtkLine ) ;
2026-06-19 23:09:50 -04:00
// Headroom: optional external proxy compression; fail open if proxy is absent.
2026-06-26 00:08:33 -04:00
const headroomDiagnostics = { } ;
2026-07-16 00:18:27 -04:00
const headroomStats = await compressWithHeadroom ( translatedBody , { enabled : tokenSaverEnabled && headroomEnabled , url : headroomUrl , model : upstreamModel , format : finalFormat , compressUserMessages : headroomCompressUserMessages , diagnostics : headroomDiagnostics } ) ;
2026-06-19 23:09:50 -04:00
const headroomLine = formatHeadroomLog ( headroomStats ) ;
2026-06-26 00:08:33 -04:00
const headroomSizeLine = formatHeadroomSizeLog ( headroomDiagnostics ) ;
if ( headroomLine ) {
log ? . info ? . ( "HEADROOM" , ` ${ headroomLine } ${ headroomSizeLine ? ` | ${ headroomSizeLine } ` : "" } ` ) ;
if ( isHeadroomPhantomSavings ( headroomStats , headroomDiagnostics ) ) {
2026-07-10 07:01:20 -04:00
log ? . warn ? . ( "HEADROOM" , ` reported token delta, but outbound JSON shrank <5%; provider may bill near-original payload | ${ formatHeadroomSizeLog ( headroomDiagnostics ) } ` ) ;
2026-06-26 00:08:33 -04:00
}
2026-07-16 00:18:27 -04:00
} else if ( tokenSaverEnabled && headroomEnabled ) log ? . warn ? . ( "HEADROOM" , ` skipped: ${ headroomDiagnostics . reason || "compression unavailable" } ${ headroomDiagnostics . endpoint ? ` ( ${ headroomDiagnostics . endpoint } ) ` : "" } ` ) ;
2026-06-19 23:09:50 -04:00
2026-07-16 07:13:51 -04:00
// Token-saver flags accumulator for the single "⚙" log line below.
const xf = [ ] ;
2026-04-30 07:00:38 -04:00
// Caveman: inject terse-style system prompt
2026-07-16 00:18:27 -04:00
if ( tokenSaverEnabled && cavemanEnabled && cavemanLevel ) {
2026-04-30 07:00:38 -04:00
injectCaveman ( translatedBody , finalFormat , cavemanLevel ) ;
2026-07-10 07:01:20 -04:00
xf . push ( ` CAVEMAN: ${ cavemanLevel } ` ) ;
2026-04-30 07:00:38 -04:00
}
2026-06-19 23:09:50 -04:00
// Ponytail: inject lazy-senior-dev system prompt
2026-07-16 00:18:27 -04:00
if ( tokenSaverEnabled && ponytailEnabled && ponytailLevel ) {
2026-06-19 23:09:50 -04:00
injectPonytail ( translatedBody , finalFormat , ponytailLevel ) ;
2026-07-10 07:01:20 -04:00
xf . push ( ` PONYTAIL: ${ ponytailLevel } ` ) ;
2026-06-19 23:09:50 -04:00
}
2026-07-10 05:10:16 -04:00
// PXPIPE: image bulky context (Claude-format bodies only), last saver before dispatch
let pxpipeSummary = null ;
if ( pxpipeEnabled ) {
const pxpipeResult = await compressWithPxpipe ( translatedBody , {
enabled : true , format : finalFormat , model : upstreamModel ,
minChars : pxpipeMinChars , timeoutMs : pxpipeTimeoutMs , transform : pxpipeTransform ,
} ) ;
pxpipeSummary = pxpipeResult . summary ;
if ( pxpipeResult . body ) translatedBody = pxpipeResult . body ;
2026-07-10 07:01:20 -04:00
if ( pxpipeSummary ? . applied ) xf . push ( ` PXPIPE: ${ pxpipeSummary . imageCount } img ` ) ;
2026-07-10 05:10:16 -04:00
try { onPxpipeEvent ? . ( { provider , model , ... pxpipeSummary } ) ; } catch { /* stats must not break requests */ }
}
2026-07-10 07:01:20 -04:00
if ( xf . length && log ? . line ) log . line ( reqTag , "⚙" , xf . join ( " · " ) ) ;
2026-08-14 05:08:30 -04:00
// Pin cache breakpoints to the final body — every saver above can reshape
// system/tools/messages, and a stale anchor costs a full prefix rewrite.
if ( passthrough && clientTool === "claude" ) anchorClaudeCache ( translatedBody ) ;
2026-01-11 09:45:01 -05:00
const executor = getExecutor ( provider ) ;
2026-01-06 13:41:53 -05:00
trackPendingRequest ( model , provider , connectionId , true ) ;
2026-05-13 11:40:42 -04:00
appendRequestLog ( { model , provider , connectionId , status : "PENDING" } ) . catch ( ( ) => { } ) ;
2026-01-06 13:41:53 -05:00
2026-02-26 23:15:12 -05:00
const msgCount = translatedBody . messages ? . length || translatedBody . input ? . length || translatedBody . contents ? . length || translatedBody . request ? . contents ? . length || 0 ;
2026-01-04 22:37:09 -05:00
log ? . debug ? . ( "REQUEST" , ` ${ provider . toUpperCase ( ) } | ${ model } | ${ msgCount } msgs ` ) ;
2026-02-26 23:15:12 -05:00
const streamController = createStreamController ( {
2026-02-14 23:51:37 -05:00
onDisconnect : ( reason ) => {
trackPendingRequest ( model , provider , connectionId , false ) ;
if ( onDisconnect ) onDisconnect ( reason ) ;
} ,
2026-02-26 23:15:12 -05:00
onError : ( ) => trackPendingRequest ( model , provider , connectionId , false ) ,
2026-07-10 07:01:20 -04:00
log , provider , model , reqTag
2026-02-14 23:51:37 -05:00
} ) ;
2026-01-04 22:37:09 -05:00
2026-03-09 04:46:06 -04:00
const proxyOptions = {
connectionProxyEnabled : credentials ? . providerSpecificData ? . connectionProxyEnabled === true ,
connectionProxyUrl : credentials ? . providerSpecificData ? . connectionProxyUrl || "" ,
connectionNoProxy : credentials ? . providerSpecificData ? . connectionNoProxy || "" ,
2026-04-12 23:08:24 -04:00
vercelRelayUrl : credentials ? . providerSpecificData ? . vercelRelayUrl || "" ,
2026-03-09 04:46:06 -04:00
} ;
2026-04-12 23:08:24 -04:00
if ( proxyOptions . vercelRelayUrl ) {
const connectionName = credentials ? . connectionName || credentials ? . connectionId || "unknown" ;
const poolId = credentials ? . providerSpecificData ? . connectionProxyPoolId || "none" ;
log ? . info ? . ( "PROXY" , ` ${ provider . toUpperCase ( ) } | ${ model } | conn= ${ connectionName } | pool= ${ poolId } | vercel-relay= ${ proxyOptions . vercelRelayUrl } ` ) ;
} else if ( proxyOptions . connectionProxyEnabled && proxyOptions . connectionProxyUrl ) {
2026-03-09 04:46:06 -04:00
let maskedProxyUrl = proxyOptions . connectionProxyUrl ;
try {
const parsed = new URL ( proxyOptions . connectionProxyUrl ) ;
const host = parsed . hostname || "" ;
const port = parsed . port ? ` : ${ parsed . port } ` : "" ;
const protocol = parsed . protocol || "http:" ;
maskedProxyUrl = ` ${ protocol } // ${ host } ${ port } ` ;
} catch {
// Keep raw if URL parsing fails
}
const poolId = credentials ? . providerSpecificData ? . connectionProxyPoolId || "none" ;
const connectionName = credentials ? . connectionName || credentials ? . connectionId || "unknown" ;
log ? . info ? . ( "PROXY" , ` ${ provider . toUpperCase ( ) } | ${ model } | conn= ${ connectionName } | pool= ${ poolId } | url= ${ maskedProxyUrl } ` ) ;
}
if ( proxyOptions . connectionProxyEnabled && proxyOptions . connectionNoProxy ) {
const connectionName = credentials ? . connectionName || credentials ? . connectionId || "unknown" ;
log ? . debug ? . ( "PROXY" , ` ${ provider . toUpperCase ( ) } | ${ model } | conn= ${ connectionName } | no_proxy= ${ proxyOptions . connectionNoProxy } ` ) ;
}
2026-02-26 23:15:12 -05:00
// Execute request
let providerResponse , providerUrl , providerHeaders , finalBody ;
2026-07-20 04:39:17 -04:00
// Most executors return their registry format. Cursor AgentService is an
// exception: it is decoded by the executor into OpenAI-compatible output.
let providerResponseFormat = targetFormat ;
2026-01-04 22:37:09 -05:00
try {
2026-03-09 04:46:06 -04:00
const result = await executor . execute ( { model , body : translatedBody , stream , credentials , signal : streamController . signal , log , proxyOptions } ) ;
2026-01-11 09:45:01 -05:00
providerResponse = result . response ;
providerUrl = result . url ;
providerHeaders = result . headers ;
finalBody = result . transformedBody ;
2026-07-20 04:39:17 -04:00
providerResponseFormat = result . responseFormat || targetFormat ;
2026-01-26 22:49:16 -05:00
reqLogger . logTargetRequest ( providerUrl , providerHeaders , finalBody ) ;
2026-01-04 22:37:09 -05:00
} catch ( error ) {
2026-02-21 04:42:46 -05:00
trackPendingRequest ( model , provider , connectionId , false , true ) ;
2026-05-13 11:40:42 -04:00
appendRequestLog ( { model , provider , connectionId , status : ` FAILED ${ error . name === "AbortError" ? 499 : HTTP _STATUS . BAD _GATEWAY } ` } ) . catch ( ( ) => { } ) ;
2026-02-26 23:15:12 -05:00
saveRequestDetail ( buildRequestDetail ( {
provider , model , connectionId ,
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
latency : { ttft : 0 , total : Date . now ( ) - requestStartTime } ,
tokens : { prompt _tokens : 0 , completion _tokens : 0 } ,
request : extractRequestConfig ( body , stream ) ,
providerRequest : translatedBody || null ,
2026-02-26 23:15:12 -05:00
response : { error : error . message || String ( error ) , status : error . name === "AbortError" ? 499 : 502 , thinking : null } ,
2026-07-10 05:10:16 -04:00
pxpipe : pxpipeSummary ,
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
status : "error"
2026-05-13 11:40:42 -04:00
} ) ) . catch ( ( ) => { } ) ;
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
2026-01-04 22:37:09 -05:00
if ( error . name === "AbortError" ) {
streamController . handleError ( error ) ;
return createErrorResult ( 499 , "Request aborted" ) ;
}
2026-02-06 23:17:06 -05:00
const errMsg = formatProviderError ( error , provider , model , HTTP _STATUS . BAD _GATEWAY ) ;
2026-07-10 07:01:20 -04:00
if ( log ? . errorLine ) {
log . errorLine ( reqTag , "✗" , ` ERROR 502 · ${ provider } / ${ model } · ${ Date . now ( ) - requestStartTime } ms \n ${ errMsg } ${ error . stack ? ` \n ${ error . stack } ` : "" } ` ) ;
}
2026-02-06 23:17:06 -05:00
return createErrorResult ( HTTP _STATUS . BAD _GATEWAY , errMsg ) ;
2026-01-04 22:37:09 -05:00
}
2026-04-13 23:14:50 -04:00
// Handle 401/403 - try token refresh (skip for noAuth providers)
if ( ! executor . noAuth && ( providerResponse . status === HTTP _STATUS . UNAUTHORIZED || providerResponse . status === HTTP _STATUS . FORBIDDEN ) ) {
2026-03-13 22:37:29 -04:00
try {
2026-07-25 06:30:11 -04:00
// Mutate credentials after each successful refresh: rotating refresh_token
// providers (xAI/grok-cli) issue a new RT on every refresh; without this,
// refreshWithRetry's 2nd/3rd attempt reuses the already-consumed RT →
// invalid_grant → auth_failed retryable=false.
const newCredentials = await refreshWithRetry ( async ( ) => {
const result = await executor . refreshCredentials ( credentials , log ) ;
if ( result ? . refreshToken && result . refreshToken !== credentials . refreshToken ) {
if ( result . accessToken ) credentials . accessToken = result . accessToken ;
credentials . refreshToken = result . refreshToken ;
}
return result ;
} , 3 , log ) ;
2026-03-13 22:37:29 -04:00
if ( newCredentials ? . accessToken || newCredentials ? . copilotToken ) {
2026-07-10 07:01:20 -04:00
if ( log ? . line ) log . line ( reqTag , "🔑" , ` TOKEN REFRESHED · ${ provider } / ${ model } ` ) ;
2026-03-13 22:37:29 -04:00
Object . assign ( credentials , newCredentials ) ;
if ( onCredentialsRefreshed ) {
try { await onCredentialsRefreshed ( newCredentials ) ; } catch ( e ) { log ? . warn ? . ( "TOKEN" , ` onCredentialsRefreshed failed: ${ e . message } ` ) ; }
}
try {
const retryResult = await executor . execute ( { model , body : translatedBody , stream , credentials , signal : streamController . signal , log , proxyOptions } ) ;
2026-07-20 04:39:17 -04:00
if ( retryResult . response . ok ) {
providerResponse = retryResult . response ;
providerUrl = retryResult . url ;
providerResponseFormat = retryResult . responseFormat || targetFormat ;
}
2026-03-13 22:37:29 -04:00
} catch { log ? . warn ? . ( "TOKEN" , ` ${ provider . toUpperCase ( ) } | retry after refresh failed ` ) ; }
} else {
log ? . warn ? . ( "TOKEN" , ` ${ provider . toUpperCase ( ) } | refresh failed ` ) ;
}
} catch ( e ) {
log ? . warn ? . ( "TOKEN" , ` ${ provider . toUpperCase ( ) } | refresh threw: ${ e . message } ` ) ;
2026-01-04 22:37:09 -05:00
}
}
2026-02-26 23:15:12 -05:00
// Provider returned error
2026-01-04 22:37:09 -05:00
if ( ! providerResponse . ok ) {
2026-02-21 04:42:46 -05:00
trackPendingRequest ( model , provider , connectionId , false , true ) ;
2026-04-24 00:36:16 -04:00
const { statusCode , message , resetsAtMs } = await parseUpstreamError ( providerResponse , executor ) ;
2026-05-13 11:40:42 -04:00
appendRequestLog ( { model , provider , connectionId , status : ` FAILED ${ statusCode } ` } ) . catch ( ( ) => { } ) ;
2026-02-26 23:15:12 -05:00
saveRequestDetail ( buildRequestDetail ( {
provider , model , connectionId ,
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
latency : { ttft : 0 , total : Date . now ( ) - requestStartTime } ,
tokens : { prompt _tokens : 0 , completion _tokens : 0 } ,
request : extractRequestConfig ( body , stream ) ,
providerRequest : finalBody || translatedBody || null ,
2026-02-26 23:15:12 -05:00
response : { error : message , status : statusCode , thinking : null } ,
2026-07-10 05:10:16 -04:00
pxpipe : pxpipeSummary ,
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
status : "error"
2026-05-13 11:40:42 -04:00
} ) ) . catch ( ( ) => { } ) ;
Feature/ai observability dashboard (#79)
* feat: add AI request details feature with latency tracking
Add comprehensive request history and debugging capability to the Usage dashboard:
**Storage Layer** (usageDb.js):
- Add saveRequestDetail() for storing full request/response details
- Implement FIFO queue with 1000-record limit in request-details.json
- Auto-sanitize sensitive headers (authorization, api-key, cookie, token)
- Add getRequestDetails() with pagination and filtering support
- Add getRequestDetailById() for single record lookup
**Pipeline Integration** (chatCore.js):
- Track request start time and calculate total latency
- Record TTFT (Time To First Token) and total latency for all requests
- Capture full request details (messages, model, parameters)
- Save response content for non-streaming, mark streaming responses
- Handle error cases with detailed error information
- Async non-blocking saves to avoid impacting request performance
**API Layer** (/api/usage/request-details):
- GET endpoint with pagination (page, pageSize: 1-100)
- Filter by provider, model, connectionId, status, date range
- Returns { details: [...], pagination: {...} } format
**UI Components**:
- Drawer.js: Right slide-out panel with backdrop blur and ESC close
- Pagination.js: Full pagination with page size selector (10/20/50)
- RequestDetailsTab.js: Complete table view with filters and detail drawer
**Dashboard Integration**:
- Add "Details" tab to Usage page (4th tab after Overview/Logger/Limits)
- Table columns: Timestamp, Model, Provider, Input Tokens, Output Tokens, Latency (TTFT/Total), Action
- Provider filter dropdown (9 providers supported)
- Date range filters (start/end datetime)
- Click "Detail" button to view full request/response JSON in slide-out drawer
**Features**:
- Real-time latency monitoring (TTFT & Total)
- Complete request/response inspection for debugging
- Filterable and searchable request history
- Responsive design with mobile-friendly filters
- Data security with automatic header sanitization
- Performance: async saves don't block request pipeline
**Files Created/Modified**:
- src/lib/usageDb.js (modified)
- open-sse/handlers/chatCore.js (modified)
- src/app/api/usage/request-details/route.js (new)
- src/shared/components/Drawer.js (new)
- src/shared/components/Pagination.js (new)
- src/app/(dashboard)/dashboard/usage/components/RequestDetailsTab.js (new)
- src/app/(dashboard)/dashboard/usage/page.js (modified)
Closes: AI Observability Dashboard feature
* feat: enhance request details with full config and streaming content capture
Improve Request Details feature to capture comprehensive request parameters
and actual streaming response content:
**Request Configuration Enhancement** (chatCore.js):
- Add extractRequestConfig() helper function to capture all request parameters
- Include temperature controls: temperature, top_p, top_k
- Include token limits: max_tokens, max_completion_tokens
- Include thinking/reasoning modes: thinking, reasoning, enable_thinking
- Include OpenAI parameters: presence_penalty, frequency_penalty, seed, stop,
tools, tool_choice, response_format, n, logprobs, top_logprobs, logit_bias,
user, parallel_tool_calls, prediction, store, metadata
- Apply to all request types: non-streaming, streaming, and error cases
**Streaming Content Capture** (chatCore.js & stream.js):
- Add onStreamComplete callback mechanism to stream processors
- Accumulate content from all formats: OpenAI, Claude, Gemini
- Track content from delta.content, delta.reasoning_content, delta.text,
delta.thinking, and Gemini content.parts
- Save initial record with "[Streaming in progress...]" marker
- Update record with actual content when stream completes
- Include usage tokens when available from stream
**Files Modified**:
- open-sse/handlers/chatCore.js - extractRequestConfig() + streaming capture
- open-sse/utils/stream.js - onStreamComplete callback + content accumulation
**Benefits**:
- View complete request configuration in Request Details (thinking mode, etc.)
- See actual streaming response content instead of placeholder
- Better debugging and observability for AI requests
Refs: #request-details-enhancement
* feat: separate thinking/reasoning content from response content
Improve Request Details to display thinking process separately from final response:
**Backend Changes**:
- stream.js: Capture content and thinking separately in streaming mode
- Add accumulatedThinking variable alongside accumulatedContent
- Route delta.content to content, delta.reasoning_content to thinking
- Support OpenAI (reasoning_content), Claude (thinking), Gemini (part.thought)
- Update onStreamComplete callback to return { content, thinking } object
- chatCore.js: Update response structure to include thinking field
- Non-streaming: Extract thinking from reasoning_content field
- Streaming: Receive { content, thinking } from stream callback
- Error responses: Include thinking: null
- Initial streaming save: Include thinking: null
**Frontend Changes**:
- RequestDetailsTab.js: Display thinking and content in separate sections
- Add amber/yellow themed "Thinking Process" section with psychology icon
- Show "Final Response" label when thinking is present
- Use distinct visual styling for thinking (amber bg) vs content (gray bg)
- Only show thinking section when thinking content exists
**Benefits**:
- Users can clearly see model's reasoning process vs final answer
- Better debugging for models with thinking capabilities (Claude, o1, etc.)
- Visual distinction makes it easy to identify thinking vs response
Refs: #thinking-content-separation
* fix: map Claude thinking to reasoning_content field
Fix Claude thinking content to be properly captured as reasoning_content
instead of regular content, enabling separate display in Request Details:
**Changes**:
- claude-to-openai.js: Use reasoning_content field for thinking blocks
- thinking start: send { reasoning_content: "" } instead of { content: "```\n```" }
- thinking delta: map to reasoning_content instead of content
- thinking stop: send { reasoning_content: "" } instead of { content: "```\n```" }
**Why This Matters**:
- Previously Claude thinking was sent as `content` field, mixed with actual response
- Now thinking uses `reasoning_content` field, matching OpenAI's o1 format
- stream.js can now properly route thinking to accumulatedThinking variable
- Request Details UI will show Claude thinking in separate "Thinking Process" section
**Supported Thinking Formats**:
- OpenAI: delta.reasoning_content → thinking
- Claude: delta.thinking → reasoning_content (now fixed)
- Gemini: part.thought === true → thinking
Refs: #claude-thinking-fix
* feat(observability): capture and display full 4-layer request chain
Capture complete request/response chain in AI Request Details:
- Add providerRequest field (translated request sent to provider)
- Add providerResponse field (raw provider response, streaming indicator)
- Update chatCore.js at all 5 saveRequestDetail() call sites
- Reorganize UI into 4 collapsible sections with Material icons
- Preserve backward compatibility for old records
- Add distinct styling for streaming indicator
* fix(observability): resolve React duplicate key warning in request details table
- Use composite key (detail.id + index) to ensure unique keys
- Prevents React warnings when database contains duplicate IDs from old ID generation
* fix(observability): display actual content in streaming request details
Change providerResponse field for streaming requests from placeholder
"[Streaming - raw response not captured]" to actual final content.
This improves debugging experience by showing the real AI response
in the "Provider Response (Raw)" section instead of a confusing
placeholder message.
Files changed:
- open-sse/handlers/chatCore.js: Save contentObj.content to providerResponse
- src/app/.../RequestDetailsTab.js: Remove special handling for placeholder
* refactor(observability): migrate request details to SQLite for improved concurrency
- Replace LowDB JSON storage with better-sqlite3
- Enable WAL mode for true concurrent read/write support
- Add 5 indexes to accelerate queries (timestamp, provider, model, connection_id, status)
- Perform pagination at the database level to reduce memory footprint
- Maintain 1000 record limit with automatic cleanup of old data
- Ensure API compatibility via re-exports, requiring no caller changes
Performance improvements:
- Concurrent Writes: Lock-free WAL mode prevents data contention
- Query Efficiency: Index-based searches replace full dataset loading
- Data Integrity: Atomic operations prevent file corruption
* fix(observability): resolve pagination statistics display issues
- Fix issue where totalItems=0 showed 'Showing 1 to 0 of 0 results'
- Hide pagination controls when totalItems=0 or totalPages<=1
- Standardize API response fields: pagination.total -> pagination.totalItems
Before: Incorrect stats shown for empty data, and pager visible even for single-page results
After: Stats hidden for empty data, pager hidden when navigation is unnecessary
* feat(observability): display friendly provider names in request details
- Add /api/usage/providers endpoint to dynamically fetch provider list with names
- Replace hardcoded provider options with dynamic loading from database
- Display friendly provider names instead of IDs in both table and detail drawer
- Support custom provider nodes (e.g., OpenAI-compatible) with user-defined names
- Add provider name caching to optimize performance
* fix(observability): use INSERT OR REPLACE for request details to handle streaming updates
* fix(observability): resolve zero-token display issue by ensuring streaming usage capture and fixing key mismatch
* fix(observability): separate TTFT and total latency calculation for streaming requests
* feat(observability): implement SQLite write queue and JSON size limits
- Added in-memory buffer and batch writing for SQLite to prevent lock contention
- Implemented with configurable 1MB limit to prevent DB bloat
- Added dashboard UI for observability performance and data management settings
- Integrated graceful shutdown handlers to prevent data loss
* fix(observability): resolve ReferenceError by declaring dbInstance
2026-02-08 22:30:42 -05:00
2026-01-04 22:57:45 -05:00
const errMsg = formatProviderError ( new Error ( message ) , provider , model , statusCode ) ;
2026-07-10 07:01:20 -04:00
if ( log ? . errorLine ) {
const urlStr = providerUrl ? ` \n URL: ${ providerUrl } ` : "" ;
log . errorLine ( reqTag , "✗" , ` ERROR ${ statusCode } · ${ provider } / ${ model } · ${ Date . now ( ) - requestStartTime } ms ${ urlStr } \n ${ errMsg } ` ) ;
}
2026-01-11 09:45:01 -05:00
reqLogger . logError ( new Error ( message ) , finalBody || translatedBody ) ;
2026-04-24 00:36:16 -04:00
return createErrorResult ( statusCode , errMsg , resetsAtMs ) ;
2026-01-04 22:37:09 -05:00
}
2026-07-10 07:01:20 -04:00
const sharedCtx = { provider , model , body , stream , translatedBody , finalBody , requestStartTime , connectionId , apiKey , clientRawRequest , onRequestSuccess , pxpipe : pxpipeSummary , reqTag , log } ;
2026-05-13 11:40:42 -04:00
const appendLog = ( extra ) => appendRequestLog ( { model , provider , connectionId , ... extra } ) . catch ( ( ) => { } ) ;
2026-02-26 23:15:12 -05:00
const trackDone = ( ) => trackPendingRequest ( model , provider , connectionId , false ) ;
2026-02-22 09:44:11 -05:00
2026-02-26 23:15:12 -05:00
// Provider forced streaming but client wants JSON
if ( ! clientRequestedStreaming && providerRequiresStreaming ) {
2026-08-05 02:23:03 -04:00
const result = await handleForcedSSEToJson ( { ... sharedCtx , providerResponse , sourceFormat , targetFormat : providerResponseFormat , customToolNames , trackDone , appendLog } ) ;
2026-03-13 22:37:29 -04:00
if ( result ) { streamController . handleComplete ( ) ; return result ; }
2026-02-14 23:47:55 -05:00
}
2026-02-26 23:15:12 -05:00
// True non-streaming response
2026-01-04 22:37:09 -05:00
if ( ! stream ) {
2026-08-05 02:23:03 -04:00
const result = await handleNonStreamingResponse ( { ... sharedCtx , providerResponse , sourceFormat , targetFormat : providerResponseFormat , reqLogger , toolNameMap , customToolNames , trackDone , appendLog } ) ;
2026-03-13 22:37:29 -04:00
streamController . handleComplete ( ) ;
return result ;
2026-01-04 22:37:09 -05:00
}
// Streaming response
2026-07-03 04:07:20 -04:00
const { onStreamComplete , streamDetailId } = buildOnStreamComplete ( { ... sharedCtx } ) ;
2026-08-05 02:23:03 -04:00
return handleStreamingResponse ( { ... sharedCtx , providerResponse , sourceFormat , targetFormat : providerResponseFormat , userAgent , reqLogger , toolNameMap , customToolNames , streamController , onStreamComplete , streamDetailId } ) ;
2026-01-04 22:37:09 -05:00
}
export function isTokenExpiringSoon ( expiresAt , bufferMs = 5 * 60 * 1000 ) {
if ( ! expiresAt ) return false ;
2026-02-26 23:15:12 -05:00
return new Date ( expiresAt ) . getTime ( ) - Date . now ( ) < bufferMs ;
2026-02-08 04:45:31 -05:00
}