Add Self-hosted STT/TTS/Embedding providers that read baseUrl per connection instead of a fixed registry endpoint, so 9Router can point at whisper.cpp, faster-whisper, Kokoro-FastAPI, llama-server, vLLM, Infinity, and similar OpenAI-compatible local servers. Self-hosted Embedding refuses to run without a baseUrl rather than falling back to api.openai.com like openaiCompatNode does, since that fallback would silently send input text and the API key to OpenAI under a provider named "Self-hosted". Also fixes embeddingsCore to catch adapter build errors as a 400 instead of letting them escape uncaught, and bounds the upstream fetch with FETCH_CONNECT_TIMEOUT_MS to avoid hanging forever on a dead endpoint. Self-hosted TTS treats a bare model value as the model rather than the voice, since the generic OpenAI TTS convention (bare = voice) is backwards for a provider where the model is the variable part.
56 lines
2.2 KiB
JavaScript
56 lines
2.2 KiB
JavaScript
// TTS provider registry
|
|
import googleTts from "./googleTts.js";
|
|
import edgeTts, { fetchEdgeTtsVoices } from "./edgeTts.js";
|
|
import localDevice, { fetchLocalDeviceVoices } from "./localDevice.js";
|
|
import elevenlabs, { fetchElevenLabsVoices } from "./elevenlabs.js";
|
|
import openai from "./openai.js";
|
|
import openrouter from "./openrouter.js";
|
|
import gemini, { fetchGeminiVoices } from "./gemini.js";
|
|
import xiaomiMimo from "./xiaomi-mimo.js";
|
|
import selfhostedTts from "./selfhostedTts.js";
|
|
import { FORMAT_HANDLERS } from "./genericFormats.js";
|
|
import { parseModelVoice } from "./_base.js";
|
|
|
|
// Special providers with custom synthesize() logic
|
|
const SPECIAL_ADAPTERS = {
|
|
"google-tts": googleTts,
|
|
"edge-tts": edgeTts,
|
|
"local-device": localDevice,
|
|
elevenlabs,
|
|
openai,
|
|
openrouter,
|
|
gemini,
|
|
"xiaomi-mimo": xiaomiMimo,
|
|
"selfhosted-tts": selfhostedTts,
|
|
};
|
|
|
|
export function getTtsAdapter(provider) {
|
|
return SPECIAL_ADAPTERS[provider] || null;
|
|
}
|
|
|
|
// Generic config-driven dispatcher (uses ttsConfig.format)
|
|
export async function synthesizeViaConfig(provider, text, model, credentials) {
|
|
const { AI_PROVIDERS } = await import("@/shared/constants/providers");
|
|
const cfg = AI_PROVIDERS[provider]?.ttsConfig;
|
|
if (!cfg) return null;
|
|
const handler = FORMAT_HANDLERS[cfg.format];
|
|
if (!handler) return null;
|
|
const apiKey = credentials?.apiKey;
|
|
if (cfg.authType !== "none" && !apiKey) throw new Error(`${provider} API key required`);
|
|
const { PROVIDER_MODELS } = await import("open-sse/config/providerModels.js");
|
|
const ttsModels = (PROVIDER_MODELS[provider] || []).filter(m => (m.kind || m.type) === "tts");
|
|
const defaultModel = ttsModels[0]?.id || "";
|
|
const { modelId, voiceId } = parseModelVoice(model, defaultModel, "", ttsModels);
|
|
return handler({ baseUrl: cfg.baseUrl, apiKey, text, modelId, voiceId });
|
|
}
|
|
|
|
// Voice fetchers (used by /api/media-providers/tts/voices route)
|
|
export const VOICE_FETCHERS = {
|
|
"edge-tts": fetchEdgeTtsVoices,
|
|
"local-device": fetchLocalDeviceVoices,
|
|
elevenlabs: fetchElevenLabsVoices,
|
|
gemini: fetchGeminiVoices,
|
|
};
|
|
|
|
// Re-export for backward compat
|
|
export { fetchEdgeTtsVoices, fetchLocalDeviceVoices, fetchElevenLabsVoices, fetchGeminiVoices };
|