mirror of
https://github.com/logancyang/obsidian-copilot.git
synced 2026-07-22 07:50:24 +00:00
560 lines
18 KiB
TypeScript
560 lines
18 KiB
TypeScript
import { BaseChatModelParams } from "@langchain/core/language_models/chat_models";
|
|
import { AIMessageChunk, BaseMessage } from "@langchain/core/messages";
|
|
import type { UsageMetadata } from "@langchain/core/messages";
|
|
import { ChatGenerationChunk } from "@langchain/core/outputs";
|
|
import { ChatOpenAI } from "@langchain/openai";
|
|
import OpenAI from "openai";
|
|
import { logInfo } from "@/logger";
|
|
|
|
type OpenRouterChatChunk = OpenAI.ChatCompletionChunk;
|
|
type OpenRouterUsage = NonNullable<OpenRouterChatChunk["usage"]>;
|
|
type OpenRouterMessageParam = OpenAI.ChatCompletionMessageParam;
|
|
|
|
/**
|
|
* ChatOpenRouter extends ChatOpenAI to support OpenRouter-specific features,
|
|
* particularly reasoning/thinking tokens.
|
|
*
|
|
* OpenRouter exposes thinking tokens via the `reasoning` request parameter
|
|
* and responds with `reasoning_details` in both streaming and non-streaming modes.
|
|
*
|
|
* @see https://openrouter.ai/docs/use-cases/reasoning-tokens
|
|
*/
|
|
export interface ChatOpenRouterInput extends BaseChatModelParams {
|
|
/**
|
|
* Enable reasoning/thinking tokens from OpenRouter
|
|
* When true, requests will include reasoning parameters
|
|
*/
|
|
enableReasoning?: boolean;
|
|
|
|
/**
|
|
* Reasoning effort level: "minimal", "low", "medium", "high", or "xhigh"
|
|
* Controls the amount of reasoning the model uses
|
|
* Note: "minimal" will be treated as "low" for OpenRouter
|
|
* Note: "xhigh" is only supported by GPT-5.4 models
|
|
*/
|
|
reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
|
|
/**
|
|
* Enable prompt caching (cache_control) for OpenRouter requests.
|
|
* Defaults to true. Set to false for Zero Data Retention (ZDR) endpoints
|
|
* that do not support prompt caching.
|
|
*/
|
|
enablePromptCaching?: boolean;
|
|
|
|
// All other ChatOpenAI parameters
|
|
modelName?: string;
|
|
apiKey?: string;
|
|
configuration?: {
|
|
baseURL?: string;
|
|
defaultHeaders?: Record<string, string>;
|
|
fetch?: (input: RequestInfo | URL, init?: RequestInit) => Promise<Response>;
|
|
[key: string]: unknown;
|
|
};
|
|
temperature?: number;
|
|
maxTokens?: number;
|
|
topP?: number;
|
|
frequencyPenalty?: number;
|
|
streaming?: boolean;
|
|
maxRetries?: number;
|
|
maxConcurrency?: number;
|
|
[key: string]: unknown;
|
|
}
|
|
|
|
export class ChatOpenRouter extends ChatOpenAI {
|
|
private enableReasoning: boolean;
|
|
private reasoningEffort?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
private enablePromptCaching: boolean;
|
|
private openaiClient: OpenAI;
|
|
/** True when the configured baseURL belongs to the OpenRouter gateway. */
|
|
private isOpenRouter: boolean;
|
|
|
|
constructor(fields: ChatOpenRouterInput) {
|
|
const {
|
|
enableReasoning = false,
|
|
reasoningEffort,
|
|
enablePromptCaching = true,
|
|
...rest
|
|
} = fields;
|
|
|
|
// Pass all other parameters to ChatOpenAI
|
|
super(rest);
|
|
|
|
this.enableReasoning = enableReasoning;
|
|
this.reasoningEffort = reasoningEffort;
|
|
this.enablePromptCaching = enablePromptCaching;
|
|
|
|
const baseURL = fields.configuration?.baseURL || "https://openrouter.ai/api/v1";
|
|
this.isOpenRouter = baseURL.includes("openrouter.ai");
|
|
|
|
// Create our own OpenAI client for raw access
|
|
this.openaiClient = new OpenAI({
|
|
apiKey: fields.apiKey,
|
|
baseURL,
|
|
defaultHeaders: fields.configuration?.defaultHeaders,
|
|
fetch: fields.configuration?.fetch,
|
|
dangerouslyAllowBrowser: true,
|
|
});
|
|
}
|
|
|
|
/**
|
|
* Override the invocation parameters to include prompt caching and reasoning when enabled.
|
|
*
|
|
* Prompt caching: The `cache_control` field opts Anthropic models (via OpenRouter) into
|
|
* automatic cache breakpoint detection, reducing token costs on repeated context. This
|
|
* field is only sent when connected to the OpenRouter gateway; other backends (LM Studio,
|
|
* Copilot Plus) use the same class but must not receive OpenRouter-specific fields.
|
|
*
|
|
* @see https://openrouter.ai/docs/features/prompt-caching
|
|
*/
|
|
override invocationParams(options?: this["ParsedCallOptions"]): Record<string, unknown> {
|
|
const baseParams = super.invocationParams(options);
|
|
|
|
// Only inject cache_control for OpenRouter endpoints. LM Studio, Copilot Plus, and
|
|
// other OpenAI-compatible backends share this class but reject unknown top-level fields.
|
|
// Skip caching when enablePromptCaching is false (e.g. for ZDR endpoints that don't
|
|
// support Anthropic's automatic caching).
|
|
const withCaching =
|
|
this.isOpenRouter && this.enablePromptCaching
|
|
? { ...baseParams, cache_control: { type: "ephemeral" } }
|
|
: baseParams;
|
|
|
|
// Add reasoning parameter if enabled
|
|
if (this.enableReasoning) {
|
|
// Per OpenRouter docs:
|
|
// - For Anthropic models: MUST use reasoning.max_tokens or reasoning.effort
|
|
// - For other models: Can use reasoning.enabled
|
|
// - max_tokens must be strictly higher than reasoning budget
|
|
|
|
// Prefer effort if provided, otherwise fall back to max_tokens
|
|
if (this.reasoningEffort) {
|
|
// Map "minimal" to "low" since OpenRouter doesn't support "minimal"
|
|
const effort = this.reasoningEffort === "minimal" ? "low" : this.reasoningEffort;
|
|
logInfo(`OpenRouter reasoning enabled with effort: ${effort}`);
|
|
return {
|
|
...withCaching,
|
|
reasoning: {
|
|
effort,
|
|
},
|
|
};
|
|
} else {
|
|
logInfo(`OpenRouter reasoning enabled with max_tokens: 1024`);
|
|
return {
|
|
...withCaching,
|
|
reasoning: {
|
|
max_tokens: 1024,
|
|
},
|
|
};
|
|
}
|
|
}
|
|
|
|
return withCaching;
|
|
}
|
|
|
|
/**
|
|
* Override to use raw OpenAI SDK to access reasoning_details
|
|
* LangChain filters out reasoning_details, so we bypass it completely
|
|
*/
|
|
override async *_streamResponseChunks(
|
|
messages: BaseMessage[],
|
|
options: this["ParsedCallOptions"],
|
|
_runManager?: { handleLLMNewToken: (token: string) => Promise<void> }
|
|
): AsyncGenerator<ChatGenerationChunk> {
|
|
const params = this.invocationParams(options);
|
|
const openaiMessages = this.toOpenRouterMessages(messages);
|
|
|
|
const stream = (await this.openaiClient.chat.completions.create({
|
|
...params,
|
|
messages: openaiMessages,
|
|
stream: true,
|
|
stream_options: {
|
|
...(params.stream_options ?? {}),
|
|
include_usage: true,
|
|
},
|
|
} as Parameters<
|
|
typeof this.openaiClient.chat.completions.create
|
|
>[0])) as unknown as AsyncIterable<OpenRouterChatChunk>;
|
|
|
|
let usageSummary: OpenRouterUsage | undefined;
|
|
|
|
for await (const rawChunk of stream) {
|
|
if (rawChunk.usage) {
|
|
usageSummary = rawChunk.usage;
|
|
}
|
|
|
|
const choice = rawChunk.choices?.[0];
|
|
const delta = choice?.delta;
|
|
if (!choice || !delta) {
|
|
continue;
|
|
}
|
|
|
|
const reasoningText = this.normalizeReasoningChunk(
|
|
(delta as Record<string, unknown>)?.reasoning
|
|
);
|
|
const reasoningDetails = this.extractReasoningDetails(choice);
|
|
const content = this.extractDeltaContent(delta.content);
|
|
|
|
const messageChunk = this.buildMessageChunk({
|
|
rawChunk,
|
|
delta: delta as unknown as Record<string, unknown>,
|
|
content,
|
|
finishReason: choice.finish_reason,
|
|
reasoningDetails,
|
|
reasoningText,
|
|
});
|
|
|
|
const generationChunk = new ChatGenerationChunk({
|
|
message: messageChunk,
|
|
text: typeof messageChunk.content === "string" ? messageChunk.content : "",
|
|
generationInfo: {
|
|
finish_reason: choice.finish_reason,
|
|
// Reason: system_fingerprint is marked deprecated by some scorecards but is still
|
|
// returned by OpenAI-style streaming APIs and is useful for telemetry.
|
|
system_fingerprint: rawChunk["system_fingerprint"],
|
|
model: rawChunk.model,
|
|
},
|
|
});
|
|
|
|
yield generationChunk;
|
|
if (generationChunk.text) {
|
|
await _runManager?.handleLLMNewToken(generationChunk.text);
|
|
}
|
|
}
|
|
|
|
if (usageSummary) {
|
|
yield this.buildUsageGenerationChunk(usageSummary);
|
|
}
|
|
|
|
if (options.signal?.aborted) {
|
|
throw new Error("AbortError");
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Convert LangChain messages to OpenRouter-ready messages.
|
|
*
|
|
* @param messages LangChain messages passed into the model
|
|
* @returns Messages formatted for the OpenRouter API
|
|
*/
|
|
private toOpenRouterMessages(messages: BaseMessage[]): OpenRouterMessageParam[] {
|
|
return messages.map((msg) => {
|
|
const msgRecord = msg as unknown as Record<string, unknown>;
|
|
const role =
|
|
typeof msg._getType === "function"
|
|
? msg._getType()
|
|
: ((msgRecord.role as string) ?? "user");
|
|
const mappedRole =
|
|
role === "human"
|
|
? "user"
|
|
: role === "ai"
|
|
? "assistant"
|
|
: (role as OpenAI.ChatCompletionRole);
|
|
|
|
if (msgRecord.tool_call_id) {
|
|
return {
|
|
role: "tool",
|
|
content: msg.content,
|
|
tool_call_id: msgRecord.tool_call_id as string,
|
|
} as OpenRouterMessageParam;
|
|
}
|
|
|
|
if (msg.additional_kwargs?.function_call) {
|
|
return {
|
|
role: mappedRole,
|
|
content: msg.content,
|
|
function_call: msg.additional_kwargs.function_call,
|
|
} as OpenRouterMessageParam;
|
|
}
|
|
|
|
// Handle modern tool_calls format (used by autonomous agent)
|
|
if (msg.additional_kwargs?.tool_calls) {
|
|
return {
|
|
role: mappedRole,
|
|
content: msg.content,
|
|
tool_calls: msg.additional_kwargs.tool_calls,
|
|
} as OpenRouterMessageParam;
|
|
}
|
|
|
|
return {
|
|
role: mappedRole,
|
|
content: msg.content,
|
|
} as OpenRouterMessageParam;
|
|
});
|
|
}
|
|
|
|
/**
|
|
* Build an `AIMessageChunk` enriched with reasoning metadata.
|
|
*
|
|
* @param config Chunk configuration values extracted from the stream
|
|
* @returns AI message chunk ready for downstream streaming utilities
|
|
*/
|
|
private buildMessageChunk(config: {
|
|
rawChunk: OpenRouterChatChunk;
|
|
delta: Record<string, unknown>;
|
|
content: string;
|
|
finishReason: string | null | undefined;
|
|
reasoningText?: string;
|
|
reasoningDetails?: unknown[];
|
|
}): AIMessageChunk {
|
|
const { rawChunk, delta, content, finishReason, reasoningText, reasoningDetails } = config;
|
|
const toolCallChunks = this.extractToolCallChunks(delta.tool_calls);
|
|
|
|
const additionalKwargs: Record<string, unknown> = {};
|
|
|
|
if (delta.function_call) {
|
|
additionalKwargs.function_call = delta.function_call;
|
|
}
|
|
|
|
if (Array.isArray(delta.tool_calls)) {
|
|
additionalKwargs.tool_calls = delta.tool_calls;
|
|
}
|
|
|
|
const deltaPayload: Record<string, unknown> = {};
|
|
if (reasoningText) {
|
|
deltaPayload.reasoning = reasoningText;
|
|
}
|
|
if (reasoningDetails && reasoningDetails.length > 0) {
|
|
deltaPayload.reasoning_details = reasoningDetails;
|
|
}
|
|
|
|
if (Object.keys(deltaPayload).length > 0) {
|
|
additionalKwargs.delta = {
|
|
...(additionalKwargs.delta as Record<string, unknown>),
|
|
...deltaPayload,
|
|
};
|
|
}
|
|
|
|
if (reasoningDetails && reasoningDetails.length > 0) {
|
|
additionalKwargs.reasoning_details = reasoningDetails;
|
|
}
|
|
|
|
const responseMetadata = this.buildResponseMetadata(rawChunk, finishReason);
|
|
|
|
return new AIMessageChunk({
|
|
content,
|
|
additional_kwargs: additionalKwargs,
|
|
tool_call_chunks: toolCallChunks,
|
|
response_metadata: responseMetadata,
|
|
id: rawChunk.id,
|
|
});
|
|
}
|
|
|
|
/**
|
|
* Normalize streamed reasoning payloads into plain text for the UI.
|
|
*
|
|
* @param reasoning Arbitrary reasoning payload returned by OpenRouter
|
|
* @returns Normalized reasoning text or undefined
|
|
*/
|
|
private normalizeReasoningChunk(reasoning: unknown): string | undefined {
|
|
if (!reasoning) {
|
|
return undefined;
|
|
}
|
|
|
|
if (typeof reasoning === "string") {
|
|
return reasoning;
|
|
}
|
|
|
|
if (Array.isArray(reasoning)) {
|
|
return reasoning
|
|
.map((item) => this.normalizeReasoningChunk(item))
|
|
.filter((item): item is string => Boolean(item))
|
|
.join("");
|
|
}
|
|
|
|
if (typeof reasoning === "object") {
|
|
const record = reasoning as Record<string, unknown>;
|
|
const candidates = [
|
|
record.output_text,
|
|
record.text,
|
|
record.reasoning,
|
|
record.thinking,
|
|
record.content,
|
|
];
|
|
|
|
const normalized = candidates.find((value) => typeof value === "string");
|
|
if (typeof normalized === "string") {
|
|
return normalized;
|
|
}
|
|
}
|
|
|
|
return undefined;
|
|
}
|
|
|
|
/**
|
|
* Extract reasoning details arrays from the streamed choice payload.
|
|
*
|
|
* @param choice Chunk choice object from the OpenRouter stream
|
|
* @returns Array of reasoning detail entries, if present
|
|
*/
|
|
private extractReasoningDetails(
|
|
choice: OpenAI.ChatCompletionChunk.Choice
|
|
): unknown[] | undefined {
|
|
const choiceRecord = choice as unknown as Record<string, Record<string, unknown>>;
|
|
const candidate =
|
|
choiceRecord?.delta?.reasoning_details ??
|
|
choiceRecord?.message?.reasoning_details ??
|
|
(choice as unknown as Record<string, unknown>)?.reasoning_details;
|
|
|
|
if (!Array.isArray(candidate)) {
|
|
return undefined;
|
|
}
|
|
|
|
return (candidate as unknown[]).filter((detail) => detail !== undefined && detail !== null);
|
|
}
|
|
|
|
/**
|
|
* Flatten OpenRouter delta content into a single text string.
|
|
*
|
|
* @param content Delta content payload
|
|
* @returns Text representation for downstream streaming
|
|
*/
|
|
private extractDeltaContent(content: unknown): string {
|
|
if (typeof content === "string") {
|
|
return content;
|
|
}
|
|
|
|
if (Array.isArray(content)) {
|
|
return content
|
|
.map((part) => {
|
|
if (typeof part === "string") {
|
|
return part;
|
|
}
|
|
if (
|
|
part &&
|
|
typeof part === "object" &&
|
|
typeof (part as { text?: unknown }).text === "string"
|
|
) {
|
|
return (part as { text: string }).text;
|
|
}
|
|
return "";
|
|
})
|
|
.join("");
|
|
}
|
|
|
|
return "";
|
|
}
|
|
|
|
/**
|
|
* Map raw OpenRouter tool call deltas into LangChain tool call chunks.
|
|
*
|
|
* @param toolCalls Tool call deltas returned by OpenRouter
|
|
* @returns Tool call chunk array compatible with LangChain
|
|
*/
|
|
private extractToolCallChunks(
|
|
toolCalls: unknown
|
|
):
|
|
| Array<{ name?: string; args?: string; id?: string; index?: number; type: "tool_call_chunk" }>
|
|
| undefined {
|
|
if (!Array.isArray(toolCalls)) {
|
|
return undefined;
|
|
}
|
|
|
|
return toolCalls.map((rawCall) => {
|
|
const call = rawCall as
|
|
| { function?: { name?: string; arguments?: string }; id?: string; index?: number }
|
|
| null
|
|
| undefined;
|
|
return {
|
|
name: call?.function?.name,
|
|
args: call?.function?.arguments,
|
|
id: call?.id,
|
|
index: call?.index,
|
|
type: "tool_call_chunk" as const,
|
|
};
|
|
});
|
|
}
|
|
|
|
/**
|
|
* Build response metadata payload with finish reason and usage info.
|
|
*
|
|
* @param rawChunk Raw streaming chunk from OpenRouter
|
|
* @param finishReason Stop reason reported by the model
|
|
* @returns Metadata object attached to each AI message chunk
|
|
*/
|
|
private buildResponseMetadata(
|
|
rawChunk: OpenRouterChatChunk,
|
|
finishReason: string | null | undefined
|
|
): Record<string, unknown> {
|
|
const metadata: Record<string, unknown> = {
|
|
model_provider: "openrouter",
|
|
};
|
|
|
|
if (finishReason) {
|
|
metadata.finish_reason = finishReason;
|
|
}
|
|
|
|
if (rawChunk.model) {
|
|
metadata.model = rawChunk.model;
|
|
}
|
|
|
|
// Reason: system_fingerprint is marked deprecated by some scorecards but is still
|
|
// returned by OpenAI-style streaming APIs and is useful for telemetry. Use bracket
|
|
// access to bypass JSDoc deprecation warnings.
|
|
const fingerprint = rawChunk["system_fingerprint"];
|
|
if (fingerprint) {
|
|
metadata.system_fingerprint = fingerprint;
|
|
}
|
|
|
|
if (rawChunk.usage) {
|
|
metadata.usage = { ...rawChunk.usage };
|
|
metadata.tokenUsage = {
|
|
promptTokens: rawChunk.usage.prompt_tokens,
|
|
completionTokens: rawChunk.usage.completion_tokens,
|
|
totalTokens: rawChunk.usage.total_tokens,
|
|
};
|
|
}
|
|
|
|
return metadata;
|
|
}
|
|
|
|
/**
|
|
* Create a terminal usage chunk so downstream consumers can capture token usage.
|
|
*
|
|
* @param usage Usage payload returned by the streaming API
|
|
* @returns Chat generation chunk containing usage metadata
|
|
*/
|
|
private buildUsageGenerationChunk(usage: OpenRouterUsage): ChatGenerationChunk {
|
|
const inputTokenDetails: Record<string, number> = {};
|
|
const outputTokenDetails: Record<string, number> = {};
|
|
|
|
const promptDetails = usage.prompt_tokens_details ?? {};
|
|
if (typeof promptDetails.audio_tokens === "number") {
|
|
inputTokenDetails.audio = promptDetails.audio_tokens;
|
|
}
|
|
if (typeof promptDetails.cached_tokens === "number") {
|
|
inputTokenDetails.cache_read = promptDetails.cached_tokens;
|
|
}
|
|
|
|
const completionDetails = usage.completion_tokens_details ?? {};
|
|
if (typeof completionDetails.audio_tokens === "number") {
|
|
outputTokenDetails.audio = completionDetails.audio_tokens;
|
|
}
|
|
if (typeof completionDetails.reasoning_tokens === "number") {
|
|
outputTokenDetails.reasoning = completionDetails.reasoning_tokens;
|
|
}
|
|
|
|
const usageMetadata: UsageMetadata = {
|
|
input_tokens: usage.prompt_tokens ?? 0,
|
|
output_tokens: usage.completion_tokens ?? 0,
|
|
total_tokens: usage.total_tokens ?? 0,
|
|
};
|
|
|
|
if (Object.keys(inputTokenDetails).length > 0) {
|
|
usageMetadata.input_token_details = inputTokenDetails;
|
|
}
|
|
|
|
if (Object.keys(outputTokenDetails).length > 0) {
|
|
usageMetadata.output_token_details = outputTokenDetails;
|
|
}
|
|
|
|
const messageChunk = new AIMessageChunk({
|
|
content: "",
|
|
response_metadata: { usage: { ...usage } },
|
|
usage_metadata: usageMetadata,
|
|
});
|
|
|
|
return new ChatGenerationChunk({
|
|
message: messageChunk,
|
|
text: "",
|
|
});
|
|
}
|
|
}
|