Plugin providers are registered from a plugin's `config` hook and credentialed from its `auth` loader, both inside the running OpenCode process. Nothing about them reaches `opencode.json` or `auth.json`, so resolution that only reads files could not see them: selecting such a model failed with "has no known API base URL" while the same model worked in chat (#2666). `GET /provider` is where that state is visible. A new `runtime-providers` module keeps one cached snapshot of it and reports, per provider, the credential and endpoint OpenCode itself resolved. Credential resolution becomes config -> runtime -> auth.json, and endpoint resolution config -> openai default -> runtime -> models.dev catalog. Providers with a dedicated wire format (Copilot, ChatGPT-plan OpenAI, Anthropic, Google) are excluded from the runtime credential: for them OpenCode reports an OAuth access token that their real transport does not accept. opencode zen is excluded when the user has no zen login. OpenCode then reports the sentinel `apiKey: "public"` and trims its catalog to free models that run on its own infrastructure; the sentinel is never read as a credential. Claude Code stays refused for background actions even when a plugin publishes an OpenAI-compatible endpoint for it, because that endpoint is a facade over the Claude Agent SDK and spawns the CLI per request. No capability probe. Asking `GET /models` does identify a plugin whose protocol lives in its own `fetch`, but measured across the 166 providers with an `api` URL in the models.dev catalog it also denies six that work and simply have no `/models` route. A provider that vanishes from the picker explains nothing, while one that fails on use says why, so availability stops at credential and endpoint. The same list drives the Small Model and Changes Walkthrough pickers. Validated against a real OpenCode with four plugin providers loaded: offered providers went from 3 to 7, zen and Claude Code stayed out, and a generation through a plugin-backed model that previously failed now returns.
327 lines
13 KiB
JavaScript
327 lines
13 KiB
JavaScript
import fs from 'fs';
|
|
import os from 'os';
|
|
import path from 'path';
|
|
import { readAuthFile } from '../opencode/auth.js';
|
|
import { readConfigLayers } from '../opencode/shared.js';
|
|
import { getModelCatalog } from './catalog.js';
|
|
import { resolveSmallModel, parseModelRef, isUsableAuthEntry, getAuthEntryForProvider } from './resolve.js';
|
|
import { DEDICATED_WIRE_FORMAT_PROVIDERS, callSmallModel, resolveProviderLogin } from './call.js';
|
|
import { getRuntimeProviderSnapshot } from './runtime-providers.js';
|
|
|
|
// Never a small model, whatever the transport looks like. A plugin can publish
|
|
// an OpenAI-compatible endpoint for Claude Code, but it is a façade over the
|
|
// Claude Agent SDK, which spawns the Claude Code CLI per request and spends
|
|
// the user's Claude subscription rate limit. Paying that for a session title
|
|
// or a summary is the wrong trade, so the refusal is unconditional rather than
|
|
// conditional on an endpoint existing.
|
|
const CLAUDE_CODE_PROVIDER = 'claude-code';
|
|
|
|
const OPENCHAMBER_SETTINGS_FILE = path.join(
|
|
process.env.OPENCHAMBER_DATA_DIR
|
|
? path.resolve(process.env.OPENCHAMBER_DATA_DIR)
|
|
: path.join(os.homedir(), '.config', 'openchamber'),
|
|
'settings.json',
|
|
);
|
|
|
|
// OpenChamber's own settings: when the user unchecks "use default small model"
|
|
// their explicit override outranks every other resolution step.
|
|
const readSmallModelSettingsOverride = () => {
|
|
try {
|
|
const raw = fs.readFileSync(OPENCHAMBER_SETTINGS_FILE, 'utf8');
|
|
const settings = JSON.parse(raw);
|
|
if (!settings || typeof settings !== 'object') return null;
|
|
if (settings.smallModelUseDefault !== false) return null;
|
|
const override = typeof settings.smallModelOverride === 'string' ? settings.smallModelOverride.trim() : '';
|
|
return override || null;
|
|
} catch {
|
|
return null;
|
|
}
|
|
};
|
|
|
|
// Rough safety clamp so a huge input never blows the model's context window.
|
|
// Token estimate is ~4 chars/token; when the catalog has no limit for the
|
|
// model (Copilot/codex utility models are not listed) a conservative default
|
|
// applies.
|
|
const DEFAULT_CONTEXT_TOKENS = 64_000;
|
|
const OUTPUT_RESERVE_TOKENS = 4_000;
|
|
|
|
/**
|
|
* Input budget in characters, given how much of the context the caller intends
|
|
* to leave for the answer. The reserve must match the output budget the caller
|
|
* will actually request, or the two disagree and the model overruns its context.
|
|
*/
|
|
export const getModelInputCharBudget = ({ catalog, providerID, modelID, outputReserveTokens }) => {
|
|
const limit = catalog?.[providerID]?.models?.[modelID]?.limit;
|
|
const known = Number(limit?.context) > 0;
|
|
const contextTokens = known ? Number(limit.context) : DEFAULT_CONTEXT_TOKENS;
|
|
const reserve = Number(outputReserveTokens) > 0 ? Number(outputReserveTokens) : OUTPUT_RESERVE_TOKENS;
|
|
const inputBudgetTokens = Math.max(1_000, contextTokens - reserve);
|
|
return { maxChars: inputBudgetTokens * 4, contextTokens, contextKnown: known };
|
|
};
|
|
|
|
/**
|
|
* The output budget to actually request: what the caller asked for, capped by
|
|
* what the model admits it can emit. Asking for more than `limit.output` is
|
|
* rejected outright by some providers and silently ignored by others.
|
|
*/
|
|
const resolveOutputTokens = ({ catalog, providerID, modelID, maxOutputTokens }) => {
|
|
const requested = Number(maxOutputTokens) > 0 ? Number(maxOutputTokens) : 0;
|
|
if (!requested) return undefined;
|
|
const limit = Number(catalog?.[providerID]?.models?.[modelID]?.limit?.output);
|
|
return limit > 0 ? Math.min(requested, limit) : requested;
|
|
};
|
|
|
|
// `truncate` keeps the historical behavior for callers whose prompt losing its
|
|
// tail is survivable (summaries, commit messages). `error` is for callers whose
|
|
// output would be quietly wrong on a clipped input — they need the failure.
|
|
const clampPromptToModelLimit = ({ prompt, catalog, providerID, modelID, onOverflow, outputReserveTokens }) => {
|
|
const { maxChars } = getModelInputCharBudget({ catalog, providerID, modelID, outputReserveTokens });
|
|
if (prompt.length <= maxChars) {
|
|
return { prompt, truncated: false };
|
|
}
|
|
if (onOverflow === 'error') {
|
|
throw Object.assign(
|
|
new Error(`Input is too large for ${providerID}/${modelID}: ${prompt.length} characters exceeds the ${maxChars} the model's context allows`),
|
|
{ statusCode: 413, code: 'context-too-small', providerID, modelID, requiredChars: prompt.length, availableChars: maxChars },
|
|
);
|
|
}
|
|
return { prompt: `${prompt.slice(0, maxChars)}…`, truncated: true };
|
|
};
|
|
|
|
const readConfiguredSmallModel = (workingDirectory) => {
|
|
try {
|
|
const { mergedConfig } = readConfigLayers(workingDirectory);
|
|
const value = mergedConfig?.small_model;
|
|
return typeof value === 'string' ? value : null;
|
|
} catch {
|
|
return null;
|
|
}
|
|
};
|
|
|
|
/**
|
|
* Generates text with the user's small model, resolved and authenticated
|
|
* entirely server-side from the OpenCode config and auth store.
|
|
*/
|
|
export async function generateSmallModelText({ prompt, system, maxOutputTokens, model, directory, preferredProviderID, preferredModelID, restrictToPreferredProvider = false, responseSchema, timeoutMs, signal, onOverflow = 'truncate' }) {
|
|
if (typeof prompt !== 'string' || !prompt.trim()) {
|
|
throw Object.assign(new Error('prompt is required'), { statusCode: 400 });
|
|
}
|
|
|
|
const auth = readAuthFile();
|
|
const catalog = await getModelCatalog().catch(() => ({}));
|
|
|
|
const explicit = parseModelRef(model);
|
|
const resolved = explicit
|
|
? { ...explicit, source: 'request' }
|
|
: resolveSmallModel({
|
|
auth,
|
|
catalog,
|
|
settingsSmallModel: readSmallModelSettingsOverride(),
|
|
configSmallModel: readConfiguredSmallModel(directory),
|
|
preferredProviderID,
|
|
preferredModelID,
|
|
});
|
|
|
|
if (!resolved) {
|
|
throw Object.assign(
|
|
new Error('No small model available — no authenticated provider has a suitable model'),
|
|
{ statusCode: 404 },
|
|
);
|
|
}
|
|
|
|
if (resolved.providerID === CLAUDE_CODE_PROVIDER) {
|
|
throw Object.assign(
|
|
new Error('Claude Code cannot be used for background small-model actions. Choose another Small Model in Settings → Sessions.'),
|
|
{ statusCode: 422, code: 'small-model-provider-unsupported' },
|
|
);
|
|
}
|
|
|
|
// Callers with a session context can forbid silently switching providers:
|
|
// an explicit user choice (settings override, opencode config, request
|
|
// model) is always allowed, anything else must stay on the session's
|
|
// provider.
|
|
if (restrictToPreferredProvider
|
|
&& !['settings', 'config', 'request'].includes(resolved.source)
|
|
&& resolved.providerID !== preferredProviderID) {
|
|
throw Object.assign(
|
|
new Error('No small model available within the session provider'),
|
|
{ statusCode: 404 },
|
|
);
|
|
}
|
|
|
|
const outputTokens = resolveOutputTokens({
|
|
catalog,
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
maxOutputTokens,
|
|
});
|
|
|
|
const clamped = clampPromptToModelLimit({
|
|
prompt: prompt.trim(),
|
|
catalog,
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
onOverflow,
|
|
outputReserveTokens: outputTokens,
|
|
});
|
|
|
|
const text = await callSmallModel({
|
|
auth,
|
|
catalog,
|
|
workingDirectory: directory,
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
prompt: clamped.prompt,
|
|
system: typeof system === 'string' && system.trim() ? system.trim() : undefined,
|
|
maxOutputTokens: outputTokens,
|
|
responseSchema,
|
|
timeoutMs,
|
|
signal,
|
|
});
|
|
|
|
return {
|
|
text: text.trim(),
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
source: resolved.source,
|
|
...(clamped.truncated ? { inputTruncated: true } : {}),
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Provider ids the small model can actually call — an auth.json login, or a
|
|
* credential and endpoint the running OpenCode resolved for a plugin. Used by
|
|
* the Small Model and Changes Walkthrough pickers to hide providers that would
|
|
* only ever fail (e.g. opencode free models without a token).
|
|
*/
|
|
export async function listAuthenticatedProviders() {
|
|
try {
|
|
const auth = readAuthFile();
|
|
const ids = new Set(
|
|
Object.keys(auth || {}).filter((providerID) => isUsableAuthEntry(auth[providerID])),
|
|
);
|
|
// The catalog id is github-copilot while legacy auth entries may sit
|
|
// under the copilot alias.
|
|
if (isUsableAuthEntry(getAuthEntryForProvider(auth, 'github-copilot'))) {
|
|
ids.add('github-copilot');
|
|
}
|
|
// Kept separate so a runtime lookup that goes wrong costs the providers it
|
|
// would have added, never the logins already established from disk.
|
|
try {
|
|
for (const providerID of await listRuntimeCallableProviders()) ids.add(providerID);
|
|
} catch {
|
|
// The auth.json set below stands on its own.
|
|
}
|
|
ids.delete(CLAUDE_CODE_PROVIDER);
|
|
return Array.from(ids);
|
|
} catch {
|
|
return [];
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Providers that only the running OpenCode knows about — plugin-registered
|
|
* ones, and any whose endpoint is resolved at startup.
|
|
*
|
|
* The test is the same one applied to an auth.json login: a credential we may
|
|
* use and somewhere to send it. Whether the endpoint answers the protocol we
|
|
* speak is not knowable from any field OpenCode reports, and guessing it wrong
|
|
* removes a working model from the picker with nothing to explain it.
|
|
*/
|
|
async function listRuntimeCallableProviders() {
|
|
const snapshot = await getRuntimeProviderSnapshot();
|
|
if (!snapshot) return [];
|
|
const ids = [];
|
|
for (const id of snapshot.connected) {
|
|
const provider = snapshot.providers.get(id);
|
|
// No credential we may use — including the zen sentinel, whose free models
|
|
// belong to OpenCode's own server.
|
|
if (!provider?.apiKey || !provider.baseURL) continue;
|
|
// Reached through a dedicated wire format and already covered by the
|
|
// auth.json scan above.
|
|
if (DEDICATED_WIRE_FORMAT_PROVIDERS.has(id)) continue;
|
|
ids.push(id);
|
|
}
|
|
return ids;
|
|
}
|
|
|
|
/**
|
|
* Reports which model would be used, without calling it.
|
|
*
|
|
* `inputCharBudget` and `structuredOutput` let callers refuse work before
|
|
* spending a request: the walkthrough needs both a big enough context and
|
|
* schema-shaped output, and would rather tell the user to pick another model
|
|
* than send a doomed prompt. `structuredOutput` is deliberately tri-state —
|
|
* the catalog omits the field for roughly half of all models (aggregators and
|
|
* proxies especially), and treating "unknown" as "unsupported" would hide
|
|
* models that work fine.
|
|
*/
|
|
/**
|
|
* The reserve, resolved against the model that was actually picked.
|
|
*
|
|
* A caller that wants "as much answer room as this model allows" cannot state a
|
|
* number up front — it does not know which model it will get. Passing a
|
|
* function lets it decide once the limits are known, and keeps the reserve and
|
|
* the eventual request the same number by construction.
|
|
*/
|
|
const resolveReserveTokens = (outputReserveTokens, limits) => (
|
|
typeof outputReserveTokens === 'function' ? outputReserveTokens(limits) : outputReserveTokens
|
|
);
|
|
|
|
export async function describeSmallModel({ directory, preferredProviderID, preferredModelID, outputReserveTokens, overrideModel } = {}) {
|
|
const auth = readAuthFile();
|
|
const catalog = await getModelCatalog().catch(() => ({}));
|
|
// A caller with its own model setting (the diff walkthrough) outranks the
|
|
// small-model chain entirely — it asked for this model on purpose.
|
|
const explicit = parseModelRef(overrideModel);
|
|
const resolved = explicit
|
|
? { ...explicit, source: 'request' }
|
|
: resolveSmallModel({
|
|
auth,
|
|
catalog,
|
|
settingsSmallModel: readSmallModelSettingsOverride(),
|
|
configSmallModel: readConfiguredSmallModel(directory),
|
|
preferredProviderID,
|
|
preferredModelID,
|
|
});
|
|
if (!resolved) return resolved;
|
|
|
|
const entry = catalog?.[resolved.providerID]?.models?.[resolved.modelID];
|
|
const outputTokenLimit = Number(entry?.limit?.output) > 0 ? Number(entry.limit.output) : null;
|
|
// Two passes: the first only to learn the context, which a caller-supplied
|
|
// reserve function needs before it can answer.
|
|
const { contextTokens, contextKnown } = getModelInputCharBudget({
|
|
catalog,
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
});
|
|
const reserveTokens = resolveReserveTokens(outputReserveTokens, { contextTokens, outputTokenLimit });
|
|
const { maxChars } = getModelInputCharBudget({
|
|
catalog,
|
|
providerID: resolved.providerID,
|
|
modelID: resolved.modelID,
|
|
outputReserveTokens: reserveTokens,
|
|
});
|
|
|
|
// Settings/config/request overrides can name a provider with no usable login.
|
|
// Report that here so readiness can refuse before the user pays for a 401.
|
|
const hasLogin = Boolean(await resolveProviderLogin({
|
|
auth,
|
|
workingDirectory: directory,
|
|
providerID: resolved.providerID,
|
|
}));
|
|
|
|
return {
|
|
...resolved,
|
|
hasLogin,
|
|
inputCharBudget: maxChars,
|
|
contextTokens,
|
|
contextKnown,
|
|
// What the caller should ask for, so the request and the reserve above
|
|
// cannot drift apart.
|
|
outputTokens: Number(reserveTokens) > 0 ? Number(reserveTokens) : null,
|
|
structuredOutput: typeof entry?.structured_output === 'boolean' ? entry.structured_output : null,
|
|
outputTokenLimit,
|
|
};
|
|
}
|