|
@@ -21,8 +21,14 @@ import type { CredentialRef } from '@deepseek-ai/dsh-credentials'
|
|
|
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
|
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
|
|
import { resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
|
import { resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
|
|
import type { ResolvedRetryPolicy, RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
|
|
import type { ResolvedRetryPolicy, RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
|
|
|
-import { resolveRouteModels, SUPPORTED_THINKING_FORMATS, THINKING_LEVELS } from './catalog.ts'
|
|
|
|
|
-import type { PiAiCompatProfile, PiAiModelOverride, PiAiModelProfile, PiAiReasoningEfforts } from './catalog.ts'
|
|
|
|
|
|
|
+import { MODALITIES, resolveRouteModels, SUPPORTED_THINKING_FORMATS, THINKING_LEVELS } from './catalog.ts'
|
|
|
|
|
+import type {
|
|
|
|
|
+ PiAiCompatProfile,
|
|
|
|
|
+ PiAiModality,
|
|
|
|
|
+ PiAiModelOverride,
|
|
|
|
|
+ PiAiModelProfile,
|
|
|
|
|
+ PiAiReasoningEfforts,
|
|
|
|
|
+} from './catalog.ts'
|
|
|
import { buildProvider, supportedProtocols } from './provider.ts'
|
|
import { buildProvider, supportedProtocols } from './provider.ts'
|
|
|
|
|
|
|
|
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
|
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
|
@@ -34,8 +40,21 @@ export const DEFAULT_CONTEXT_WINDOW = 262_144
|
|
|
/** Output capability assumed for a model neither configuration nor the catalog sizes. */
|
|
/** Output capability assumed for a model neither configuration nor the catalog sizes. */
|
|
|
export const DEFAULT_MAX_TOKENS = 32_768
|
|
export const DEFAULT_MAX_TOKENS = 32_768
|
|
|
|
|
|
|
|
|
|
+/**
|
|
|
|
|
+ * Modalities assumed for a model neither configuration nor the catalog
|
|
|
|
|
+ * declares. Text is the floor every supported protocol certainly carries, so
|
|
|
|
|
+ * this is the absence of a declaration rather than a guess at the endpoint:
|
|
|
|
|
+ * nothing can interrogate a gateway for its modalities, and the two wrong
|
|
|
|
|
+ * answers do not cost the same. Under-claiming refuses the image before it is
|
|
|
|
|
+ * attached, naming the model. Over-claiming admits one the provider then
|
|
|
|
|
+ * rejects mid-turn, after the message is durable, leaving the session
|
|
|
|
|
+ * repeating a request that cannot succeed.
|
|
|
|
|
+ */
|
|
|
|
|
+export const DEFAULT_INPUT: readonly PiAiModality[] = ['text']
|
|
|
|
|
+
|
|
|
export type {
|
|
export type {
|
|
|
PiAiCompatProfile,
|
|
PiAiCompatProfile,
|
|
|
|
|
+ PiAiModality,
|
|
|
PiAiModelOverride,
|
|
PiAiModelOverride,
|
|
|
PiAiModelProfile,
|
|
PiAiModelProfile,
|
|
|
PiAiReasoningEfforts,
|
|
PiAiReasoningEfforts,
|
|
@@ -90,6 +109,17 @@ export interface PiAiProviderProfile {
|
|
|
* never becomes a per-request cap on its own.
|
|
* never becomes a per-request cap on its own.
|
|
|
*/
|
|
*/
|
|
|
defaultMaxTokens?: number
|
|
defaultMaxTokens?: number
|
|
|
|
|
+ /**
|
|
|
|
|
+ * Request modalities for a model this route lists that neither its entry's
|
|
|
|
|
+ * {@link PiAiModelProfile.input} nor the installed catalog declares (default
|
|
|
|
|
+ * `[text]`). A fallback like the capacities above, not an override: a
|
|
|
|
|
+ * catalog model keeps the modalities the catalog records for it, and this
|
|
|
|
|
+ * value never narrows one. A gateway serving vision models the catalog does
|
|
|
|
|
+ * not describe declares `[text, image]` once here instead of on every entry.
|
|
|
|
|
+ * Unlike an entry's list, this one may not be empty — nothing sits below it
|
|
|
|
|
+ * to answer instead.
|
|
|
|
|
+ */
|
|
|
|
|
+ defaultInput?: PiAiModality[]
|
|
|
/** Provider request headers; Harness attribution wins reserved names. */
|
|
/** Provider request headers; Harness attribution wins reserved names. */
|
|
|
headers?: Record<string, string>
|
|
headers?: Record<string, string>
|
|
|
/** Provider-neutral pi-ai reasoning level. */
|
|
/** Provider-neutral pi-ai reasoning level. */
|
|
@@ -180,6 +210,10 @@ const modelFields = {
|
|
|
name: z.string(),
|
|
name: z.string(),
|
|
|
contextWindow: z.number().step(1).min(1),
|
|
contextWindow: z.number().step(1).min(1),
|
|
|
maxTokens: z.number().step(1).min(1),
|
|
maxTokens: z.number().step(1).min(1),
|
|
|
|
|
+ // No explicit default, unlike the route's `defaultInput`: schemastery
|
|
|
|
|
+ // materializes `[]` for an absent array, and resolution reads that as "no
|
|
|
|
|
+ // answer here" so the catalog entry below still applies.
|
|
|
|
|
+ input: z.array(z.union(MODALITIES)),
|
|
|
// The union, not a bare dict: schemastery materializes an absent dict as
|
|
// The union, not a bare dict: schemastery materializes an absent dict as
|
|
|
// `{}`, and absent must stay distinguishable — it means "inherit the
|
|
// `{}`, and absent must stay distinguishable — it means "inherit the
|
|
|
// installed catalog's capability", while `false` disables reasoning.
|
|
// installed catalog's capability", while `false` disables reasoning.
|
|
@@ -205,6 +239,7 @@ const profile = z.object({
|
|
|
compat: compatProfile,
|
|
compat: compatProfile,
|
|
|
defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
|
|
defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
|
|
|
defaultMaxTokens: z.number().step(1).min(1).default(DEFAULT_MAX_TOKENS),
|
|
defaultMaxTokens: z.number().step(1).min(1).default(DEFAULT_MAX_TOKENS),
|
|
|
|
|
+ defaultInput: z.array(z.union(MODALITIES)).default([...DEFAULT_INPUT]),
|
|
|
headers: z.dict(z.string()),
|
|
headers: z.dict(z.string()),
|
|
|
reasoning: z.union(THINKING_LEVELS),
|
|
reasoning: z.union(THINKING_LEVELS),
|
|
|
thinkingBudgets,
|
|
thinkingBudgets,
|
|
@@ -288,6 +323,15 @@ export function resolveProfiles(
|
|
|
`llm-pi-ai: provider "${provider}" streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
|
|
`llm-pi-ai: provider "${provider}" streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
|
|
|
)
|
|
)
|
|
|
}
|
|
}
|
|
|
|
|
+ // Detached from the configuration object because pi-ai types `Model.input`
|
|
|
|
|
+ // mutable. The schema's explicit default covers an absent key, so an empty
|
|
|
|
|
+ // list here is always one someone typed — and unlike an entry's, nothing
|
|
|
|
|
+ // below it can answer instead — so it is refused rather than read as "no
|
|
|
|
|
+ // answer".
|
|
|
|
|
+ const defaultInput = [...source.defaultInput ?? DEFAULT_INPUT]
|
|
|
|
|
+ if (defaultInput.length === 0) {
|
|
|
|
|
+ throw new Error(`llm-pi-ai: provider "${provider}" defaultInput must name at least one modality`)
|
|
|
|
|
+ }
|
|
|
// The route key, not the installed provider's own name: the directory has
|
|
// The route key, not the installed provider's own name: the directory has
|
|
|
// always shown route keys, and a catalog route must not silently rename
|
|
// always shown route keys, and a catalog route must not silently rename
|
|
|
// itself on every configuration surface just because it gained a profile.
|
|
// itself on every configuration surface just because it gained a profile.
|
|
@@ -299,6 +343,7 @@ export function resolveProfiles(
|
|
|
...source.models === undefined ? {} : { models: source.models },
|
|
...source.models === undefined ? {} : { models: source.models },
|
|
|
...source.modelOverrides === undefined ? {} : { modelOverrides: source.modelOverrides },
|
|
...source.modelOverrides === undefined ? {} : { modelOverrides: source.modelOverrides },
|
|
|
...source.compat === undefined ? {} : { compat: source.compat },
|
|
...source.compat === undefined ? {} : { compat: source.compat },
|
|
|
|
|
+ defaultInput,
|
|
|
defaultContextWindow: source.defaultContextWindow ?? DEFAULT_CONTEXT_WINDOW,
|
|
defaultContextWindow: source.defaultContextWindow ?? DEFAULT_CONTEXT_WINDOW,
|
|
|
defaultMaxTokens: source.defaultMaxTokens ?? DEFAULT_MAX_TOKENS,
|
|
defaultMaxTokens: source.defaultMaxTokens ?? DEFAULT_MAX_TOKENS,
|
|
|
})
|
|
})
|