Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions fork/manifest.json
Original file line number Diff line number Diff line change
Expand Up @@ -192,6 +192,7 @@
"packages/opencode/src/peer/claude/sidecar-manager.ts",
"packages/opencode/test/peer/claude/sidecar-e2e.test.ts",
"packages/opencode/test/peer/claude/sidecar-manager.test.ts",
"packages/opencode/test/peer/claude/lifecycle-inject.test.ts",
"packages/opencode/src/peer/claude/lifecycle.ts",
"packages/opencode/src/peer/delegate.ts",
"packages/opencode/test/peer/delegate.test.ts",
Expand Down
4 changes: 0 additions & 4 deletions packages/core/src/v1/config/config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -209,10 +209,6 @@ export const Info = Schema.Struct({
description:
"When the parent session runs on a local llama-skein provider, place subagents on an idle peer provider instead of queueing behind the parent (default: true)",
}),
local_subagent_placement_models: Schema.optional(Schema.mutable(Schema.Array(Schema.String))).annotate({
description:
"Model IDs trusted for local subagent placement (the parent's own model is always trusted). Unset: any tool-capable model on an idle provider qualifies",
}),
peer_delegation: Schema.optional(Schema.Boolean).annotate({
description:
"When no local host has a free slot for a subagent, hand the task to an idle peer agent (Claude Code or opencode) over A2A instead of failing (default: true)",
Expand Down
36 changes: 19 additions & 17 deletions packages/opencode/src/local/placement.ts
Original file line number Diff line number Diff line change
Expand Up @@ -128,14 +128,16 @@ export type Probe = {
providerID: ProviderV2.ID
hardware: ResourceSnapshot
fit: FitReport
/** The host's own configured default model (`GET /api/config/info`), if any. */
defaultModel?: string
}

async function probe(providerID: ProviderV2.ID, baseURL: string, signal: AbortSignal): Promise<Probe | null> {
const llama = new LlamaSkeinClient({
client: createClient(createConfig({ baseUrl: normalizeBaseURL(baseURL) })),
key: `placement:${providerID}`,
})
const [hardware, fit] = await Promise.all([
const [hardware, fit, configInfo] = await Promise.all([
llama
.getHardware({ signal })
.then((res) => res.data ?? null)
Expand All @@ -144,35 +146,34 @@ async function probe(providerID: ProviderV2.ID, baseURL: string, signal: AbortSi
.getFitReport({ signal })
.then((res) => res.data ?? null)
.catch(() => null),
// Best-effort: an older host without this endpoint simply has no default
// model preference, not a reason to skip placement on it.
llama
.getConfigInfo({ signal })
.then((res) => res.data?.default_model ?? undefined)
.catch(() => undefined),
])
// Both signals are required: hardware to judge busyness, fit to judge which
// model can actually serve a subagent. A plain llama-swap host without the
// llama-skein API yields nulls and is skipped — capacity we can't see is
// capacity we don't schedule on.
if (!hardware || !fit) return null
return { providerID, hardware, fit }
return { providerID, hardware, fit, defaultModel: configInfo }
}

export function bestModel(input: {
probe: Probe
info: Provider.Info
parentModelID: string
requiredCtx: number
allowedModels?: readonly string[]
}): { modelID: ModelV2.ID; score: number; maxSafeCtx: number } | null {
const loadedID = input.probe.hardware.loaded_model?.id
const defaultID = input.probe.defaultModel
let best: { modelID: ModelV2.ID; score: number; maxSafeCtx: number } | null = null
for (const fit of input.probe.fit.models) {
const model = input.info.models[fit.model]
if (!model) continue // not registered with opencode — can't be prompted
// Discovered local models default toolcall to true, so this filter alone
// can't be trusted — the allowlist below is the real vetting mechanism.
if (!model.capabilities.toolcall) continue
// Curated list of models proven to handle subagent tool calls. Anything
// else may be loaded and fast yet flub tool-call JSON, wasting the whole
// task. The parent's own model is always trusted — the user picked it.
if (input.allowedModels?.length && fit.model !== input.parentModelID && !input.allowedModels.includes(fit.model))
continue
// Hard context filter. A model whose usable context (max_safe_ctx) can't
// hold the subagent prompt will 413 or silently truncate it — never pick
// it, regardless of how well it fits VRAM or how fast it is. This is the
Expand All @@ -188,12 +189,15 @@ export function bestModel(input: {
// Swapping models on a host mid-session evicts what the user (or skein)
// deliberately keeps loaded and costs a multi-second reload both ways.
// An eligible resident model always beats anything that needs a load;
// eligibility (allowlist, ctx, fit) is still enforced by the filters
// above, so an unvetted or too-small resident model never wins by
// residency alone.
// eligibility (ctx, fit, toolcall) is still enforced by the filters
// above, so a too-small resident model never wins by residency alone.
(fit.model === loadedID ? 100_000 : 0) +
// Among models that would need a load, prefer the parent's own model:
// proven behavior beats an arbitrary pick.
// Among models that would need a load, prefer the host's own configured
// default (GET /api/config/info) — the operator chose it for this host
// specifically, which beats guessing from fit/speed alone.
(fit.model === defaultID ? 20_000 : 0) +
// Failing that, prefer the parent's own model: proven behavior beats an
// arbitrary pick.
(fit.model === input.parentModelID ? 5_000 : 0) +
Math.min(fit.est_tokens_per_sec ?? 0, 500) -
// Host-bandwidth-paced placements rank below every GPU-resident
Expand Down Expand Up @@ -415,7 +419,6 @@ export type PickOutcome =
export async function pick(input: {
parent: Placement
providers: Record<string, Provider.Info>
allowedModels?: readonly string[]
promptText?: string
requiredCtx?: number
timeoutMs?: number
Expand Down Expand Up @@ -496,7 +499,6 @@ export async function pick(input: {
info: candidates[i].info,
parentModelID: input.parent.modelID,
requiredCtx,
allowedModels: input.allowedModels,
})
if (!model) {
scored.push({ providerID, eligible: false })
Expand Down
1 change: 0 additions & 1 deletion packages/opencode/src/tool/task.ts
Original file line number Diff line number Diff line change
Expand Up @@ -283,7 +283,6 @@ export const TaskTool = Tool.define(
LocalPlacement.pick({
parent: inherited,
providers,
allowedModels: cfg.experimental?.local_subagent_placement_models,
promptText: params.prompt,
target: params.provider,
prefer: rolePlacement,
Expand Down
62 changes: 49 additions & 13 deletions packages/opencode/test/local/placement.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,11 +27,12 @@ function info(...models: string[]): Provider.Info {
} as unknown as Provider.Info
}

function probe(models: ModelFit[], loadedID?: string): Probe {
function probe(models: ModelFit[], loadedID?: string, defaultModel?: string): Probe {
return {
providerID: "host" as Probe["providerID"],
hardware: hw({ slots_total: 1, in_flight: 0, busy: false }, loadedID),
fit: { models } as Probe["fit"],
defaultModel,
}
}

Expand Down Expand Up @@ -110,7 +111,7 @@ describe("bestModel context-adequacy filter", () => {
})
})

describe("bestModel resident-model tier (skein rule: loaded > preferred > any)", () => {
describe("bestModel scoring tiers: loaded > host default > parent's model > any", () => {
const requiredCtx = 9_000

test("an eligible resident model beats swapping to the parent's own model", () => {
Expand All @@ -134,22 +135,57 @@ describe("bestModel resident-model tier (skein rule: loaded > preferred > any)",
expect(result?.modelID as string | undefined).toBe("parentm")
})

test("residency never overrides vetting — an unlisted resident model loses to an allowed cold one", () => {
test("a resident model beats a model that would need a fresh load, tool-capable or not", () => {
// No allowlist anymore — every registered, tool-capable model is eligible.
// Residency still wins over a cold load regardless of raw fit/speed.
const p = probe(
[
fitModel("resident", { fit_level: "perfect", max_safe_ctx: 32_768 }),
fitModel("vetted", { fit_level: "good", max_safe_ctx: 32_768 }),
fitModel("resident", { fit_level: "good", max_safe_ctx: 32_768, est_tokens_per_sec: 20 }),
fitModel("faster", { fit_level: "perfect", max_safe_ctx: 32_768, est_tokens_per_sec: 400 }),
],
"resident",
)
const result = bestModel({
probe: p,
info: info("resident", "vetted"),
parentModelID: "cloud",
requiredCtx,
allowedModels: ["vetted"],
})
expect(result?.modelID as string | undefined).toBe("vetted")
const result = bestModel({ probe: p, info: info("resident", "faster"), parentModelID: "cloud", requiredCtx })
expect(result?.modelID as string | undefined).toBe("resident")
})

test("with nothing resident, the host's own default model is preferred over a better fit", () => {
const p = probe(
[
fitModel("default", { fit_level: "good", max_safe_ctx: 32_768, est_tokens_per_sec: 20 }),
fitModel("other", { fit_level: "perfect", max_safe_ctx: 32_768, est_tokens_per_sec: 400 }),
],
undefined,
"default",
)
const result = bestModel({ probe: p, info: info("default", "other"), parentModelID: "cloud", requiredCtx })
expect(result?.modelID as string | undefined).toBe("default")
})

test("the host's default model outranks the parent's own model", () => {
const p = probe(
[
fitModel("default", { fit_level: "good", max_safe_ctx: 32_768 }),
fitModel("parentm", { fit_level: "good", max_safe_ctx: 32_768 }),
],
undefined,
"default",
)
const result = bestModel({ probe: p, info: info("default", "parentm"), parentModelID: "parentm", requiredCtx })
expect(result?.modelID as string | undefined).toBe("default")
})

test("residency still outranks the host's own default model", () => {
const p = probe(
[
fitModel("resident", { fit_level: "good", max_safe_ctx: 32_768 }),
fitModel("default", { fit_level: "good", max_safe_ctx: 32_768 }),
],
"resident",
"default",
)
const result = bestModel({ probe: p, info: info("resident", "default"), parentModelID: "cloud", requiredCtx })
expect(result?.modelID as string | undefined).toBe("resident")
})
})

Expand Down
Loading