From 112c74c898195e82d84d3f3a0b764a4f3b2d4648 Mon Sep 17 00:00:00 2001 From: santoshkumarradha Date: Sun, 27 Sep 2026 12:25:59 -0400 Subject: [PATCH 1/3] fix(chat): defer independent pool judges in one-model runs --- cmd/codeaf/chatv3.go | 28 ++++++++------ cmd/codeaf/pooljudge_close_test.go | 50 +++++++++++++++++++++++++ internal/manual/chat/models-and-cost.md | 4 ++ 3 files changed, 71 insertions(+), 11 deletions(-) diff --git a/cmd/codeaf/chatv3.go b/cmd/codeaf/chatv3.go index 03a8837920..9cbec89f49 100644 --- a/cmd/codeaf/chatv3.go +++ b/cmd/codeaf/chatv3.go @@ -940,17 +940,23 @@ func openV3Launch(proc *v3Process, opts v3Options) (*v3Launch, error) { // door: a mode that forbids reading builds no hook, and a nil hook is the // engine's own nothing. The ask is built once and a client is made from it // per call, each billed to the judge's own seat. - taskLanded := poolJudgeHook(settings, settings.ProfileDir, workspace, - config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now, "task") - // The runs a live process would have judged but a process death left unjudged, - // and the headless doors that never had this hook: at start, on a goroutine - // nobody waits on, judge the resumed session's own final-state nodes and the - // pending file's rows, each exactly once, bounded so it never holds the prompt. The - // process tracker cancels and joins it at close. - poolErrandGoCtx(settings.ProfileDir, "pool/judge-sweep", func(ctx context.Context) { - poolJudgeSweepRun(ctx, settings, settings.ProfileDir, found.Place.Tasks(), - config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now) - }) + var taskLanded func(session.TaskLanding) + // AN INDEPENDENT JUDGE CANNOT SHARE THE CREW MODEL. A one-model + // launch therefore leaves this optional scoring to an ordinary launch; + // it must neither judge new landings nor sweep earlier pending work. + if !opts.OneModel { + taskLanded = poolJudgeHook(settings, settings.ProfileDir, workspace, + config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now, "task") + // The runs a live process would have judged but a process death left unjudged, + // and the headless doors that never had this hook: at start, on a goroutine + // nobody waits on, judge the resumed session's own final-state nodes and the + // pending file's rows, each exactly once, bounded so it never holds the prompt. The + // process tracker cancels and joins it at close. + poolErrandGoCtx(settings.ProfileDir, "pool/judge-sweep", func(ctx context.Context) { + poolJudgeSweepRun(ctx, settings, settings.ProfileDir, found.Place.Tasks(), + config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now) + }) + } cfg := session.Config{ Workspace: workspace, diff --git a/cmd/codeaf/pooljudge_close_test.go b/cmd/codeaf/pooljudge_close_test.go index 7cf45ae9a7..38423c65a2 100644 --- a/cmd/codeaf/pooljudge_close_test.go +++ b/cmd/codeaf/pooljudge_close_test.go @@ -1,7 +1,9 @@ package main import ( + "bytes" "context" + "fmt" "os" "path/filepath" "sync" @@ -188,3 +190,51 @@ func TestJudgeLandingLeftUnjudgedWhenCancelledMidJudge(t *testing.T) { t.Fatal("a landing cancelled mid-judge was marked judged; it will never be scored or rejudged") } } + +// The launch policy covers both routes into optional independent judgments. +func TestOneModelLaunchWithholdsPoolHookAndSweep(t *testing.T) { + for _, one := range []bool{true, false} { + t.Run(fmt.Sprint(one), func(t *testing.T) { + proc := v3TestProcess(t) + t.Setenv("CODEAF_MODEL_POOL", "read") + if err := writePendingLanding(proc.ProfileDir, "do", poolTestLanding()); err != nil { + t.Fatal(err) + } + pending := pendingPath(config.ProfilePath(proc.ProfileDir, "pool")) + before, err := os.ReadFile(pending) + if err != nil { + t.Fatal(err) + } + started := make(chan struct{}, 1) + old := poolJudgeSweepRun + poolJudgeSweepRun = func(context.Context, config.Config, string, string, func() []catalog.Model, func(string) judge.Ask, func() time.Time) { + started <- struct{}{} + } + t.Cleanup(func() { poolJudgeSweepRun = old }) + launch, err := openV3Launch(proc, v3Options{Model: "test/model", Workspace: t.TempDir(), OneModel: one}) + if err != nil { + t.Fatal(err) + } + if (launch.Config.TaskLanded == nil) != one { + t.Fatalf("one-model=%v: hook absent=%v", one, launch.Config.TaskLanded == nil) + } + proc.closeAll() + if one { + after, err := os.ReadFile(pending) + if err != nil || !bytes.Equal(before, after) { + t.Fatalf("single-model launch consumed or changed pending judgments: %v", err) + } + } + select { + case <-started: + if one { + t.Fatal("one-model launch started an independent judge sweep") + } + default: + if !one { + t.Fatal("ordinary launch lost its judge sweep") + } + } + }) + } +} diff --git a/internal/manual/chat/models-and-cost.md b/internal/manual/chat/models-and-cost.md index a75f0bfff7..82d7e7f4ff 100644 --- a/internal/manual/chat/models-and-cost.md +++ b/internal/manual/chat/models-and-cost.md @@ -1239,6 +1239,10 @@ rather than handed to the model that just wrote the answer. Under this flag they model like everything else, because you have said your model is the crew. Without the flag and without a planner to seat, a move that needs them says `no second model is set`. +**Model Pool judging waits for an ordinary launch.** Its judge must be independent +of the crew, so `--one-model` runs neither the task-landing judge nor the startup sweep +of pending judgments. It does not substitute your worker as its own independent judge. + **It changes no setting and writes nothing.** Your pins and rows are untouched, `/crew` still says what it said, and the next session without the flag reads them exactly as before. It is a posture for one run, not an edit. From ffe8a20646a551afeed39c915951e8f70ed670e6 Mon Sep 17 00:00:00 2001 From: santoshkumarradha Date: Sun, 27 Sep 2026 12:27:37 -0400 Subject: [PATCH 2/3] docs: record single-model pool judge policy for PR 1610 --- docs/changes/unreleased/1610-one-model-pool.md | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 docs/changes/unreleased/1610-one-model-pool.md diff --git a/docs/changes/unreleased/1610-one-model-pool.md b/docs/changes/unreleased/1610-one-model-pool.md new file mode 100644 index 0000000000..13e2c3486b --- /dev/null +++ b/docs/changes/unreleased/1610-one-model-pool.md @@ -0,0 +1,11 @@ +--- +kind: fixed +title: Single-model chat defers independent Model Pool judging +pr: 1610 +surface: [chat] +invalidates: + - "The Model Pool could call another model after a task landed under --one-model. Single-model launches now defer independent judging and leave pending judgments for an ordinary launch." +--- + +The task-landing hook and startup sweep stay absent during a single-model launch. +The chosen worker is never substituted as its own independent pool judge. From f1a6773837faa0b5284a446f3b0a8b2901b2f9ac Mon Sep 17 00:00:00 2001 From: santoshkumarradha Date: Sun, 27 Sep 2026 12:46:51 -0400 Subject: [PATCH 3/3] test: preserve machine gate settings when pinning run models --- cmd/codeaf/do_engine_test.go | 47 ++++++++++++++++++------------------ 1 file changed, 23 insertions(+), 24 deletions(-) diff --git a/cmd/codeaf/do_engine_test.go b/cmd/codeaf/do_engine_test.go index 90df20d0dc..f3384da34d 100644 --- a/cmd/codeaf/do_engine_test.go +++ b/cmd/codeaf/do_engine_test.go @@ -149,6 +149,23 @@ func beltRunEnv(t *testing.T) string { return home } +// beltModelPins preserves the fixture's resource settings while changing the +// seats a test intends to exercise. Replacing the profile wholesale would +// silently re-enable the real machine gate on a busy test host (#1525). +func beltModelPins(t *testing.T, profileDir string, pins map[string]string) { + t.Helper() + settings := config.NewSettings(config.SettingsOptions{ProfileDir: profileDir}) + for key, value := range pins { + row, ok := settings.Row(key) + if !ok { + t.Fatalf("missing setting %s", key) + } + if err := row.Apply(value); err != nil { + t.Fatal(err) + } + } +} + // beltPlandbDoor is the real plandb CLI behind the resolver's override, built // once for the package. THE LOOP ENDS IN THE STORE: a worker's task is done // when `plandb done` marks it so and no other way, so a scripted worker that @@ -376,16 +393,10 @@ func TestDoOnTheRunEngineSeatsEveryLaunchOnTheDoorsModels(t *testing.T) { if err := os.MkdirAll(profileDir, 0o700); err != nil { t.Fatal(err) } - rows, err := json.Marshal(map[string]string{ + beltModelPins(t, profileDir, map[string]string{ config.KeyTierWorkerModel: "vendor/profile-worker", config.KeyTierMastermindModel: "vendor/profile-thinking", }) - if err != nil { - t.Fatal(err) - } - if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil { - t.Fatal(err) - } workspace := beltRepoWorkspace(t) const ( @@ -433,7 +444,7 @@ func TestDoOnTheRunEngineSeatsEveryLaunchOnTheDoorsModels(t *testing.T) { } var stdout, stderr strings.Builder - err = doErrand(doRequest{ + err := doErrand(doRequest{ task: "write out.txt and say what you did", workspace: workspace, asJSON: true, timeout: 60 * time.Second, slots: bound(1), model: workModel, planModel: planModel, checkModel: planModel, stdout: &stdout, stderr: &stderr, newBeltCompleter: newBelt, @@ -668,18 +679,12 @@ func TestDoOnTheRunEngineSeatsACheckOnTheCheckModel(t *testing.T) { if err := os.MkdirAll(profileDir, 0o700); err != nil { t.Fatal(err) } - rows, err := json.Marshal(map[string]string{ + beltModelPins(t, profileDir, map[string]string{ config.KeyTierLowModel: "vendor/profile-small", config.KeyTierWorkerModel: "vendor/profile-worker", config.KeyTierHighModel: "vendor/profile-careful", config.KeyTierMastermindModel: "vendor/profile-thinking", }) - if err != nil { - t.Fatal(err) - } - if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil { - t.Fatal(err) - } workspace := beltRepoWorkspace(t) const ( @@ -722,7 +727,7 @@ func TestDoOnTheRunEngineSeatsACheckOnTheCheckModel(t *testing.T) { } var stdout, stderr strings.Builder - err = doErrand(doRequest{ + err := doErrand(doRequest{ task: "write out.txt and say what you did", workspace: workspace, asJSON: true, timeout: 60 * time.Second, slots: bound(1), model: workModel, planModel: planModel, checkModel: checkModel, @@ -767,17 +772,11 @@ func TestDoOnTheRunEngineSeatsAnUnpinnedCheckOnTheCrewsChecker(t *testing.T) { if err := os.MkdirAll(profileDir, 0o700); err != nil { t.Fatal(err) } - rows, err := json.Marshal(map[string]string{ + beltModelPins(t, profileDir, map[string]string{ config.KeyTierWorkerModel: "vendor/profile-worker", config.KeyTierHighModel: "vendor/profile-careful", config.KeyTierMastermindModel: "vendor/profile-thinking", }) - if err != nil { - t.Fatal(err) - } - if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil { - t.Fatal(err) - } workspace := beltRepoWorkspace(t) // The same one-seat shape as the two tests above. @@ -813,7 +812,7 @@ func TestDoOnTheRunEngineSeatsAnUnpinnedCheckOnTheCrewsChecker(t *testing.T) { } var stdout, stderr strings.Builder - err = doErrand(doRequest{ + err := doErrand(doRequest{ task: "write out.txt and say what you did", workspace: workspace, asJSON: true, timeout: 60 * time.Second, slots: bound(1), stdout: &stdout, stderr: &stderr, newBeltCompleter: newBelt,