Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 17 additions & 11 deletions cmd/codeaf/chatv3.go
Original file line number Diff line number Diff line change
Expand Up @@ -940,17 +940,23 @@ func openV3Launch(proc *v3Process, opts v3Options) (*v3Launch, error) {
// door: a mode that forbids reading builds no hook, and a nil hook is the
// engine's own nothing. The ask is built once and a client is made from it
// per call, each billed to the judge's own seat.
taskLanded := poolJudgeHook(settings, settings.ProfileDir, workspace,
config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now, "task")
// The runs a live process would have judged but a process death left unjudged,
// and the headless doors that never had this hook: at start, on a goroutine
// nobody waits on, judge the resumed session's own final-state nodes and the
// pending file's rows, each exactly once, bounded so it never holds the prompt. The
// process tracker cancels and joins it at close.
poolErrandGoCtx(settings.ProfileDir, "pool/judge-sweep", func(ctx context.Context) {
poolJudgeSweepRun(ctx, settings, settings.ProfileDir, found.Place.Tasks(),
config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now)
})
var taskLanded func(session.TaskLanding)
// AN INDEPENDENT JUDGE CANNOT SHARE THE CREW MODEL. A one-model
// launch therefore leaves this optional scoring to an ordinary launch;
// it must neither judge new landings nor sweep earlier pending work.
if !opts.OneModel {
taskLanded = poolJudgeHook(settings, settings.ProfileDir, workspace,
config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now, "task")
// The runs a live process would have judged but a process death left unjudged,
// and the headless doors that never had this hook: at start, on a goroutine
// nobody waits on, judge the resumed session's own final-state nodes and the
// pending file's rows, each exactly once, bounded so it never holds the prompt. The
// process tracker cancels and joins it at close.
poolErrandGoCtx(settings.ProfileDir, "pool/judge-sweep", func(ctx context.Context) {
poolJudgeSweepRun(ctx, settings, settings.ProfileDir, found.Place.Tasks(),
config.CrewCatalog, poolJudgeAsk(proc.liveSettings(settings), settings.ProfileDir), time.Now)
})
}

cfg := session.Config{
Workspace: workspace,
Expand Down
47 changes: 23 additions & 24 deletions cmd/codeaf/do_engine_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -149,6 +149,23 @@ func beltRunEnv(t *testing.T) string {
return home
}

// beltModelPins preserves the fixture's resource settings while changing the
// seats a test intends to exercise. Replacing the profile wholesale would
// silently re-enable the real machine gate on a busy test host (#1525).
func beltModelPins(t *testing.T, profileDir string, pins map[string]string) {
t.Helper()
settings := config.NewSettings(config.SettingsOptions{ProfileDir: profileDir})
for key, value := range pins {
row, ok := settings.Row(key)
if !ok {
t.Fatalf("missing setting %s", key)
}
if err := row.Apply(value); err != nil {
t.Fatal(err)
}
}
}

// beltPlandbDoor is the real plandb CLI behind the resolver's override, built
// once for the package. THE LOOP ENDS IN THE STORE: a worker's task is done
// when `plandb done` marks it so and no other way, so a scripted worker that
Expand Down Expand Up @@ -376,16 +393,10 @@ func TestDoOnTheRunEngineSeatsEveryLaunchOnTheDoorsModels(t *testing.T) {
if err := os.MkdirAll(profileDir, 0o700); err != nil {
t.Fatal(err)
}
rows, err := json.Marshal(map[string]string{
beltModelPins(t, profileDir, map[string]string{
config.KeyTierWorkerModel: "vendor/profile-worker",
config.KeyTierMastermindModel: "vendor/profile-thinking",
})
if err != nil {
t.Fatal(err)
}
if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil {
t.Fatal(err)
}

workspace := beltRepoWorkspace(t)
const (
Expand Down Expand Up @@ -433,7 +444,7 @@ func TestDoOnTheRunEngineSeatsEveryLaunchOnTheDoorsModels(t *testing.T) {
}

var stdout, stderr strings.Builder
err = doErrand(doRequest{
err := doErrand(doRequest{
task: "write out.txt and say what you did", workspace: workspace, asJSON: true,
timeout: 60 * time.Second, slots: bound(1), model: workModel, planModel: planModel, checkModel: planModel,
stdout: &stdout, stderr: &stderr, newBeltCompleter: newBelt,
Expand Down Expand Up @@ -668,18 +679,12 @@ func TestDoOnTheRunEngineSeatsACheckOnTheCheckModel(t *testing.T) {
if err := os.MkdirAll(profileDir, 0o700); err != nil {
t.Fatal(err)
}
rows, err := json.Marshal(map[string]string{
beltModelPins(t, profileDir, map[string]string{
config.KeyTierLowModel: "vendor/profile-small",
config.KeyTierWorkerModel: "vendor/profile-worker",
config.KeyTierHighModel: "vendor/profile-careful",
config.KeyTierMastermindModel: "vendor/profile-thinking",
})
if err != nil {
t.Fatal(err)
}
if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil {
t.Fatal(err)
}

workspace := beltRepoWorkspace(t)
const (
Expand Down Expand Up @@ -722,7 +727,7 @@ func TestDoOnTheRunEngineSeatsACheckOnTheCheckModel(t *testing.T) {
}

var stdout, stderr strings.Builder
err = doErrand(doRequest{
err := doErrand(doRequest{
task: "write out.txt and say what you did", workspace: workspace, asJSON: true,
timeout: 60 * time.Second, slots: bound(1),
model: workModel, planModel: planModel, checkModel: checkModel,
Expand Down Expand Up @@ -767,17 +772,11 @@ func TestDoOnTheRunEngineSeatsAnUnpinnedCheckOnTheCrewsChecker(t *testing.T) {
if err := os.MkdirAll(profileDir, 0o700); err != nil {
t.Fatal(err)
}
rows, err := json.Marshal(map[string]string{
beltModelPins(t, profileDir, map[string]string{
config.KeyTierWorkerModel: "vendor/profile-worker",
config.KeyTierHighModel: "vendor/profile-careful",
config.KeyTierMastermindModel: "vendor/profile-thinking",
})
if err != nil {
t.Fatal(err)
}
if err := os.WriteFile(config.BudgetConfigPath(profileDir), rows, 0o600); err != nil {
t.Fatal(err)
}

workspace := beltRepoWorkspace(t)
// The same one-seat shape as the two tests above.
Expand Down Expand Up @@ -813,7 +812,7 @@ func TestDoOnTheRunEngineSeatsAnUnpinnedCheckOnTheCrewsChecker(t *testing.T) {
}

var stdout, stderr strings.Builder
err = doErrand(doRequest{
err := doErrand(doRequest{
task: "write out.txt and say what you did", workspace: workspace, asJSON: true,
timeout: 60 * time.Second, slots: bound(1),
stdout: &stdout, stderr: &stderr, newBeltCompleter: newBelt,
Expand Down
50 changes: 50 additions & 0 deletions cmd/codeaf/pooljudge_close_test.go
Original file line number Diff line number Diff line change
@@ -1,7 +1,9 @@
package main

import (
"bytes"
"context"
"fmt"
"os"
"path/filepath"
"sync"
Expand Down Expand Up @@ -188,3 +190,51 @@ func TestJudgeLandingLeftUnjudgedWhenCancelledMidJudge(t *testing.T) {
t.Fatal("a landing cancelled mid-judge was marked judged; it will never be scored or rejudged")
}
}

// The launch policy covers both routes into optional independent judgments.
func TestOneModelLaunchWithholdsPoolHookAndSweep(t *testing.T) {
for _, one := range []bool{true, false} {
t.Run(fmt.Sprint(one), func(t *testing.T) {
proc := v3TestProcess(t)
t.Setenv("CODEAF_MODEL_POOL", "read")
if err := writePendingLanding(proc.ProfileDir, "do", poolTestLanding()); err != nil {
t.Fatal(err)
}
pending := pendingPath(config.ProfilePath(proc.ProfileDir, "pool"))
before, err := os.ReadFile(pending)
if err != nil {
t.Fatal(err)
}
started := make(chan struct{}, 1)
old := poolJudgeSweepRun
poolJudgeSweepRun = func(context.Context, config.Config, string, string, func() []catalog.Model, func(string) judge.Ask, func() time.Time) {
started <- struct{}{}
}
t.Cleanup(func() { poolJudgeSweepRun = old })
launch, err := openV3Launch(proc, v3Options{Model: "test/model", Workspace: t.TempDir(), OneModel: one})
if err != nil {
t.Fatal(err)
}
if (launch.Config.TaskLanded == nil) != one {
t.Fatalf("one-model=%v: hook absent=%v", one, launch.Config.TaskLanded == nil)
}
proc.closeAll()
if one {
after, err := os.ReadFile(pending)
if err != nil || !bytes.Equal(before, after) {
t.Fatalf("single-model launch consumed or changed pending judgments: %v", err)
}
}
select {
case <-started:
if one {
t.Fatal("one-model launch started an independent judge sweep")
}
default:
if !one {
t.Fatal("ordinary launch lost its judge sweep")
}
}
})
}
}
11 changes: 11 additions & 0 deletions docs/changes/unreleased/1610-one-model-pool.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
---
kind: fixed
title: Single-model chat defers independent Model Pool judging
pr: 1610
surface: [chat]
invalidates:
- "The Model Pool could call another model after a task landed under --one-model. Single-model launches now defer independent judging and leave pending judgments for an ordinary launch."
---

The task-landing hook and startup sweep stay absent during a single-model launch.
The chosen worker is never substituted as its own independent pool judge.
4 changes: 4 additions & 0 deletions internal/manual/chat/models-and-cost.md
Original file line number Diff line number Diff line change
Expand Up @@ -1239,6 +1239,10 @@ rather than handed to the model that just wrote the answer. Under this flag they
model like everything else, because you have said your model is the crew. Without the flag and
without a planner to seat, a move that needs them says `no second model is set`.

**Model Pool judging waits for an ordinary launch.** Its judge must be independent
of the crew, so `--one-model` runs neither the task-landing judge nor the startup sweep
of pending judgments. It does not substitute your worker as its own independent judge.

**It changes no setting and writes nothing.** Your pins and rows are untouched, `/crew`
still says what it said, and the next session without the flag reads them exactly as before.
It is a posture for one run, not an edit.
Expand Down
Loading