From 8479a9bed007ca342c22858de1731c607d3826c2 Mon Sep 17 00:00:00 2001 From: Sam Bretz Date: Tue, 15 Sep 2026 21:03:26 -0700 Subject: [PATCH] feat: report the model each role actually ran, not the configured one Every surface named a stage's models from configuration. A role with no configured model is invoked without --model and the harness chooses for itself, so "harness default" was as much as any of them could say, and an alias like sonnet never showed the version behind it. After a run finished there was no way to answer which model it had used. Each attempt now records what its harness reports running. Claude names its model on the init event of its stream, the same event the session identity already comes from, so the adapter reads both in one pass. The backend records it per role beside the session, the observation carries it, and the engine writes it onto the attempt, preferring a reported model over a configured one because it is the more precise of the two: it names what an unset role resolved to and the version behind an alias. The TUI stage view and the dashboard stage panel show what ran, falling back to the configured value when a harness reports nothing rather than guessing. Codex documents no model on its events, so its adapter reads one only if a stream provides it under the usual key and otherwise reports nothing, which leaves those surfaces exactly as they were. Also fixes the dashboard timeline, which tested whether either role had a model and then always printed the worker's: a stage with a supervisor model and no worker model announced "with harness default", naming the wrong role. It now names each role that reported one, and the supervisor appears there at all for the first time. Refs #38 Co-Authored-By: Claude Opus 5 (1M context) --- docs/src/content/docs/config-reference.mdx | 18 ++++++++++++ internal/agent/claude.go | 28 ++++++++++++++++-- internal/agent/claude_test.go | 18 ++++++++++++ internal/agent/codex.go | 16 +++++++++++ internal/agent/harness.go | 6 ++++ internal/engine/attempt.go | 33 ++++++++++++++++++++++ internal/engine/attempt_test.go | 24 ++++++++++++++++ internal/engine/engine.go | 3 ++ internal/localexec/attempt.go | 20 +++++++++---- internal/tui/agents_test.go | 22 +++++++++++++++ internal/tui/model.go | 30 +++++++++++++++++++- internal/web/assets/app.js | 15 ++++++++-- internal/web/server_test.go | 32 +++++++++++++++++++++ internal/web/view.go | 14 +++++++-- 14 files changed, 264 insertions(+), 15 deletions(-) create mode 100644 internal/engine/attempt_test.go diff --git a/docs/src/content/docs/config-reference.mdx b/docs/src/content/docs/config-reference.mdx index 0a3edca..9e82fbe 100644 --- a/docs/src/content/docs/config-reference.mdx +++ b/docs/src/content/docs/config-reference.mdx @@ -162,6 +162,24 @@ harness's own default (no `--model` flag). Readiness probes for `harness.worker`/`harness.supervisor` always use the run-wide model, since a connection probe has no stage in scope. +### Which model actually ran + +A role with no configured model is invoked without `--model`, and the harness +chooses for itself, so the configuration cannot say what it used. Each attempt +therefore records what its harness reports running, and the TUI stage view, +the dashboard stage panel and the run timeline show that in preference to the +configured value: + +| Configured | Reported | Shown | +| --- | --- | --- | +| unset | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (harness default)` | +| `sonnet` | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (configured sonnet)` | +| `opus` | `opus` | `opus` | +| `opus` | nothing reported | `opus` | + +Claude reports its model on the `init` event of its stream. A harness that +reports nothing leaves the configured value showing, rather than a guess. + `envctl run show` and `envctl run readiness` print each node's effective worker and supervisor model next to its attempt budget. diff --git a/internal/agent/claude.go b/internal/agent/claude.go index a766dac..a5c3a82 100644 --- a/internal/agent/claude.go +++ b/internal/agent/claude.go @@ -157,17 +157,39 @@ func (c Claude) Result(ctx context.Context, id string) (json.RawMessage, error) return readResult(ctx, c.Guest, id) } func (c Claude) Session(stream string) string { + session, _ := claudeInit(stream) + return session +} + +// Model reports the model Claude reports on the init event, which is the one +// it resolved for itself when the invocation named none. +func (c Claude) Model(stream string) string { + _, model := claudeInit(stream) + return model +} + +// claudeInit reads the session and model out of the stream's init event, the +// first event Claude emits. +func claudeInit(stream string) (session, model string) { for _, line := range strings.Split(stream, "\n") { var event struct { Type string `json:"type"` Subtype string `json:"subtype"` Session string `json:"session_id"` + Model string `json:"model"` + } + if json.Unmarshal([]byte(line), &event) != nil || event.Type != "system" || event.Subtype != "init" { + continue + } + if sessionID.MatchString(event.Session) { + session = event.Session } - if json.Unmarshal([]byte(line), &event) == nil && event.Type == "system" && event.Subtype == "init" && sessionID.MatchString(event.Session) { - return event.Session + model = event.Model + if session != "" { + return session, model } } - return "" + return session, model } // Preserve the native event stream for progress/reconnect. Only a successful diff --git a/internal/agent/claude_test.go b/internal/agent/claude_test.go index 45397ea..3409ee9 100644 --- a/internal/agent/claude_test.go +++ b/internal/agent/claude_test.go @@ -85,3 +85,21 @@ func TestInvocationEnvironmentIsScopedToEnvctlValues(t *testing.T) { } } } + +func TestClaudeReportsTheModelItResolvedForItself(t *testing.T) { + // An invocation with no --model still says which model it ran, on the + // same init event the session comes from. + stream := `{"type":"system","subtype":"init","session_id":"01a08f59-7073-7a83-b959-867ca896ce48","model":"claude-sonnet-5-20260115"} +{"type":"assistant"}` + c := Claude{} + if got := c.Model(stream); got != "claude-sonnet-5-20260115" { + t.Fatalf("model not read from the init event: %q", got) + } + if got := c.Session(stream); got != "01a08f59-7073-7a83-b959-867ca896ce48" { + t.Fatalf("reading the model broke the session: %q", got) + } + // A stream that never says leaves the surfaces on the configured value. + if got := c.Model(`{"type":"assistant"}`); got != "" { + t.Fatalf("invented a model: %q", got) + } +} diff --git a/internal/agent/codex.go b/internal/agent/codex.go index 98211fd..eb85027 100644 --- a/internal/agent/codex.go +++ b/internal/agent/codex.go @@ -341,6 +341,22 @@ func (b *limitBuffer) Write(p []byte) (int, error) { return b.data.Write(p) } +// Model reports the model an event stream says it ran. Codex's documented +// events carry no model, so this finds one only if a stream reports it under +// the usual key; an empty result leaves the surfaces showing the configured +// model rather than a guess. +func Model(stream string) string { + for _, line := range strings.Split(stream, "\n") { + var event struct { + Model string `json:"model"` + } + if json.Unmarshal([]byte(line), &event) == nil && event.Model != "" { + return event.Model + } + } + return "" +} + // Session extracts the explicit session identity from Codex's structured stream. // It never selects the globally most recent conversation when resuming. func Session(stream string) string { diff --git a/internal/agent/harness.go b/internal/agent/harness.go index 6ff5d9f..cffc32d 100644 --- a/internal/agent/harness.go +++ b/internal/agent/harness.go @@ -16,6 +16,11 @@ type Harness interface { Start(context.Context, Invocation, Credential) (guestjob.Status, error) Result(context.Context, string) (json.RawMessage, error) Session(string) string + // Model reports the model the harness actually ran, read from its own + // event stream. An invocation with no model names one the harness chose, + // which the configuration never records; empty means the stream did not + // say. + Model(string) string // Activity summarizes the native event stream for live progress display. Activity(string) []string // Usage sums the model usage the native event stream reports. @@ -52,3 +57,4 @@ func ResolveCredential(root, kind, reference string) (Credential, error) { } func (c Codex) Session(stream string) string { return Session(stream) } +func (c Codex) Model(stream string) string { return Model(stream) } diff --git a/internal/engine/attempt.go b/internal/engine/attempt.go index ff87f16..f2a7ccd 100644 --- a/internal/engine/attempt.go +++ b/internal/engine/attempt.go @@ -131,6 +131,17 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w }) return true, err } + if models := newModels(a.Models, observation.Models); len(models) > 0 { + _, err = e.update(ctx, id, revision, "attempt.models", func(_ *workflow.Run, v *workflow.Revision) error { + current := v.Attempt(a.ID) + if current == nil { + return workflow.ErrConflict + } + current.Models = models + return nil + }) + return true, err + } if observation.Session != "" && a.Session != observation.Session { _, err = e.update(ctx, id, revision, "attempt.session", func(_ *workflow.Run, v *workflow.Revision) error { current := v.Attempt(a.ID) @@ -226,3 +237,25 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w } return false, nil } + +// newModels merges what a harness reported running over what an attempt +// recorded from configuration, returning nil when nothing changed. A reported +// model is the more precise of the two: it names what an unset role resolved +// to, and the exact version behind an alias. +func newModels(recorded, reported map[string]string) map[string]string { + changed := false + merged := map[string]string{} + for role, model := range recorded { + merged[role] = model + } + for role, model := range reported { + if model != "" && merged[role] != model { + merged[role] = model + changed = true + } + } + if !changed { + return nil + } + return merged +} diff --git a/internal/engine/attempt_test.go b/internal/engine/attempt_test.go new file mode 100644 index 0000000..2b93328 --- /dev/null +++ b/internal/engine/attempt_test.go @@ -0,0 +1,24 @@ +package engine + +import "testing" + +func TestAReportedModelReplacesTheConfiguredOneOnTheAttempt(t *testing.T) { + // A role with nothing configured is exactly the case the configuration + // cannot answer, so what the harness reports has to win. + got := newModels(map[string]string{"supervisor": "opus"}, map[string]string{"worker": "claude-sonnet-5-20260115"}) + if got["worker"] != "claude-sonnet-5-20260115" || got["supervisor"] != "opus" { + t.Fatalf("merge lost a role: %v", got) + } + // An alias resolves to its version. + got = newModels(map[string]string{"worker": "sonnet"}, map[string]string{"worker": "claude-sonnet-5-20260115"}) + if got["worker"] != "claude-sonnet-5-20260115" { + t.Fatalf("alias not replaced by the version that ran: %v", got) + } + // Nothing new means no write, so an unchanged attempt is not rewritten. + if got = newModels(map[string]string{"worker": "opus"}, map[string]string{"worker": "opus"}); got != nil { + t.Fatalf("rewrote an unchanged attempt: %v", got) + } + if got = newModels(map[string]string{"worker": "opus"}, nil); got != nil { + t.Fatalf("rewrote when the harness said nothing: %v", got) + } +} diff --git a/internal/engine/engine.go b/internal/engine/engine.go index 1e82912..6d7c57c 100644 --- a/internal/engine/engine.go +++ b/internal/engine/engine.go @@ -32,6 +32,9 @@ type Observation struct { Result *workflow.Result Detail string // already redacted by the backend Session string // explicit harness session identity for durable continuation + // Models is the model each role's harness reported running, by role. It + // is what actually ran, as opposed to what the configuration asked for. + Models map[string]string // Progress and Delivered are display state for running attempts. Delivered // lists user messages already submitted to an agent invocation. Progress *workflow.Progress diff --git a/internal/localexec/attempt.go b/internal/localexec/attempt.go index f285ee1..a686fda 100644 --- a/internal/localexec/attempt.go +++ b/internal/localexec/attempt.go @@ -44,6 +44,10 @@ type attemptRecord struct { Steering []workflow.Delivery `json:"steering,omitempty"` Live map[string]*liveRole `json:"live,omitempty"` SupervisorSession string `json:"supervisor_session,omitempty"` + // Models is the model each role's harness reported running, by role. It + // records what an invocation with no configured model resolved to, which + // the configuration cannot say. + Models map[string]string `json:"models,omitempty"` // Stall is the durable no-output watch per agent role. Stall map[string]*stallState `json:"stall,omitempty"` } @@ -426,6 +430,12 @@ func (b *Backend) pollJob(ctx context.Context, a engine.Assignment, r *attemptRe } else { r.Session = harness.Session(r.Logs[id]) } + if model := harness.Model(r.Logs[id]); model != "" { + if r.Models == nil { + r.Models = map[string]string{} + } + r.Models[role] = model + } } if err = b.save(a, r); err != nil { return status, err @@ -446,7 +456,7 @@ func (b *Backend) failed(a engine.Assignment, r *attemptRecord, detail string) ( if err := b.save(a, r); err != nil { return engine.Observation{}, err } - return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil } func (b *Backend) workerFailed(ctx context.Context, a engine.Assignment, r *attemptRecord, detail string) (engine.Observation, error) { @@ -524,7 +534,7 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa return engine.Observation{}, err } running := func() (engine.Observation, error) { - return engine.Observation{State: "running", Session: r.Session, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "running", Session: r.Session, Models: r.Models, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil } // agentJob reconciles one role's active generation: it finishes stopping a // superseded generation, starts a missing job, records delivery once the @@ -822,11 +832,11 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa if err = b.save(a, r); err != nil { return engine.Observation{}, err } - return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil case "completed": - return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil case "failed": - return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil default: return engine.Observation{}, errors.New("unknown durable attempt phase") } diff --git a/internal/tui/agents_test.go b/internal/tui/agents_test.go index eb19168..9cb9588 100644 --- a/internal/tui/agents_test.go +++ b/internal/tui/agents_test.go @@ -48,3 +48,25 @@ func TestConversationModelLineCoexistsWithAttemptDetail(t *testing.T) { t.Fatalf("conversation lost either the model line or the attempt detail:\n%s", view) } } + +func TestTheStageViewNamesTheModelThatActuallyRan(t *testing.T) { + // The whole point: "harness default" never said which model that was. + if got := stageModelTUI("", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (harness default)" { + t.Fatalf("unconfigured role: %q", got) + } + // An alias shows the version behind it. + if got := stageModelTUI("sonnet", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (configured sonnet)" { + t.Fatalf("alias: %q", got) + } + // Agreement is not worth two names. + if got := stageModelTUI("opus", "opus"); got != "opus" { + t.Fatalf("agreement: %q", got) + } + // Nothing reported falls back rather than inventing. + if got := stageModelTUI("", ""); got != "harness default" { + t.Fatalf("nothing known: %q", got) + } + if got := stageModelTUI("opus", ""); got != "opus" { + t.Fatalf("configured only: %q", got) + } +} diff --git a/internal/tui/model.go b/internal/tui/model.go index ddeb98a..3630d93 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -901,7 +901,8 @@ func (m Model) details() string { switch panels[m.Panel] { case "Conversation": agents := rev.Config.NodeAgents(node) - lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", formatModelTUI(agents.Worker.Model), formatModelTUI(agents.Supervisor.Model))} + ran := ranModels(rev, node) + lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", stageModelTUI(agents.Worker.Model, ran["worker"]), stageModelTUI(agents.Supervisor.Model, ran["supervisor"]))} for _, a := range rev.Attempts { if a.Node == node { lines = append(lines, fmt.Sprintf("Worker attempt %d: %s", a.Number, a.State)) @@ -1014,6 +1015,33 @@ func formatModelTUI(model string) string { } return model } + +// ranModels is what a node's latest attempt reported actually running, which +// names the model an unconfigured role resolved to. +func ranModels(rev *workflow.Revision, node string) map[string]string { + out := map[string]string{} + for _, a := range rev.Attempts { + if a.Node == node && len(a.Models) > 0 { + out = a.Models + } + } + return out +} + +// stageModelTUI prefers what a role actually ran over what was configured for +// it, and says so when the two differ, because "harness default" on its own +// never answered which model that was. +func stageModelTUI(configured, ran string) string { + switch { + case ran == "": + return formatModelTUI(configured) + case configured == "": + return ran + " (harness default)" + case ran != configured: + return ran + " (configured " + configured + ")" + } + return ran +} func pretty(v any) string { b, err := json.MarshalIndent(v, "", " ") if err != nil { diff --git a/internal/web/assets/app.js b/internal/web/assets/app.js index de71674..ca3d0f0 100644 --- a/internal/web/assets/app.js +++ b/internal/web/assets/app.js @@ -56,6 +56,14 @@ const bytes = (n) => (n >= 1 << 20 ? `${(n / (1 << 20)).toFixed(1)} MB` : n >= 1024 ? `${Math.round(n / 1024)} KB` : `${n} bytes`); const sha = (s) => (s ? s.slice(0, 7) : ''); const model = (m) => m || 'harness default'; + // stageModel prefers the model a role actually ran over the one configured + // for it: "harness default" never answered which model that turned out to + // be, and an alias never showed the version behind it. + const stageModel = (configured, ran) => { + if (!ran) return model(configured); + if (!configured) return `${ran} (harness default)`; + return ran === configured ? ran : `${ran} (configured ${configured})`; + }; const tokens = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${Math.round(n / 1e3)}K` : String(n)); const usageTokens = (u) => (u.input || 0) + (u.cache_write || 0) + (u.cache_read || 0) + (u.output || 0); // issueLink names a run's linked issue the way its tracker does: a Linear @@ -615,7 +623,8 @@ const push = (at, el) => items.push({ at: at || '', el }); for (const a of stage.attempts) { const models = a.models || {}; - const modelText = models.worker || models.supervisor ? ` with ${model(models.worker)}` : ''; + const ran = [models.worker && `worker on ${models.worker}`, models.supervisor && `supervisor on ${models.supervisor}`].filter(Boolean); + const modelText = ran.length ? ` (${ran.join(', ')})` : ''; push(a.started_at, h('div', { class: 'event' }, h('span', {}, `Attempt ${a.number} started${modelText} `, timeEl(a.started_at)))); const p = a.progress; if (a.state === 'running') { @@ -878,8 +887,8 @@ path, h('button', { type: 'submit', class: 'btn btn-small', text: 'Attach' }))); const stageFacts = stage ? h('section', {}, h('h3', { text: `${stage.id} settings` }), h('dl', { class: 'facts' }, - h('dt', { text: 'Worker' }), h('dd', { text: model(stage.models.worker) }), - h('dt', { text: 'Supervisor' }), h('dd', { text: model(stage.models.supervisor) }), + h('dt', { text: 'Worker' }), h('dd', { text: stageModel(stage.models.worker, stage.models.ran_worker) }), + h('dt', { text: 'Supervisor' }), h('dd', { text: stageModel(stage.models.supervisor, stage.models.ran_supervisor) }), h('dt', { text: 'Attempts' }), h('dd', { text: `${stage.limits.attempts} of ${stage.limits.max_attempts}` }), h('dt', { text: 'Time limit' }), h('dd', { text: `${duration(stage.limits.attempt_seconds)} each` }), h('dt', { text: 'Nudged after' }), h('dd', { text: `${duration(stage.limits.stall_seconds)} silent` }))) : null; diff --git a/internal/web/server_test.go b/internal/web/server_test.go index e0f68a9..9555856 100644 --- a/internal/web/server_test.go +++ b/internal/web/server_test.go @@ -348,3 +348,35 @@ func TestAScreenshotReachesThePageAsAnImageNotAPlaceholder(t *testing.T) { t.Fatalf("the screenshot is not in the page:\n%s", page.String()) } } + +func TestTheStagePanelReportsTheModelAnAttemptActuallyRan(t *testing.T) { + c, err := workflow.Parse([]byte(config)) + if err != nil { + t.Fatal(err) + } + r := run(t, c) + rev := r.Current() + // Nothing is configured for task, so the configuration cannot say which + // model it used; the harness reporting one is the only source. + rev.Attempts = append(rev.Attempts, workflow.Attempt{ + ID: "attempt_1", Node: "task", Number: 1, State: "running", + Models: map[string]string{"worker": "claude-sonnet-5-20260115"}, + }) + + view := runView(*r, time.Now()) + var stage StageView + for _, s := range view.Revisions[0].Stages { + if s.ID == "task" { + stage = s + } + } + if stage.Models.Worker != "" { + t.Fatalf("task has a configured worker model: %q", stage.Models.Worker) + } + if stage.Models.RanWorker != "claude-sonnet-5-20260115" { + t.Fatalf("the model that ran is not reported: %+v", stage.Models) + } + if stage.Models.RanSupervisor != "" { + t.Fatalf("invented a supervisor model: %q", stage.Models.RanSupervisor) + } +} diff --git a/internal/web/view.go b/internal/web/view.go index ae4d195..bfe304a 100644 --- a/internal/web/view.go +++ b/internal/web/view.go @@ -119,10 +119,15 @@ type LimitView struct { StallSeconds int `json:"stall_seconds"` } -// ModelView names each role's effective model; empty is the harness default. +// ModelView names each role's model. Worker and Supervisor are what the +// configuration asks for, empty meaning the harness default; RanWorker and +// RanSupervisor are what the latest attempt's harness reported actually +// running, which is the only place an unconfigured role's model is known. type ModelView struct { - Worker string `json:"worker"` - Supervisor string `json:"supervisor"` + Worker string `json:"worker"` + Supervisor string `json:"supervisor"` + RanWorker string `json:"ran_worker,omitempty"` + RanSupervisor string `json:"ran_supervisor,omitempty"` } // AwaitingView is the exact result a person approves. @@ -277,6 +282,9 @@ func revisionView(v *workflow.Revision, current bool, now time.Time) RevisionVie continue } s.Attempts = append(s.Attempts, a) + if len(a.Models) > 0 { + s.Models.RanWorker, s.Models.RanSupervisor = a.Models["worker"], a.Models["supervisor"] + } s.Limits.Attempts++ if a.State == "awaiting-approval" && a.Result != nil { s.Awaiting = &AwaitingView{Attempt: a.ID, WorkDigest: a.Result.WorkDigest(), Result: *a.Result}