diff --git a/docs/src/content/docs/config-reference.mdx b/docs/src/content/docs/config-reference.mdx index 0a3edca..9e82fbe 100644 --- a/docs/src/content/docs/config-reference.mdx +++ b/docs/src/content/docs/config-reference.mdx @@ -162,6 +162,24 @@ harness's own default (no `--model` flag). Readiness probes for `harness.worker`/`harness.supervisor` always use the run-wide model, since a connection probe has no stage in scope. +### Which model actually ran + +A role with no configured model is invoked without `--model`, and the harness +chooses for itself, so the configuration cannot say what it used. Each attempt +therefore records what its harness reports running, and the TUI stage view, +the dashboard stage panel and the run timeline show that in preference to the +configured value: + +| Configured | Reported | Shown | +| --- | --- | --- | +| unset | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (harness default)` | +| `sonnet` | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (configured sonnet)` | +| `opus` | `opus` | `opus` | +| `opus` | nothing reported | `opus` | + +Claude reports its model on the `init` event of its stream. A harness that +reports nothing leaves the configured value showing, rather than a guess. + `envctl run show` and `envctl run readiness` print each node's effective worker and supervisor model next to its attempt budget. diff --git a/internal/agent/claude.go b/internal/agent/claude.go index a766dac..a5c3a82 100644 --- a/internal/agent/claude.go +++ b/internal/agent/claude.go @@ -157,17 +157,39 @@ func (c Claude) Result(ctx context.Context, id string) (json.RawMessage, error) return readResult(ctx, c.Guest, id) } func (c Claude) Session(stream string) string { + session, _ := claudeInit(stream) + return session +} + +// Model reports the model Claude reports on the init event, which is the one +// it resolved for itself when the invocation named none. +func (c Claude) Model(stream string) string { + _, model := claudeInit(stream) + return model +} + +// claudeInit reads the session and model out of the stream's init event, the +// first event Claude emits. +func claudeInit(stream string) (session, model string) { for _, line := range strings.Split(stream, "\n") { var event struct { Type string `json:"type"` Subtype string `json:"subtype"` Session string `json:"session_id"` + Model string `json:"model"` + } + if json.Unmarshal([]byte(line), &event) != nil || event.Type != "system" || event.Subtype != "init" { + continue + } + if sessionID.MatchString(event.Session) { + session = event.Session } - if json.Unmarshal([]byte(line), &event) == nil && event.Type == "system" && event.Subtype == "init" && sessionID.MatchString(event.Session) { - return event.Session + model = event.Model + if session != "" { + return session, model } } - return "" + return session, model } // Preserve the native event stream for progress/reconnect. Only a successful diff --git a/internal/agent/claude_test.go b/internal/agent/claude_test.go index 45397ea..3409ee9 100644 --- a/internal/agent/claude_test.go +++ b/internal/agent/claude_test.go @@ -85,3 +85,21 @@ func TestInvocationEnvironmentIsScopedToEnvctlValues(t *testing.T) { } } } + +func TestClaudeReportsTheModelItResolvedForItself(t *testing.T) { + // An invocation with no --model still says which model it ran, on the + // same init event the session comes from. + stream := `{"type":"system","subtype":"init","session_id":"01a08f59-7073-7a83-b959-867ca896ce48","model":"claude-sonnet-5-20260115"} +{"type":"assistant"}` + c := Claude{} + if got := c.Model(stream); got != "claude-sonnet-5-20260115" { + t.Fatalf("model not read from the init event: %q", got) + } + if got := c.Session(stream); got != "01a08f59-7073-7a83-b959-867ca896ce48" { + t.Fatalf("reading the model broke the session: %q", got) + } + // A stream that never says leaves the surfaces on the configured value. + if got := c.Model(`{"type":"assistant"}`); got != "" { + t.Fatalf("invented a model: %q", got) + } +} diff --git a/internal/agent/codex.go b/internal/agent/codex.go index 98211fd..eb85027 100644 --- a/internal/agent/codex.go +++ b/internal/agent/codex.go @@ -341,6 +341,22 @@ func (b *limitBuffer) Write(p []byte) (int, error) { return b.data.Write(p) } +// Model reports the model an event stream says it ran. Codex's documented +// events carry no model, so this finds one only if a stream reports it under +// the usual key; an empty result leaves the surfaces showing the configured +// model rather than a guess. +func Model(stream string) string { + for _, line := range strings.Split(stream, "\n") { + var event struct { + Model string `json:"model"` + } + if json.Unmarshal([]byte(line), &event) == nil && event.Model != "" { + return event.Model + } + } + return "" +} + // Session extracts the explicit session identity from Codex's structured stream. // It never selects the globally most recent conversation when resuming. func Session(stream string) string { diff --git a/internal/agent/harness.go b/internal/agent/harness.go index 6ff5d9f..cffc32d 100644 --- a/internal/agent/harness.go +++ b/internal/agent/harness.go @@ -16,6 +16,11 @@ type Harness interface { Start(context.Context, Invocation, Credential) (guestjob.Status, error) Result(context.Context, string) (json.RawMessage, error) Session(string) string + // Model reports the model the harness actually ran, read from its own + // event stream. An invocation with no model names one the harness chose, + // which the configuration never records; empty means the stream did not + // say. + Model(string) string // Activity summarizes the native event stream for live progress display. Activity(string) []string // Usage sums the model usage the native event stream reports. @@ -52,3 +57,4 @@ func ResolveCredential(root, kind, reference string) (Credential, error) { } func (c Codex) Session(stream string) string { return Session(stream) } +func (c Codex) Model(stream string) string { return Model(stream) } diff --git a/internal/engine/attempt.go b/internal/engine/attempt.go index ff87f16..f2a7ccd 100644 --- a/internal/engine/attempt.go +++ b/internal/engine/attempt.go @@ -131,6 +131,17 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w }) return true, err } + if models := newModels(a.Models, observation.Models); len(models) > 0 { + _, err = e.update(ctx, id, revision, "attempt.models", func(_ *workflow.Run, v *workflow.Revision) error { + current := v.Attempt(a.ID) + if current == nil { + return workflow.ErrConflict + } + current.Models = models + return nil + }) + return true, err + } if observation.Session != "" && a.Session != observation.Session { _, err = e.update(ctx, id, revision, "attempt.session", func(_ *workflow.Run, v *workflow.Revision) error { current := v.Attempt(a.ID) @@ -226,3 +237,25 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w } return false, nil } + +// newModels merges what a harness reported running over what an attempt +// recorded from configuration, returning nil when nothing changed. A reported +// model is the more precise of the two: it names what an unset role resolved +// to, and the exact version behind an alias. +func newModels(recorded, reported map[string]string) map[string]string { + changed := false + merged := map[string]string{} + for role, model := range recorded { + merged[role] = model + } + for role, model := range reported { + if model != "" && merged[role] != model { + merged[role] = model + changed = true + } + } + if !changed { + return nil + } + return merged +} diff --git a/internal/engine/attempt_test.go b/internal/engine/attempt_test.go new file mode 100644 index 0000000..2b93328 --- /dev/null +++ b/internal/engine/attempt_test.go @@ -0,0 +1,24 @@ +package engine + +import "testing" + +func TestAReportedModelReplacesTheConfiguredOneOnTheAttempt(t *testing.T) { + // A role with nothing configured is exactly the case the configuration + // cannot answer, so what the harness reports has to win. + got := newModels(map[string]string{"supervisor": "opus"}, map[string]string{"worker": "claude-sonnet-5-20260115"}) + if got["worker"] != "claude-sonnet-5-20260115" || got["supervisor"] != "opus" { + t.Fatalf("merge lost a role: %v", got) + } + // An alias resolves to its version. + got = newModels(map[string]string{"worker": "sonnet"}, map[string]string{"worker": "claude-sonnet-5-20260115"}) + if got["worker"] != "claude-sonnet-5-20260115" { + t.Fatalf("alias not replaced by the version that ran: %v", got) + } + // Nothing new means no write, so an unchanged attempt is not rewritten. + if got = newModels(map[string]string{"worker": "opus"}, map[string]string{"worker": "opus"}); got != nil { + t.Fatalf("rewrote an unchanged attempt: %v", got) + } + if got = newModels(map[string]string{"worker": "opus"}, nil); got != nil { + t.Fatalf("rewrote when the harness said nothing: %v", got) + } +} diff --git a/internal/engine/engine.go b/internal/engine/engine.go index 1e82912..6d7c57c 100644 --- a/internal/engine/engine.go +++ b/internal/engine/engine.go @@ -32,6 +32,9 @@ type Observation struct { Result *workflow.Result Detail string // already redacted by the backend Session string // explicit harness session identity for durable continuation + // Models is the model each role's harness reported running, by role. It + // is what actually ran, as opposed to what the configuration asked for. + Models map[string]string // Progress and Delivered are display state for running attempts. Delivered // lists user messages already submitted to an agent invocation. Progress *workflow.Progress diff --git a/internal/localexec/attempt.go b/internal/localexec/attempt.go index f285ee1..a686fda 100644 --- a/internal/localexec/attempt.go +++ b/internal/localexec/attempt.go @@ -44,6 +44,10 @@ type attemptRecord struct { Steering []workflow.Delivery `json:"steering,omitempty"` Live map[string]*liveRole `json:"live,omitempty"` SupervisorSession string `json:"supervisor_session,omitempty"` + // Models is the model each role's harness reported running, by role. It + // records what an invocation with no configured model resolved to, which + // the configuration cannot say. + Models map[string]string `json:"models,omitempty"` // Stall is the durable no-output watch per agent role. Stall map[string]*stallState `json:"stall,omitempty"` } @@ -426,6 +430,12 @@ func (b *Backend) pollJob(ctx context.Context, a engine.Assignment, r *attemptRe } else { r.Session = harness.Session(r.Logs[id]) } + if model := harness.Model(r.Logs[id]); model != "" { + if r.Models == nil { + r.Models = map[string]string{} + } + r.Models[role] = model + } } if err = b.save(a, r); err != nil { return status, err @@ -446,7 +456,7 @@ func (b *Backend) failed(a engine.Assignment, r *attemptRecord, detail string) ( if err := b.save(a, r); err != nil { return engine.Observation{}, err } - return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil } func (b *Backend) workerFailed(ctx context.Context, a engine.Assignment, r *attemptRecord, detail string) (engine.Observation, error) { @@ -524,7 +534,7 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa return engine.Observation{}, err } running := func() (engine.Observation, error) { - return engine.Observation{State: "running", Session: r.Session, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "running", Session: r.Session, Models: r.Models, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil } // agentJob reconciles one role's active generation: it finishes stopping a // superseded generation, starts a missing job, records delivery once the @@ -822,11 +832,11 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa if err = b.save(a, r); err != nil { return engine.Observation{}, err } - return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil case "completed": - return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil case "failed": - return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil + return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil default: return engine.Observation{}, errors.New("unknown durable attempt phase") } diff --git a/internal/tui/agents_test.go b/internal/tui/agents_test.go index eb19168..9cb9588 100644 --- a/internal/tui/agents_test.go +++ b/internal/tui/agents_test.go @@ -48,3 +48,25 @@ func TestConversationModelLineCoexistsWithAttemptDetail(t *testing.T) { t.Fatalf("conversation lost either the model line or the attempt detail:\n%s", view) } } + +func TestTheStageViewNamesTheModelThatActuallyRan(t *testing.T) { + // The whole point: "harness default" never said which model that was. + if got := stageModelTUI("", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (harness default)" { + t.Fatalf("unconfigured role: %q", got) + } + // An alias shows the version behind it. + if got := stageModelTUI("sonnet", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (configured sonnet)" { + t.Fatalf("alias: %q", got) + } + // Agreement is not worth two names. + if got := stageModelTUI("opus", "opus"); got != "opus" { + t.Fatalf("agreement: %q", got) + } + // Nothing reported falls back rather than inventing. + if got := stageModelTUI("", ""); got != "harness default" { + t.Fatalf("nothing known: %q", got) + } + if got := stageModelTUI("opus", ""); got != "opus" { + t.Fatalf("configured only: %q", got) + } +} diff --git a/internal/tui/model.go b/internal/tui/model.go index ddeb98a..3630d93 100644 --- a/internal/tui/model.go +++ b/internal/tui/model.go @@ -901,7 +901,8 @@ func (m Model) details() string { switch panels[m.Panel] { case "Conversation": agents := rev.Config.NodeAgents(node) - lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", formatModelTUI(agents.Worker.Model), formatModelTUI(agents.Supervisor.Model))} + ran := ranModels(rev, node) + lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", stageModelTUI(agents.Worker.Model, ran["worker"]), stageModelTUI(agents.Supervisor.Model, ran["supervisor"]))} for _, a := range rev.Attempts { if a.Node == node { lines = append(lines, fmt.Sprintf("Worker attempt %d: %s", a.Number, a.State)) @@ -1014,6 +1015,33 @@ func formatModelTUI(model string) string { } return model } + +// ranModels is what a node's latest attempt reported actually running, which +// names the model an unconfigured role resolved to. +func ranModels(rev *workflow.Revision, node string) map[string]string { + out := map[string]string{} + for _, a := range rev.Attempts { + if a.Node == node && len(a.Models) > 0 { + out = a.Models + } + } + return out +} + +// stageModelTUI prefers what a role actually ran over what was configured for +// it, and says so when the two differ, because "harness default" on its own +// never answered which model that was. +func stageModelTUI(configured, ran string) string { + switch { + case ran == "": + return formatModelTUI(configured) + case configured == "": + return ran + " (harness default)" + case ran != configured: + return ran + " (configured " + configured + ")" + } + return ran +} func pretty(v any) string { b, err := json.MarshalIndent(v, "", " ") if err != nil { diff --git a/internal/web/assets/app.js b/internal/web/assets/app.js index de71674..ca3d0f0 100644 --- a/internal/web/assets/app.js +++ b/internal/web/assets/app.js @@ -56,6 +56,14 @@ const bytes = (n) => (n >= 1 << 20 ? `${(n / (1 << 20)).toFixed(1)} MB` : n >= 1024 ? `${Math.round(n / 1024)} KB` : `${n} bytes`); const sha = (s) => (s ? s.slice(0, 7) : ''); const model = (m) => m || 'harness default'; + // stageModel prefers the model a role actually ran over the one configured + // for it: "harness default" never answered which model that turned out to + // be, and an alias never showed the version behind it. + const stageModel = (configured, ran) => { + if (!ran) return model(configured); + if (!configured) return `${ran} (harness default)`; + return ran === configured ? ran : `${ran} (configured ${configured})`; + }; const tokens = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${Math.round(n / 1e3)}K` : String(n)); const usageTokens = (u) => (u.input || 0) + (u.cache_write || 0) + (u.cache_read || 0) + (u.output || 0); // issueLink names a run's linked issue the way its tracker does: a Linear @@ -615,7 +623,8 @@ const push = (at, el) => items.push({ at: at || '', el }); for (const a of stage.attempts) { const models = a.models || {}; - const modelText = models.worker || models.supervisor ? ` with ${model(models.worker)}` : ''; + const ran = [models.worker && `worker on ${models.worker}`, models.supervisor && `supervisor on ${models.supervisor}`].filter(Boolean); + const modelText = ran.length ? ` (${ran.join(', ')})` : ''; push(a.started_at, h('div', { class: 'event' }, h('span', {}, `Attempt ${a.number} started${modelText} `, timeEl(a.started_at)))); const p = a.progress; if (a.state === 'running') { @@ -878,8 +887,8 @@ path, h('button', { type: 'submit', class: 'btn btn-small', text: 'Attach' }))); const stageFacts = stage ? h('section', {}, h('h3', { text: `${stage.id} settings` }), h('dl', { class: 'facts' }, - h('dt', { text: 'Worker' }), h('dd', { text: model(stage.models.worker) }), - h('dt', { text: 'Supervisor' }), h('dd', { text: model(stage.models.supervisor) }), + h('dt', { text: 'Worker' }), h('dd', { text: stageModel(stage.models.worker, stage.models.ran_worker) }), + h('dt', { text: 'Supervisor' }), h('dd', { text: stageModel(stage.models.supervisor, stage.models.ran_supervisor) }), h('dt', { text: 'Attempts' }), h('dd', { text: `${stage.limits.attempts} of ${stage.limits.max_attempts}` }), h('dt', { text: 'Time limit' }), h('dd', { text: `${duration(stage.limits.attempt_seconds)} each` }), h('dt', { text: 'Nudged after' }), h('dd', { text: `${duration(stage.limits.stall_seconds)} silent` }))) : null; diff --git a/internal/web/server_test.go b/internal/web/server_test.go index e0f68a9..9555856 100644 --- a/internal/web/server_test.go +++ b/internal/web/server_test.go @@ -348,3 +348,35 @@ func TestAScreenshotReachesThePageAsAnImageNotAPlaceholder(t *testing.T) { t.Fatalf("the screenshot is not in the page:\n%s", page.String()) } } + +func TestTheStagePanelReportsTheModelAnAttemptActuallyRan(t *testing.T) { + c, err := workflow.Parse([]byte(config)) + if err != nil { + t.Fatal(err) + } + r := run(t, c) + rev := r.Current() + // Nothing is configured for task, so the configuration cannot say which + // model it used; the harness reporting one is the only source. + rev.Attempts = append(rev.Attempts, workflow.Attempt{ + ID: "attempt_1", Node: "task", Number: 1, State: "running", + Models: map[string]string{"worker": "claude-sonnet-5-20260115"}, + }) + + view := runView(*r, time.Now()) + var stage StageView + for _, s := range view.Revisions[0].Stages { + if s.ID == "task" { + stage = s + } + } + if stage.Models.Worker != "" { + t.Fatalf("task has a configured worker model: %q", stage.Models.Worker) + } + if stage.Models.RanWorker != "claude-sonnet-5-20260115" { + t.Fatalf("the model that ran is not reported: %+v", stage.Models) + } + if stage.Models.RanSupervisor != "" { + t.Fatalf("invented a supervisor model: %q", stage.Models.RanSupervisor) + } +} diff --git a/internal/web/view.go b/internal/web/view.go index ae4d195..bfe304a 100644 --- a/internal/web/view.go +++ b/internal/web/view.go @@ -119,10 +119,15 @@ type LimitView struct { StallSeconds int `json:"stall_seconds"` } -// ModelView names each role's effective model; empty is the harness default. +// ModelView names each role's model. Worker and Supervisor are what the +// configuration asks for, empty meaning the harness default; RanWorker and +// RanSupervisor are what the latest attempt's harness reported actually +// running, which is the only place an unconfigured role's model is known. type ModelView struct { - Worker string `json:"worker"` - Supervisor string `json:"supervisor"` + Worker string `json:"worker"` + Supervisor string `json:"supervisor"` + RanWorker string `json:"ran_worker,omitempty"` + RanSupervisor string `json:"ran_supervisor,omitempty"` } // AwaitingView is the exact result a person approves. @@ -277,6 +282,9 @@ func revisionView(v *workflow.Revision, current bool, now time.Time) RevisionVie continue } s.Attempts = append(s.Attempts, a) + if len(a.Models) > 0 { + s.Models.RanWorker, s.Models.RanSupervisor = a.Models["worker"], a.Models["supervisor"] + } s.Limits.Attempts++ if a.State == "awaiting-approval" && a.Result != nil { s.Awaiting = &AwaitingView{Attempt: a.ID, WorkDigest: a.Result.WorkDigest(), Result: *a.Result}