Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions docs/src/content/docs/config-reference.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -162,6 +162,24 @@ harness's own default (no `--model` flag).
Readiness probes for `harness.worker`/`harness.supervisor` always use the
run-wide model, since a connection probe has no stage in scope.

### Which model actually ran

A role with no configured model is invoked without `--model`, and the harness
chooses for itself, so the configuration cannot say what it used. Each attempt
therefore records what its harness reports running, and the TUI stage view,
the dashboard stage panel and the run timeline show that in preference to the
configured value:

| Configured | Reported | Shown |
| --- | --- | --- |
| unset | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (harness default)` |
| `sonnet` | `claude-sonnet-5-20260115` | `claude-sonnet-5-20260115 (configured sonnet)` |
| `opus` | `opus` | `opus` |
| `opus` | nothing reported | `opus` |

Claude reports its model on the `init` event of its stream. A harness that
reports nothing leaves the configured value showing, rather than a guess.

`envctl run show` and `envctl run readiness` print each node's effective
worker and supervisor model next to its attempt budget.

Expand Down
28 changes: 25 additions & 3 deletions internal/agent/claude.go
Original file line number Diff line number Diff line change
Expand Up @@ -157,17 +157,39 @@ func (c Claude) Result(ctx context.Context, id string) (json.RawMessage, error)
return readResult(ctx, c.Guest, id)
}
func (c Claude) Session(stream string) string {
session, _ := claudeInit(stream)
return session
}

// Model reports the model Claude reports on the init event, which is the one
// it resolved for itself when the invocation named none.
func (c Claude) Model(stream string) string {
_, model := claudeInit(stream)
return model
}

// claudeInit reads the session and model out of the stream's init event, the
// first event Claude emits.
func claudeInit(stream string) (session, model string) {
for _, line := range strings.Split(stream, "\n") {
var event struct {
Type string `json:"type"`
Subtype string `json:"subtype"`
Session string `json:"session_id"`
Model string `json:"model"`
}
if json.Unmarshal([]byte(line), &event) != nil || event.Type != "system" || event.Subtype != "init" {
continue
}
if sessionID.MatchString(event.Session) {
session = event.Session
}
if json.Unmarshal([]byte(line), &event) == nil && event.Type == "system" && event.Subtype == "init" && sessionID.MatchString(event.Session) {
return event.Session
model = event.Model
if session != "" {
return session, model
}
}
return ""
return session, model
}

// Preserve the native event stream for progress/reconnect. Only a successful
Expand Down
18 changes: 18 additions & 0 deletions internal/agent/claude_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -85,3 +85,21 @@ func TestInvocationEnvironmentIsScopedToEnvctlValues(t *testing.T) {
}
}
}

func TestClaudeReportsTheModelItResolvedForItself(t *testing.T) {
// An invocation with no --model still says which model it ran, on the
// same init event the session comes from.
stream := `{"type":"system","subtype":"init","session_id":"01a08f59-7073-7a83-b959-867ca896ce48","model":"claude-sonnet-5-20260115"}
{"type":"assistant"}`
c := Claude{}
if got := c.Model(stream); got != "claude-sonnet-5-20260115" {
t.Fatalf("model not read from the init event: %q", got)
}
if got := c.Session(stream); got != "01a08f59-7073-7a83-b959-867ca896ce48" {
t.Fatalf("reading the model broke the session: %q", got)
}
// A stream that never says leaves the surfaces on the configured value.
if got := c.Model(`{"type":"assistant"}`); got != "" {
t.Fatalf("invented a model: %q", got)
}
}
16 changes: 16 additions & 0 deletions internal/agent/codex.go
Original file line number Diff line number Diff line change
Expand Up @@ -341,6 +341,22 @@ func (b *limitBuffer) Write(p []byte) (int, error) {
return b.data.Write(p)
}

// Model reports the model an event stream says it ran. Codex's documented
// events carry no model, so this finds one only if a stream reports it under
// the usual key; an empty result leaves the surfaces showing the configured
// model rather than a guess.
func Model(stream string) string {
for _, line := range strings.Split(stream, "\n") {
var event struct {
Model string `json:"model"`
}
if json.Unmarshal([]byte(line), &event) == nil && event.Model != "" {
return event.Model
}
}
return ""
}

// Session extracts the explicit session identity from Codex's structured stream.
// It never selects the globally most recent conversation when resuming.
func Session(stream string) string {
Expand Down
6 changes: 6 additions & 0 deletions internal/agent/harness.go
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,11 @@ type Harness interface {
Start(context.Context, Invocation, Credential) (guestjob.Status, error)
Result(context.Context, string) (json.RawMessage, error)
Session(string) string
// Model reports the model the harness actually ran, read from its own
// event stream. An invocation with no model names one the harness chose,
// which the configuration never records; empty means the stream did not
// say.
Model(string) string
// Activity summarizes the native event stream for live progress display.
Activity(string) []string
// Usage sums the model usage the native event stream reports.
Expand Down Expand Up @@ -52,3 +57,4 @@ func ResolveCredential(root, kind, reference string) (Credential, error) {
}

func (c Codex) Session(stream string) string { return Session(stream) }
func (c Codex) Model(stream string) string { return Model(stream) }
33 changes: 33 additions & 0 deletions internal/engine/attempt.go
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,17 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w
})
return true, err
}
if models := newModels(a.Models, observation.Models); len(models) > 0 {
_, err = e.update(ctx, id, revision, "attempt.models", func(_ *workflow.Run, v *workflow.Revision) error {
current := v.Attempt(a.ID)
if current == nil {
return workflow.ErrConflict
}
current.Models = models
return nil
})
return true, err
}
if observation.Session != "" && a.Session != observation.Session {
_, err = e.update(ctx, id, revision, "attempt.session", func(_ *workflow.Run, v *workflow.Revision) error {
current := v.Attempt(a.ID)
Expand Down Expand Up @@ -226,3 +237,25 @@ func (e *Engine) reconcileAttempt(ctx context.Context, run *workflow.Run, rev *w
}
return false, nil
}

// newModels merges what a harness reported running over what an attempt
// recorded from configuration, returning nil when nothing changed. A reported
// model is the more precise of the two: it names what an unset role resolved
// to, and the exact version behind an alias.
func newModels(recorded, reported map[string]string) map[string]string {
changed := false
merged := map[string]string{}
for role, model := range recorded {
merged[role] = model
}
for role, model := range reported {
if model != "" && merged[role] != model {
merged[role] = model
changed = true
}
}
if !changed {
return nil
}
return merged
}
24 changes: 24 additions & 0 deletions internal/engine/attempt_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
package engine

import "testing"

func TestAReportedModelReplacesTheConfiguredOneOnTheAttempt(t *testing.T) {
// A role with nothing configured is exactly the case the configuration
// cannot answer, so what the harness reports has to win.
got := newModels(map[string]string{"supervisor": "opus"}, map[string]string{"worker": "claude-sonnet-5-20260115"})
if got["worker"] != "claude-sonnet-5-20260115" || got["supervisor"] != "opus" {
t.Fatalf("merge lost a role: %v", got)
}
// An alias resolves to its version.
got = newModels(map[string]string{"worker": "sonnet"}, map[string]string{"worker": "claude-sonnet-5-20260115"})
if got["worker"] != "claude-sonnet-5-20260115" {
t.Fatalf("alias not replaced by the version that ran: %v", got)
}
// Nothing new means no write, so an unchanged attempt is not rewritten.
if got = newModels(map[string]string{"worker": "opus"}, map[string]string{"worker": "opus"}); got != nil {
t.Fatalf("rewrote an unchanged attempt: %v", got)
}
if got = newModels(map[string]string{"worker": "opus"}, nil); got != nil {
t.Fatalf("rewrote when the harness said nothing: %v", got)
}
}
3 changes: 3 additions & 0 deletions internal/engine/engine.go
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,9 @@ type Observation struct {
Result *workflow.Result
Detail string // already redacted by the backend
Session string // explicit harness session identity for durable continuation
// Models is the model each role's harness reported running, by role. It
// is what actually ran, as opposed to what the configuration asked for.
Models map[string]string
// Progress and Delivered are display state for running attempts. Delivered
// lists user messages already submitted to an agent invocation.
Progress *workflow.Progress
Expand Down
20 changes: 15 additions & 5 deletions internal/localexec/attempt.go
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,10 @@ type attemptRecord struct {
Steering []workflow.Delivery `json:"steering,omitempty"`
Live map[string]*liveRole `json:"live,omitempty"`
SupervisorSession string `json:"supervisor_session,omitempty"`
// Models is the model each role's harness reported running, by role. It
// records what an invocation with no configured model resolved to, which
// the configuration cannot say.
Models map[string]string `json:"models,omitempty"`
// Stall is the durable no-output watch per agent role.
Stall map[string]*stallState `json:"stall,omitempty"`
}
Expand Down Expand Up @@ -426,6 +430,12 @@ func (b *Backend) pollJob(ctx context.Context, a engine.Assignment, r *attemptRe
} else {
r.Session = harness.Session(r.Logs[id])
}
if model := harness.Model(r.Logs[id]); model != "" {
if r.Models == nil {
r.Models = map[string]string{}
}
r.Models[role] = model
}
}
if err = b.save(a, r); err != nil {
return status, err
Expand All @@ -446,7 +456,7 @@ func (b *Backend) failed(a engine.Assignment, r *attemptRecord, detail string) (
if err := b.save(a, r); err != nil {
return engine.Observation{}, err
}
return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
return engine.Observation{State: "failed", Detail: detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
}

func (b *Backend) workerFailed(ctx context.Context, a engine.Assignment, r *attemptRecord, detail string) (engine.Observation, error) {
Expand Down Expand Up @@ -524,7 +534,7 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa
return engine.Observation{}, err
}
running := func() (engine.Observation, error) {
return engine.Observation{State: "running", Session: r.Session, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
return engine.Observation{State: "running", Session: r.Session, Models: r.Models, Progress: b.progress(a, r), Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
}
// agentJob reconciles one role's active generation: it finishes stopping a
// superseded generation, starts a missing job, records delivery once the
Expand Down Expand Up @@ -822,11 +832,11 @@ func (b *Backend) Poll(ctx context.Context, a engine.Assignment) (engine.Observa
if err = b.save(a, r); err != nil {
return engine.Observation{}, err
}
return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
case "completed":
return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
return engine.Observation{State: "completed", Result: &r.Result, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
case "failed":
return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
return engine.Observation{State: "failed", Detail: r.Detail, Session: r.Session, Models: r.Models, Delivered: r.delivered(), Usage: b.usage(a, r)}, nil
default:
return engine.Observation{}, errors.New("unknown durable attempt phase")
}
Expand Down
22 changes: 22 additions & 0 deletions internal/tui/agents_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -48,3 +48,25 @@ func TestConversationModelLineCoexistsWithAttemptDetail(t *testing.T) {
t.Fatalf("conversation lost either the model line or the attempt detail:\n%s", view)
}
}

func TestTheStageViewNamesTheModelThatActuallyRan(t *testing.T) {
// The whole point: "harness default" never said which model that was.
if got := stageModelTUI("", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (harness default)" {
t.Fatalf("unconfigured role: %q", got)
}
// An alias shows the version behind it.
if got := stageModelTUI("sonnet", "claude-sonnet-5-20260115"); got != "claude-sonnet-5-20260115 (configured sonnet)" {
t.Fatalf("alias: %q", got)
}
// Agreement is not worth two names.
if got := stageModelTUI("opus", "opus"); got != "opus" {
t.Fatalf("agreement: %q", got)
}
// Nothing reported falls back rather than inventing.
if got := stageModelTUI("", ""); got != "harness default" {
t.Fatalf("nothing known: %q", got)
}
if got := stageModelTUI("opus", ""); got != "opus" {
t.Fatalf("configured only: %q", got)
}
}
30 changes: 29 additions & 1 deletion internal/tui/model.go
Original file line number Diff line number Diff line change
Expand Up @@ -901,7 +901,8 @@ func (m Model) details() string {
switch panels[m.Panel] {
case "Conversation":
agents := rev.Config.NodeAgents(node)
lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", formatModelTUI(agents.Worker.Model), formatModelTUI(agents.Supervisor.Model))}
ran := ranModels(rev, node)
lines := []string{fmt.Sprintf("Worker model: %s · Supervisor model: %s", stageModelTUI(agents.Worker.Model, ran["worker"]), stageModelTUI(agents.Supervisor.Model, ran["supervisor"]))}
for _, a := range rev.Attempts {
if a.Node == node {
lines = append(lines, fmt.Sprintf("Worker attempt %d: %s", a.Number, a.State))
Expand Down Expand Up @@ -1014,6 +1015,33 @@ func formatModelTUI(model string) string {
}
return model
}

// ranModels is what a node's latest attempt reported actually running, which
// names the model an unconfigured role resolved to.
func ranModels(rev *workflow.Revision, node string) map[string]string {
out := map[string]string{}
for _, a := range rev.Attempts {
if a.Node == node && len(a.Models) > 0 {
out = a.Models
}
}
return out
}

// stageModelTUI prefers what a role actually ran over what was configured for
// it, and says so when the two differ, because "harness default" on its own
// never answered which model that was.
func stageModelTUI(configured, ran string) string {
switch {
case ran == "":
return formatModelTUI(configured)
case configured == "":
return ran + " (harness default)"
case ran != configured:
return ran + " (configured " + configured + ")"
}
return ran
}
func pretty(v any) string {
b, err := json.MarshalIndent(v, "", " ")
if err != nil {
Expand Down
15 changes: 12 additions & 3 deletions internal/web/assets/app.js
Original file line number Diff line number Diff line change
Expand Up @@ -56,6 +56,14 @@
const bytes = (n) => (n >= 1 << 20 ? `${(n / (1 << 20)).toFixed(1)} MB` : n >= 1024 ? `${Math.round(n / 1024)} KB` : `${n} bytes`);
const sha = (s) => (s ? s.slice(0, 7) : '');
const model = (m) => m || 'harness default';
// stageModel prefers the model a role actually ran over the one configured
// for it: "harness default" never answered which model that turned out to
// be, and an alias never showed the version behind it.
const stageModel = (configured, ran) => {
if (!ran) return model(configured);
if (!configured) return `${ran} (harness default)`;
return ran === configured ? ran : `${ran} (configured ${configured})`;
};
const tokens = (n) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1e3 ? `${Math.round(n / 1e3)}K` : String(n));
const usageTokens = (u) => (u.input || 0) + (u.cache_write || 0) + (u.cache_read || 0) + (u.output || 0);
// issueLink names a run's linked issue the way its tracker does: a Linear
Expand Down Expand Up @@ -615,7 +623,8 @@
const push = (at, el) => items.push({ at: at || '', el });
for (const a of stage.attempts) {
const models = a.models || {};
const modelText = models.worker || models.supervisor ? ` with ${model(models.worker)}` : '';
const ran = [models.worker && `worker on ${models.worker}`, models.supervisor && `supervisor on ${models.supervisor}`].filter(Boolean);
const modelText = ran.length ? ` (${ran.join(', ')})` : '';
push(a.started_at, h('div', { class: 'event' }, h('span', {}, `Attempt ${a.number} started${modelText} `, timeEl(a.started_at))));
const p = a.progress;
if (a.state === 'running') {
Expand Down Expand Up @@ -878,8 +887,8 @@
path, h('button', { type: 'submit', class: 'btn btn-small', text: 'Attach' })));

const stageFacts = stage ? h('section', {}, h('h3', { text: `${stage.id} settings` }), h('dl', { class: 'facts' },
h('dt', { text: 'Worker' }), h('dd', { text: model(stage.models.worker) }),
h('dt', { text: 'Supervisor' }), h('dd', { text: model(stage.models.supervisor) }),
h('dt', { text: 'Worker' }), h('dd', { text: stageModel(stage.models.worker, stage.models.ran_worker) }),
h('dt', { text: 'Supervisor' }), h('dd', { text: stageModel(stage.models.supervisor, stage.models.ran_supervisor) }),
h('dt', { text: 'Attempts' }), h('dd', { text: `${stage.limits.attempts} of ${stage.limits.max_attempts}` }),
h('dt', { text: 'Time limit' }), h('dd', { text: `${duration(stage.limits.attempt_seconds)} each` }),
h('dt', { text: 'Nudged after' }), h('dd', { text: `${duration(stage.limits.stall_seconds)} silent` }))) : null;
Expand Down
Loading
Loading