diff --git a/crates/tracedecay-agent-hosts/src/agents/codex.rs b/crates/tracedecay-agent-hosts/src/agents/codex.rs index 5a3740c85d..6d29c4388e 100644 --- a/crates/tracedecay-agent-hosts/src/agents/codex.rs +++ b/crates/tracedecay-agent-hosts/src/agents/codex.rs @@ -89,6 +89,23 @@ impl AgentIntegration for CodexIntegration { ctx: &InstallContext, ) -> Result { install_codex_plugin(&ctx.home, &ctx.tracedecay_bin)?; + // Core apply drives `codex plugin add` when the host CLI is present. + // When it is not, stop with the same backtick remediation preflight + // uses so operators (and lifecycle tests) can activate natively. + if plugin_registry::require_codex_plugin_cli().is_err() { + let marketplace_name = codex_exact_personal_marketplace_name(&ctx.home) + .ok() + .flatten() + .unwrap_or_else(|| codex_cached_marketplace_name(&ctx.home)); + return Ok(NonInteractiveInstallOutcome::DeferredUserAction( + DeferredUserAction { + remediation: format!( + "Codex activates plugins through its native cache. Run `codex plugin add tracedecay@{marketplace_name}` after TraceDecay stages the source package." + ), + staged_paths: Vec::new(), + }, + )); + } Ok(NonInteractiveInstallOutcome::Ready) } diff --git a/crates/tracedecay-cli/src/agent_cmd.rs b/crates/tracedecay-cli/src/agent_cmd.rs index e8107c66f1..082805f642 100644 --- a/crates/tracedecay-cli/src/agent_cmd.rs +++ b/crates/tracedecay-cli/src/agent_cmd.rs @@ -1175,9 +1175,22 @@ fn host_bundle_error_for_agent( == tracedecay_agent_hosts::agents::host_bundle::HostBundleError::UnsupportedCapability { return tracedecay_domain::errors::TraceDecayError::Config { - message: "Codex plugin activation could not be completed through `codex plugin add`. \ - Confirm the `codex` CLI is on PATH and retry; hook trust still requires \ - `/hooks` inside Codex after a successful add." + message: "Codex activates plugins through its native cache. Run `codex plugin add \ + tracedecay@personal` after TraceDecay stages the source package. Confirm \ + the `codex` CLI is on PATH and retry; hook trust still requires `/hooks` \ + inside Codex after a successful add." + .to_string(), + }; + } + if matches!( + &error, + tracedecay_agent_hosts::agents::host_bundle::HostBundleError::HostCliUnavailable { .. } + ) && agent_id == "codex" + { + return tracedecay_domain::errors::TraceDecayError::Config { + message: "Codex activates plugins through its native cache. Run `codex plugin add \ + tracedecay@personal` after TraceDecay stages the source package. Install \ + the `codex` CLI or add it to PATH, then retry." .to_string(), }; } diff --git a/crates/tracedecay-session-runtime/src/session_retrieval/lcm.rs b/crates/tracedecay-session-runtime/src/session_retrieval/lcm.rs index 855d5310e7..53cb997df1 100644 --- a/crates/tracedecay-session-runtime/src/session_retrieval/lcm.rs +++ b/crates/tracedecay-session-runtime/src/session_retrieval/lcm.rs @@ -169,6 +169,27 @@ impl DaemonSessionRetrievalService { /// projection is pending — so the worker's own serving state, with its /// backlog and blocker, is the answer instead. Once the worker is current /// the temporal outcome stands: nothing is going to project the session. + /// + /// A missing refresh worker is not convergence: core/direct mounts never + /// attach one, and zero rows there are evidence of absence, not a pending + /// catch-up. Only historical-convergence staleness remaps absence. + fn historically_converging_unavailable(&self) -> Option { + let unavailable = self.refresh_not_current()?; + match unavailable.reason { + SessionRetrievalUnavailableReason::HistoricalConvergence + | SessionRetrievalUnavailableReason::HistoricalRetry + | SessionRetrievalUnavailableReason::HistoricalBlocked => Some(unavailable), + SessionRetrievalUnavailableReason::ServiceNotConfigured + | SessionRetrievalUnavailableReason::RefreshWorkerMissing + | SessionRetrievalUnavailableReason::RefreshWorkerRecovering + | SessionRetrievalUnavailableReason::RefreshWorkerStalled + | SessionRetrievalUnavailableReason::RefreshWorkerStopped + | SessionRetrievalUnavailableReason::TemporalStoreUnavailable + | SessionRetrievalUnavailableReason::TemporalStoreReadFailed + | SessionRetrievalUnavailableReason::HydrationUnavailable => None, + } + } + fn converging_projection_unavailable( &self, unavailable: &SessionRetrievalUnavailable, @@ -176,7 +197,7 @@ impl DaemonSessionRetrievalService { if unavailable.reason != SessionRetrievalUnavailableReason::TemporalStoreUnavailable { return None; } - self.refresh_not_current() + self.historically_converging_unavailable() } #[hotpath::measure(label = "daemon.session_retrieval.lcm_describe", future = true)] @@ -302,11 +323,12 @@ impl DaemonSessionRetrievalService { (Some(result), retrieval) } // Zero temporal rows for the session is only evidence of absence - // once the refresh worker is current; while it is still - // converging history the honest answer is that state, not a - // complete description at generation zero. + // once history is not still converging. A missing worker (core / + // direct mounts) is not convergence — treat CompleteZero as + // absence there. While historical catch-up is in flight, surface + // that state instead of a complete description at generation zero. SessionRetrievalOutcome::CompleteZero { .. } - if direct.is_none() && self.refresh_not_current().is_some() => + if direct.is_none() && self.historically_converging_unavailable().is_some() => { return describe_retrieval_outcome( outcome, diff --git a/crates/tracedecay/src/daemon/core_proxy.rs b/crates/tracedecay/src/daemon/core_proxy.rs index fa9502d1c2..cda1573ed3 100644 --- a/crates/tracedecay/src/daemon/core_proxy.rs +++ b/crates/tracedecay/src/daemon/core_proxy.rs @@ -160,8 +160,13 @@ pub(crate) async fn proxy_transport_to_daemon_with_drain_bound( let (mut reader, mut writer) = transport.split(); let (input_tx, mut input_rx) = tokio::sync::mpsc::unbounded_channel(); let (eof_tx, mut eof_rx) = tokio::sync::watch::channel(false); + // Keep a Sender alive for the whole proxy lifetime. `read_host` only + // marks EOF; if it owned the sole Sender, dropping it on host close would + // make later `eof.changed()` calls fail as "monitor closed" and abort + // before the in-flight daemon response could be drained to the host. + let eof_signal = eof_tx.clone(); - let read_host = async { + let read_host = async move { loop { match reader.read_line().await { Ok(Some(line)) => { @@ -170,7 +175,7 @@ pub(crate) async fn proxy_transport_to_daemon_with_drain_bound( } } Ok(None) => { - let _ = eof_tx.send(true); + let _ = eof_signal.send(true); return Ok(()); } Err(error) => return Err(error.into()), @@ -186,7 +191,9 @@ pub(crate) async fn proxy_transport_to_daemon_with_drain_bound( &mut writer, drain_bound, ); - tokio::try_join!(read_host, proxy)?; + let result = tokio::try_join!(read_host, proxy); + drop(eof_tx); + result?; Ok(()) } diff --git a/crates/tracedecay/tests/transport_acceptance_suite/graph_rebuild_status_test.rs b/crates/tracedecay/tests/transport_acceptance_suite/graph_rebuild_status_test.rs index 51d23de6ed..5b02449a74 100644 --- a/crates/tracedecay/tests/transport_acceptance_suite/graph_rebuild_status_test.rs +++ b/crates/tracedecay/tests/transport_acceptance_suite/graph_rebuild_status_test.rs @@ -116,11 +116,14 @@ async fn search( project: &Path, query: &str, ) -> Value { + // Keep the page tiny: a generation-scale refresh batch otherwise returns + // multi-dozen-KiB candidate bodies that MCP truncates into a handle, and + // the wait helpers never see top-level `results` / `code_generation`. tool( harness, project, "tracedecay_search", - json!({"query": query, "limit": 20, "format": "json"}), + json!({"query": query, "limit": 3, "format": "json"}), ) .await } diff --git a/crates/tracedecay/tests/transport_acceptance_suite/typed_terminal_restart_acceptance.rs b/crates/tracedecay/tests/transport_acceptance_suite/typed_terminal_restart_acceptance.rs index 76c5468628..8c395ff1d4 100644 --- a/crates/tracedecay/tests/transport_acceptance_suite/typed_terminal_restart_acceptance.rs +++ b/crates/tracedecay/tests/transport_acceptance_suite/typed_terminal_restart_acceptance.rs @@ -175,6 +175,11 @@ fn add_fact_settling_after_its_deadline( barrier_dir: &Path, content: &str, ) -> Value { + // One-shot markers: clear a previous park so a leftover `arrived` cannot + // make this helper release before the CLI request reaches the boundary. + for marker in ["armed", "claimed", "arrived", "release"] { + let _ = std::fs::remove_file(barrier_dir.join(marker)); + } std::fs::write(barrier_dir.join("armed"), b"armed\n").expect("arm the fact commit barrier"); let mut command = tool_command( @@ -226,8 +231,18 @@ fn add_fact_settling_after_its_deadline( // test spawned it, so `spawn + deadline` can still be earlier than the real // expiry. Arrival is strictly after that clock started, so holding a full // deadline plus a margin beyond arrival always outlives it. + // + // Keep asserting the CLI is still blocked: an early success here means the + // commit path skipped the barrier (or a foreign claim wrote `arrived`). let release_at = arrived_at + PARTIAL_EFFECT_DEADLINE + Duration::from_secs(1); while Instant::now() < release_at { + assert!( + child + .try_wait() + .expect("inspect the parked fact_store add") + .is_none(), + "the fact_store add settled before its request deadline could expire at the commit barrier" + ); std::thread::sleep(Duration::from_millis(20)); } std::fs::write(barrier_dir.join("release"), b"release\n")