Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ import type { CapabilityConfigurationEditor } from "../../data/chat";
import { PeriodicReportScheduleField } from "./periodic-report-schedule-field";
import { useWorkspaceI18n } from "./i18n";

type FieldCopy = Record<string, { description?: string; label?: string }>;
type FieldCopy = Record<string, { description?: string; label?: string; options?: Record<string, string> }>;
type ConfigurationField = CapabilityConfigurationEditor["fields"][number];
type FieldValue = boolean | number | string | string[] | Record<string, unknown> | null;
type FieldChange = (key: string, value: FieldValue) => void;
Expand Down Expand Up @@ -67,7 +67,7 @@ function ConfigurationFieldControl({ copy, field, id, onChange, value, timezone
<span>{label}</span>
<select disabled={readOnly} id={id} onChange={onChange ? (event) => onChange(field.key, event.target.value) : undefined} value={typeof value === "string" ? value : ""}>
<option value="" />
{(field.options ?? []).map((option) => <option key={option} value={option}>{field.key === "review_order" ? directionLabel(option) : option}</option>)}
{(field.options ?? []).map((option) => <option key={option} value={option}>{copy[field.key]?.options?.[option] ?? (field.key === "review_order" ? directionLabel(option) : option)}</option>)}
</select>
</label>
);
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@ type LocalizedCopy = Readonly<{
readOnlyReason?: string;
}>;

type FieldCopy = Record<string, Readonly<{ description?: string; label: string }>>;
type FieldCopy = Record<string, Readonly<{ description?: string; label: string; options?: Record<string, string> }>>;

const capabilityCopy: Record<WorkspaceLocale, Record<string, LocalizedCopy>> = {
en: {
Expand Down Expand Up @@ -36,7 +36,7 @@ const capabilityCopy: Record<WorkspaceLocale, Record<string, LocalizedCopy>> = {
},
explore_harness: {
displayName: "Explore Harness",
description: "Selects a capability-owned planning and research harness profile for bounded multi-step exploration.",
description: "Keeps an evidence graph of exploration, with optional branch planning. Planning includes evidence; worker permissions remain separate.",
},
lark_event_inbox: {
displayName: "Lark event inbox",
Expand Down Expand Up @@ -102,7 +102,7 @@ const capabilityCopy: Record<WorkspaceLocale, Record<string, LocalizedCopy>> = {
},
explore_harness: {
displayName: "探索 Harness",
description: "为有界的多步探索选择由能力负责的规划与研究 Harness profile。",
description: "记录探索证据图谱,可选开启分支规划;规划自动配套证据层,派生 Agent 权限独立。",
},
lark_event_inbox: {
displayName: "飞书事件收件箱",
Expand Down Expand Up @@ -225,7 +225,14 @@ export function localizeCapability(
};
}

export function localizedCapabilityFieldCopy(locale: WorkspaceLocale): FieldCopy {
export function localizedCapabilityFieldCopy(locale: WorkspaceLocale, capabilityId?: string): FieldCopy {
if (capabilityId === "explore_harness") return {...fieldCopy[locale], mode: locale === "zh-CN" ? {
label: "探索模式", description: "规划包含证据图谱;派生 Agent 和执行权限仍单独控制。",
options: {off: "关闭", evidence: "仅记录证据", planning: "证据与规划"},
} : {
label: "Exploration mode", description: "Planning includes the evidence graph; worker and execution permissions remain separate.",
options: {off: "Off", evidence: "Evidence only", planning: "Evidence and planning"},
}};
return fieldCopy[locale];
}

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -156,7 +156,7 @@ export function CapabilityDetailHeader({ capability, locale, source }: Readonly<
<summary>{locale === "zh-CN" ? "配置说明" : "Configuration help"}</summary>
<p>{localized.description}</p>
<dl>{capability.configuration_editor.fields.map((field) => {
const copy = localizedCapabilityFieldCopy(locale)[field.key];
const copy = localizedCapabilityFieldCopy(locale, capability.capability_id)[field.key];
const description = copy?.description ?? field.description;
return description ? <div key={field.key}><dt>{copy?.label ?? field.label}</dt><dd>{description}</dd></div> : null;
})}</dl>
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -308,7 +308,7 @@ function CapabilityCatalog({ callbacks, catalog, goalId, notification, onApplied
{editorMode === "guided" ? <section className="personal-capability-field-summary">
<CapabilityConfigurationFields
disabled={Boolean(busy)}
copy={localizedCapabilityFieldCopy(locale)}
copy={localizedCapabilityFieldCopy(locale, localizedSelected.capability_id)}
editor={localizedSelected.configuration_editor}
onChange={changeDraft}
value={draft}
Expand Down
11 changes: 11 additions & 0 deletions docs/architecture/rfcs/research-exploration-control-plane-v0.md
Original file line number Diff line number Diff line change
Expand Up @@ -1038,6 +1038,17 @@ This is a living RFC, not an append-only diary.
| 2026-08-13 | Separate eligibility from ranking: the control plane owns a legal bounded candidate set, while the model autonomously prioritizes among multiple eligible candidates. A selection receipt proves a scheduling choice, not research truth. Defer the protocol to M4 rather than adding it to the #3173 runtime slice. |
| 2026-10-04 | Propose §11.5 as a bounded M3 prerequisite: exact result -> scoped interpretation -> adopted task step/successor. Reuse Explore and task/replan receipts; keep Turn settlement, native scoring and scientific qualification independent. This RFC update delivers no automatic policy. |

The Explore Harness product entry now groups evidence-only and evidence-with-planning
modes under the existing `explore` capability. The typed configuration owner retains
legacy storage and flags; planning includes the graph without granting spawn or
publication authority. A turn-start hook requests bounded evidence and branch context
through the existing hook contract. This is an M3 entry/adoption prerequisite, not
qualification of experiment selection, result interpretation or scientific benefit.
CLI/file-state and packaged settings verify mode changes, stale-preview recovery and
feature-off behavior; live long-horizon trajectories must still establish actual
model adoption and continuation. The mode vocabulary is local to Explore configuration,
not a new kernel lifecycle or scheduler contract.

## 20. Acceptance Criteria for the RFC

Merge accepts this design basis. Implementation qualification still requires evidence that:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -950,6 +950,14 @@ evidence-backed terminal result。
| 2026-08-13 | 将 eligibility 与 ranking 分离:控制面拥有合法、有界 candidate set,模型在多个 eligible candidate 中自主择优;selection receipt 只证明调度选择,不构成 research truth。该协议延后到 M4,不进入 #3173 的首期 runtime。 |
| 2026-10-04 | 提议 §11.5 作为有界 M3 prerequisite:精确 result -> 有范围 interpretation -> adopted task step/successor。复用 Explore 和 task/replan receipt;Turn settlement、原生评分与科学资格独立。本次 RFC 更新不交付自动策略。 |

探索 Harness 的产品入口现在将仅证据和证据加规划模式统一到既有 `explore`
能力。类型化配置 owner 保留旧存储和 flag;规划配套图谱,但不授予 spawn 或外部
发布权限。轮次开始 hook 通过既有 hook 契约请求有界证据和分支上下文。这是 M3
入口与采纳的前置条件,不是实验选择、结果解释或科学收益的验收。
CLI/File 状态与打包设置验证模式切换、过期预览恢复和关闭态行为;真实长程轨迹
仍须验证模型实际采纳和后续推进。模式词汇仅属于 Explore 配置,不是新的 kernel
生命周期或调度契约。

## 20. RFC 验收标准

RFC 合入即接受设计依据。实现资格仍需要以下证据:
Expand Down
13 changes: 10 additions & 3 deletions examples/explore-configure-goal-smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -250,7 +250,7 @@ def main() -> int:
assert closed["written"] is True, closed
assert goal(registry)["spawn_policy"]["explore_harness"] == {"enabled": False}

graph_closed_harness_open = run_cli(
rejected_graphless_planning = run_cli(
registry,
runtime_root,
"configure-goal",
Expand All @@ -259,9 +259,16 @@ def main() -> int:
"--no-explore-graph-enabled",
"--explore-harness-enabled",
"--execute",
check=False,
)
assert graph_closed_harness_open["written"] is True, graph_closed_harness_open
assert goal(registry)["explore_graph"] == {"enabled": False}
assert rejected_graphless_planning["ok"] is False, rejected_graphless_planning
assert "planning requires its evidence graph" in rejected_graphless_planning["error"]
assert goal(registry)["explore_graph"] == {"enabled": True}
assert goal(registry)["spawn_policy"]["explore_harness"] == {"enabled": False}
unified = run_cli(registry, runtime_root, "configure-goal", "--goal-id", GOAL_ID,
"--explore-mode", "planning", "--execute")
assert unified["ok"]
assert goal(registry)["explore_graph"] == {"enabled": True}
assert goal(registry)["spawn_policy"]["explore_harness"] == {"enabled": True}

conflict = run_cli(
Expand Down
5 changes: 3 additions & 2 deletions loopx/capabilities/configuration_ui.py
Original file line number Diff line number Diff line change
Expand Up @@ -297,7 +297,8 @@ def capability_configuration_editor(
"supported_scopes": ["goal"],
"writable_scopes": ["goal"],
"fields": [
_field("enabled", "Enabled", "boolean"),
_field("mode", "Exploration mode", "select", options=("off", "evidence", "planning"),
description="Evidence only, or evidence with planning. Spawn authority is separate."),
_field(
"profile",
"Planner profile",
Expand Down Expand Up @@ -574,7 +575,7 @@ def _merge_goal_feature(
entry[field] = deepcopy(feature[field])
entry["configuration_editor"] = capability_configuration_editor(
capability_id,
explore_harness_profiles=explore_harness_profiles,
explore_harness_profiles=feature.get("profiles") or explore_harness_profiles,
)
if "machine" in entry["available_scopes"]:
entry["effective_value_policy"] = "goal_override_over_live_machine_default"
Expand Down
96 changes: 48 additions & 48 deletions loopx/capabilities/explore/README.md
Original file line number Diff line number Diff line change
@@ -1,19 +1,19 @@
# Exploration Result Layer
# Explore Harness

Status: supported optional capability; default-off harness execution contract.

## At a Glance

LoopX Explore is a supported, default-off optional capability for
Explore Harness is a supported, default-off optional capability for
long-running exploration goals (software research, security attack-surface
mapping, domain studies). It turns "go look around" into a bounded,
observable, gated process with three pillars:

1. **Explore Graph** - an append-only, public-safe evidence topology
1. **Evidence graph** (Explore Graph) - an append-only, public-safe evidence topology
(nodes / edges / findings) plus bounded projections, Mermaid export, and
canonical/executive presentation. It answers: what has been explored,
where the loop is blocked and why, and what was found.
2. **Explore Harness** - deny-by-default, read-only branch planners
2. **Optional planning** - deny-by-default, read-only branch planners
(`todo-branch-plan`, `worker-branch-plan`) that rank and bundle next
steps (DSpark-style confidence/prefix/load, `adaptive-resilient` and
`moe-router` profiles, resource-aware portfolio), without claiming,
Expand All @@ -36,11 +36,11 @@ executes it through the normal LoopX lifecycle.

## Quick Start

Enable the gates, record evidence, project, and plan:
Choose a mode, record evidence, project, and plan:

```bash
loopx configure-goal --goal-id <id> --explore-graph-enabled \
--explore-harness-enabled --explore-harness-profile adaptive-resilient --execute
loopx configure-goal --goal-id <id> --explore-mode planning \
--explore-harness-profile adaptive-resilient --execute

loopx explore node --goal-id <id> --title "Attack surface A" --status exploring
loopx explore edge --goal-id <id> --from A --to B --type leads_to
Expand All @@ -51,8 +51,7 @@ loopx explore graph --goal-id <id> --graph-format mermaid --out explore.mmd
loopx explore worker-branch-plan --goal-id <id> --harness-profile adaptive-resilient --worker-width 3
```

Both gates are separate and default-off (see "Independent Per-Goal Opt-In
Gates"). When closed, evidence-backed surfaces have an explicit reason to be
Explore Harness is off by default (see "Explore Harness Modes"). When closed, evidence-backed surfaces have an explicit reason to be
tested together, the next `quota should-run` / turn packet projects a
composition gap and can derive a joint-experiment successor todo (see
"Composition Frontier"). The detailed contract follows.
Expand Down Expand Up @@ -357,51 +356,52 @@ deny-by-default disabled packet carries the `boundary` block plus the opt-in
operator to decide which workers to start, but it cannot launch workers or
mutate the control plane on its own.

### Independent Per-Goal Opt-In Gates

Explore Graph and Explore Harness are separate optional capabilities. Enabling
one never enables the other:

- `explore_graph.enabled` controls durable graph projection and any already
configured presentation sink. After each successful material
`refresh-state` transaction, LoopX folds the canonical Explore evidence and
runs the configured sink. Semantic digests make an unchanged refresh a
zero-write operation. A configured row sink is complete only after a
row/result-id readback verifies the projection. A failed sync or readback
does not advance its digest, so the next material refresh retries it.
Visual sinks also preflight their deterministic delivery marker: an existing
marker reconciles the prior write without publishing again, while a bounded
readback timeout stops further calls in that stage batch and leaves a
retryable receipt instead of blindly repeating remote writes.
- `spawn_policy.explore_harness.enabled` controls only the read-only branch
planners described below. It does not create, update, or publish a graph.

Both gates are absent/false by default. A common operating mode is Graph on
and Harness off: keep an operator-facing topology current without changing
how work is planned.
### Explore Harness Modes

```yaml
# inside the registered goal entry
explore_graph:
enabled: true

spawn_policy:
explore_harness:
enabled: false
```
Explore Harness is one optional capability with an evidence graph and optional
read-only planning. Goal settings offer one editor with three modes:

Configure the gates independently instead of editing the registry:
| Mode | Evidence graph | Branch planning |
| --- | --- | --- |
| `off` (default) | Inactive | Inactive |
| `evidence` | Active | Inactive |
| `planning` | Active | Active |

```bash
loopx configure-goal --goal-id <id> \
--explore-graph-enabled \
--no-explore-harness-enabled \
--execute
loopx configure-goal --goal-id <id> --explore-mode planning # preview
loopx configure-goal --goal-id <id> --explore-mode planning --execute
loopx configure-goal --goal-id <id> # read back
loopx explore turn-context --goal-id <id> --agent-id <registered-agent>
loopx configure-goal --goal-id <id> --explore-mode evidence --execute
loopx configure-goal --goal-id <id> --explore-mode off --execute
```

Use `--no-explore-graph-enabled` to stop automatic graph work. Disabling the
gate preserves existing evidence and display state; it only prevents future
automatic projection and sink writes.
Existing `explore_graph.enabled` and `spawn_policy.explore_harness` storage,
record ids and CLI aliases remain supported. A legacy Graph-only Goal maps to
`evidence`; a legacy Harness-enabled Goal maps to `planning`, now including the
graph. Enabling planning writes both internal flags. Disabling planning through
the legacy Harness flag retains the evidence layer. Disabling the Graph while
planning remains enabled fails with an actionable mode command. Do not combine
`--explore-mode` with the legacy enable flags in one request.

Both enabled modes register an `explore.turn_context` turn-start hook. Its
`required_reads` entry is a **before-work read obligation** in the normal packet.
The command returns at most three recent nodes/findings and, in planning mode,
three suggested Todo branches, plus structured commands for detail or evidence
recording. The read folds existing history but bounds the returned context;
it does not claim to reduce history IO. The agent chooses evidence-backed work;
planner suggestions do not require branching on every turn or recording empty
ceremonial nodes. Use the detail command when the short view is insufficient.

Mode selection does not grant spawn, claim, lease, execution, quota or external
publication authority. `spawn_allowed=false` retains analysis-only planning.
Feature-off adds no Explore hook or evidence reads. Turning off preserves all
recorded evidence and existing display state.

The evidence layer reuses the material-refresh projection and any separately
configured, authorized sink. Semantic digests avoid unchanged writes. Row sinks
must read back result ids; failed sync/readback leaves the digest retryable.
Visual sinks reconcile deterministic delivery markers before retrying writes.

When a single run may update local state but is not authorized to write any
configured external sink, keep the graph enabled and pass
Expand Down
12 changes: 10 additions & 2 deletions loopx/capabilities/explore/activation.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,14 +82,14 @@ def sync_explore_graph_after_material_refresh(
) -> dict[str, Any]:
"""Flush an enabled graph after the goal-level refresh transaction.

Explore Graph activation is independent from Explore Harness planning. A
Explore Harness planning includes its evidence graph. A
configured graph reuses the existing idempotent projection/sink adapter;
disabled or absent policy performs no graph reads or writes.
"""

goal = load_goal_from_registry(registry_path, goal_id)
policy = compact_explore_graph_policy(
goal.get("explore_graph") if isinstance(goal, dict) else None
(goal or {}).get("explore_graph"), ((goal or {}).get("spawn_policy") or {}).get("explore_harness")
)
base = {
"ok": True,
Expand Down Expand Up @@ -125,6 +125,14 @@ def sync_explore_graph_after_material_refresh(
),
}

# Disabled activation preserves the caller's authorization observation without
# reading or publishing anything. An enabled legacy Harness does not grant a
# sink permission merely by supplying its local evidence graph.
external_sink_delivery_authorized = external_sink_delivery_authorized and (
((goal or {}).get("explore_graph") or {}).get("enabled") is True
)
base["external_sink_delivery_authorized"] = external_sink_delivery_authorized

if syncer is None:
status = "projection_sink_provider_unavailable"
return {
Expand Down
12 changes: 11 additions & 1 deletion loopx/capabilities/explore/catalog_entry.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@
"site_root": "capabilities/explore",
"canonical": "README.md",
},
"title": "Explore evidence topology",
"title": "Explore Harness",
"status": "active-preview",
"real_world_anchor": (
"goal-scoped questions, hypotheses, experiments, findings, and result projections"
Expand All @@ -24,6 +24,16 @@
),
"entry_command": "loopx explore summary --goal-id <goal-id> --format json",
"commands": [
{
"command": "loopx configure-goal --goal-id <goal-id> --explore-mode planning",
"purpose": "Preview evidence-and-planning mode; use evidence for graph only or off to disable.",
"write_boundary": "preview only; --execute applies configuration without granting spawn authority",
},
{
"command": "loopx explore turn-context --goal-id <goal-id> --agent-id <agent-id> --format json",
"purpose": "Read the bounded context requested by the enabled turn-start hook.",
"write_boundary": "read-only; no evidence, claim, lease, worker or quota writes",
},
{
"command": "loopx explore finding --goal-id <goal-id> --title <finding> --evidence-ref <ref> --format json",
"purpose": "Append one public-safe, attributable finding to the canonical result log.",
Expand Down
Loading
Loading