Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
39 commits
Select commit Hold shift + click to select a range
9f5511a
refactor: expose prompt composition seams
cedricvidal Sep 9, 2026
2bceee9
feat: add static prompt evaluation suites
cedricvidal Sep 9, 2026
5123706
docs: document prompt evaluation maintenance
cedricvidal Sep 9, 2026
a3b86de
test: use synthetic redaction fixture
cedricvidal Sep 9, 2026
768b627
fix: isolate concurrent judge evaluation workspaces
cedricvidal Sep 9, 2026
da10457
fix: delete temporary red-team agents
cedricvidal Sep 9, 2026
7d9b913
fix: report actual red-team target mode
cedricvidal Sep 9, 2026
bdf91dc
feat: guide developers when uv is missing
cedricvidal Sep 9, 2026
0751e51
feat: generate markdown evaluation reports
cedricvidal Sep 9, 2026
65b06e7
docs: explain evaluation policy gating in reports
cedricvidal Sep 9, 2026
20ad442
style: remove trailing blank lines
cedricvidal Sep 9, 2026
b7d5b6a
fix(evals): cap aggregate pass-rate policy at eighty percent
cedricvidal Sep 9, 2026
a182160
fix(evals): parse native grader outputs and preserve task context
cedricvidal Sep 9, 2026
4ace354
feat(evals): add auditable gate decisions and immutable selective replay
cedricvidal Sep 9, 2026
170260d
fix(evals): standardize decision summary and read-only legacy exports
cedricvidal Sep 9, 2026
e43a420
fix(evals): keep eighty-percent policy solely in rubric configuration
cedricvidal Sep 9, 2026
4a1061a
fix(evals): handle pnpm separator in report CLI
cedricvidal Sep 10, 2026
c0fbd7a
fix(evals): resolve incomplete sample votes and reject source overlap…
cedricvidal Sep 10, 2026
1ad33a6
fix(evals): normalize SDK tool-call fields and retain evaluator diagn…
cedricvidal Sep 10, 2026
8e2813b
fix(evals): explain unresolved report gates and format scores
cedricvidal Sep 10, 2026
e4b0195
docs(evals): explain case context in review clients
cedricvidal Sep 10, 2026
7485982
Merge upstream main and preserve prompt evaluation builders
cedricvidal Sep 10, 2026
9aba7f0
fix: restore clean evaluation builds and scope hash scan exemption
cedricvidal Sep 10, 2026
719bb66
Merge current upstream main for replacement evaluation PR
cedricvidal Sep 29, 2026
5cafe1b
fix(ci): restore upstream repository gates safely
cedricvidal Sep 29, 2026
d28f298
fix(ci): limit repository gate correction to four replacements
cedricvidal Sep 29, 2026
0c5015b
fix(ci): repair integration and image workflow execution
cedricvidal Sep 29, 2026
74f20b4
fix(ci): remove obsolete VS Code image tag branches
cedricvidal Sep 29, 2026
0a577d7
ci: separate public validation from internal automation
cedricvidal Sep 29, 2026
d326467
ci: make Windows builds and CLI releases self-contained
cedricvidal Sep 29, 2026
58fe058
Merge CI foundation for dependent prompt evaluation PR
cedricvidal Sep 29, 2026
8bffe1f
fix(docs): route website CLI installs to public releases
cedricvidal Sep 29, 2026
a61416d
fix(cli): isolate standalone bundle and require an explicit API URL
cedricvidal Sep 29, 2026
d60a71e
fix(cli): isolate standalone bundle and require an explicit API URL
cedricvidal Sep 29, 2026
c3689bc
Merge upstream main into CI and standalone release fixes
cedricvidal Sep 29, 2026
f266d76
Merge resolved CI foundation and upstream ACP test utilities
cedricvidal Sep 29, 2026
216d21b
Merge published CI foundation into evaluation branch
cedricvidal Sep 30, 2026
50a9590
refactor(evals): extract unused red-team execution from quality suite
cedricvidal Sep 30, 2026
781b4c5
docs(evals): document quality-only scope after red-team extraction
cedricvidal Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -14,9 +14,12 @@ yarn.lock
.vscode/settings.json
.azure/
.venv/
__pycache__/
*.pyc
downloads/
ctrf/
coverage/
evaluations/static-prompts/results/
.auth/
test-videos/
test-snapshots/
Expand Down
10 changes: 10 additions & 0 deletions .gitleaks.toml
Original file line number Diff line number Diff line change
Expand Up @@ -63,3 +63,13 @@ paths = [
'''.*\.test\.ts$''',
'''.*\.integration\.test\.ts$''',
]

# The harvested OpenAPI fingerprint is a content hash, not an API credential.
# Match only this verified digest in the dataset manifest, including history.
[[allowlists]]
description = "Static prompt dataset OpenAPI SHA-256 fingerprint"
targetRules = ["generic-api-key"]
condition = "AND"
regexTarget = "secret"
paths = ['''^evaluations/static-prompts/datasets/manifest\.json$''']
regexes = ['''^4f6776ac4602e279c8723acd9a6cb22fa504c95538ab9491e0b5acab267c6a6b$''']
22 changes: 22 additions & 0 deletions AGENTS.md
Original file line number Diff line number Diff line change
Expand Up @@ -195,6 +195,27 @@ pnpm test:integration # Integration tests (requires .env + Docker)

> **Portal Storybook stories run under Vitest**: `apps/portal/src/components/ui/stories.play.test.tsx` composes the `ui/*` stories and executes their `play` (interaction) functions inside the regular Vitest suite (no `@storybook/addon-vitest` required). It binds a Testing Library `canvas` to the rendered container, so story `play` functions must keep depending only on `canvas` plus values imported directly from `storybook/test` (`userEvent`, `screen`, `expect`). When you add a new `ui/*` story with a `play` function, register its module in that harness so it's covered.

### Static prompt evaluations

Read [docs/architecture/prompt-evaluations.md](docs/architecture/prompt-evaluations.md)
before changing any AI-facing instruction surface.

- Whenever hardcoded system/user prompt text, prompt-building logic, output
instructions, or AI-facing tool descriptions are added or modified, update
the corresponding production target adapter, curated cases, deterministic
checks, composition contract, and/or rubric.
- Register every new runtime static prompt family in the documented inventory,
committed evaluation manifest, and JSONL adapter registry.
- Run the smallest relevant explicit prompt-evaluation command before treating
a prompt change as complete. These suites are developer-invoked and are not
part of normal `pnpm test` or CI.
- A change to user-authored/configurable prompt content does not by itself
create a static prompt family.
- Whenever a user-controlled text field that reaches an AI is added or its
insertion point, trusted wrapper, role, tools, or security boundary changes,
preserve its benign composition contract. Cloud red-team profiles and
execution are maintained in a separate follow-up contribution.

## Contributing (Pull Requests)

This repository is commonly worked on from a **fork**. When opening a pull
Expand Down Expand Up @@ -232,6 +253,7 @@ not open the PR against the fork unless the user explicitly asks you to.
|----------|-------------|
| [docs/architecture/overview.md](docs/architecture/overview.md) | System architecture, component interactions, data flow |
| [docs/architecture/app-design.md](docs/architecture/app-design.md) | Data models, API design, package dependency graph |
| [docs/architecture/prompt-evaluations.md](docs/architecture/prompt-evaluations.md) | Static prompt quality, datasets, commands, and maintenance rules |
| [docs/architecture/data-organization-projects.md](docs/architecture/data-organization-projects.md) | Projects (a single container) to isolate/group data within a cluster; composes with data-tags and auth-rbac |
| [docs/architecture/auth-rbac.md](docs/architecture/auth-rbac.md) | Explicit-login IdP auth, Redis user-access cache, Portal handshake; deferred RBAC/internal-token roadmap |
| [docs/architecture/vscode-web-worker.md](docs/architecture/vscode-web-worker.md) | XState chat machine, GitHub auth flow, ARIA snapshots |
Expand Down
97 changes: 97 additions & 0 deletions ENV_VARIABLES.md
Original file line number Diff line number Diff line change
Expand Up @@ -183,6 +183,103 @@ inference errors at runtime. Learned compatibility is cached in each API
process by endpoint and deployment name. It is relearned after a process
restart or when Azure rejects a previously accepted request shape.

## Prompt Evaluation Configuration

These variables are consumed only by the developer-run tooling in
`evaluations/static-prompts`. They do not enable the suite in normal tests or
CI. See [Prompt Evaluations](docs/architecture/prompt-evaluations.md) for the
quality commands and artifact policy. Cloud red teaming is a separate follow-up.

Run `az login` before cloud evaluation. The Python tooling uses Azure Identity;
the quality graders also accept an explicit Azure OpenAI API key when required.
Do not commit credentials, endpoints, tenant/subscription IDs, or a populated
environment file.

### PROMPT_EVAL_MODEL
**Default:** resolved inference credential model, then `LLM_MODEL`, then `gpt-4.1`
**Type:** string
**Used by:** Static prompt TypeScript generation adapters

Model/deployment used to generate production-path responses for the quality
track. This setting is independent from the Azure-assisted evaluator deployment
so generator and grader identities are explicit in result metadata.

Generation uses the existing inference credential chain described in
[LLM Configuration](#llm-configuration-portal-ai-features), including
`AZURE_AI_INFERENCE_ENDPOINT` and `AZURE_AI_INFERENCE_API_KEY` when configured.

### SCOPE_EVAL_AZURE_OPENAI_ENDPOINT
**Required:** AI-assisted quality evaluation
**Fallback:** `AZURE_OPENAI_ENDPOINT`
**Type:** URL string
**Used by:** Azure AI Evaluation SDK quality graders

Azure OpenAI resource endpoint used by built-in and configurable quality
graders. This is the model resource endpoint, not the Foundry project endpoint.

### SCOPE_EVAL_AZURE_OPENAI_DEPLOYMENT
**Required:** AI-assisted quality evaluation
**Fallback:** `AZURE_OPENAI_DEPLOYMENT`
**Type:** string
**Used by:** Azure AI Evaluation SDK quality graders

Deployment used to grade generated quality rows. Keep it distinct from
`PROMPT_EVAL_MODEL` when evaluating one model's output with another.

### SCOPE_EVAL_AZURE_OPENAI_API_KEY
**Default:** unset
**Fallback:** `AZURE_OPENAI_API_KEY`
**Type:** string
**Used by:** Azure AI Evaluation SDK quality graders

Optional API key for the quality grader endpoint. When neither key variable is
set, the runner uses `DefaultAzureCredential`; `az login` is the recommended
local authentication path.

### SCOPE_EVAL_AZURE_OPENAI_API_VERSION
**Default:** SDK default
**Fallback:** `AZURE_OPENAI_API_VERSION`
**Type:** string
**Used by:** Azure AI Evaluation SDK quality graders

Optional Azure OpenAI API version override.

### SCOPE_EVAL_AZURE_AI_PROJECT_ENDPOINT
**Default:** unset
**Fallback:** `AZURE_AI_PROJECT_ENDPOINT`
**Type:** URL string
**Used by:** Azure AI Evaluation SDK quality runs

Optional Foundry project endpoint for publishing a quality evaluation portal
view. Local JSON/JSONL artifacts remain the source of truth.

### AZURE_AI_PROJECT_ENDPOINT
**Default:** unset
**Type:** URL string
**Used by:** Azure AI Evaluation SDK quality runs

Fallback for `SCOPE_EVAL_AZURE_AI_PROJECT_ENDPOINT` in quality runs.

Use the project endpoint for the intended Foundry project, not an inference
`/models` endpoint.

### Quality runner diagnostic overrides

`SCOPE_EVAL_GENERATOR_COMMAND` overrides the production JSONL generator command
using `{dataset}`, `{output}`, `{samples}`, and `{smoke}` placeholders.
`SCOPE_EVAL_GENERATOR_TIMEOUT_SECONDS` changes its 1,800-second timeout.
`SCOPE_EVAL_GENERATED_ROWS` copies an existing JSONL file instead of invoking
generation. These are test/debugging controls; normal baseline runs should use
the declared package generator.

### Harvester bearer token

The dataset harvester accepts the **name** of an environment variable through
its token option rather than reading a fixed secret name. This supports
protected integration environments without establishing a repository-wide
credential variable. Never pass the token value on the command line and never
write it to curated rows, provenance, logs, or results.

## Prompt Storage Configuration

### PROMPT_INLINE_MAX_BYTES
Expand Down
6 changes: 6 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -242,6 +242,7 @@ Portal and a Rust AI gateway.
| [`apps/workers/`](./apps/workers/) | Coding-agent, post-processing, and report workers |
| [`apps/gateway/`](./apps/gateway/) and [`apps/token-manager/`](./apps/token-manager/) | AI traffic capture and credential management |
| [`packages/`](./packages/) | Shared types, storage clients, migrations, and supporting libraries |
| [`evaluations/`](./evaluations/) | Developer-run static prompt quality tooling |
| [`config/`](./config/) and [`docs/`](./docs/) | Evaluation examples and documentation |

Useful commands from the repository root:
Expand All @@ -254,6 +255,10 @@ pnpm storybook # Portal component catalog
pnpm test:integration # Integration tests; requires .env and backing services
```

Evaluate Scope's own AI prompts with `pnpm eval:prompts -- --mode quality`
(the default mode). See the [prompt evaluation guide](./docs/architecture/prompt-evaluations.md)
for setup, model credentials, and offline validation.

For service-by-service development, Rust commands, migrations, and code
conventions, read [CONTRIBUTING.md](./CONTRIBUTING.md).

Expand All @@ -265,6 +270,7 @@ conventions, read [CONTRIBUTING.md](./CONTRIBUTING.md).
| Domain models and API design | [Application design](./docs/architecture/app-design.md) |
| Project organization | [Projects](./docs/architecture/data-organization-projects.md) |
| Evaluation and criteria DAGs | [Criteria provider](./docs/architecture/criteria-provider.md) |
| Static prompt quality | [Prompt evaluations](./docs/architecture/prompt-evaluations.md) |
| Agent context | [Skills](./docs/architecture/skills.md) and [codebases](./docs/architecture/codebases.md) |
| Scheduling and recovery | [Queue scheduler](./docs/architecture/queue-scheduler.md) |
| Configuration and authentication | [Environment variables](./ENV_VARIABLES.md) |
Expand Down
157 changes: 123 additions & 34 deletions apps/api/src/llm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -166,6 +166,16 @@ export interface GenerateResult {
suggestedChildren: string[];
}

export interface CriteriaPromptRequest {
messages: [
{ role: "system"; content: string },
{ role: "user"; content: string },
];
model: string;
temperature: number;
max_tokens: number;
}

export function isLlmAvailable(): boolean {
return inferenceAvailable();
}
Expand All @@ -192,22 +202,106 @@ function parseJson(content: string): any {
}
}

export function buildCriteriaAuthoringRequest(
behavior: string,
gates: GateId[] | undefined,
model: string,
): CriteriaPromptRequest {
return {
messages: [
{ role: "system", content: SYSTEM_PROMPT_AUTHOR },
{
role: "user",
content: `NEW CRITERION TO CREATE:\n${behavior}${authorGateHint(gates)}`,
},
],
model,
temperature: 0.3,
max_tokens: 512,
};
}

export function parseCriteriaAuthoringResponse(
content: string,
): { prompt: string; suggestedId: string } {
const parsed: unknown = parseJson(content);
if (
!parsed ||
typeof parsed !== "object" ||
!("prompt" in parsed) ||
!parsed.prompt
) {
throw new Error("Missing required field: prompt");
}
const value = parsed as { prompt: unknown; suggestedId?: unknown };
return {
prompt: String(value.prompt).trim(),
suggestedId: sanitizeId(value.suggestedId),
};
}

export function buildCriteriaDependencySuggestionRequest(
direction: SuggestDirection,
behavior: string,
pool: ExistingCriterion[],
model: string,
): CriteriaPromptRequest {
return {
messages: [
{ role: "system", content: suggestSystemPrompt(direction) },
{
role: "user",
content: buildSuggestMessage(direction, behavior, pool),
},
],
model,
temperature: 0.3,
max_tokens: 512,
};
}

export function parseCriteriaDependencySuggestionResponse(
content: string,
pool: ExistingCriterion[],
): string[] {
const parsed: unknown = parseJson(content);
const suggestions =
parsed && typeof parsed === "object" && "suggestions" in parsed
? (parsed as { suggestions?: unknown }).suggestions
: undefined;
if (!Array.isArray(suggestions)) return [];
const poolIds = new Set(pool.map((criterion) => criterion.id));
return suggestions.filter(
(id): id is string => typeof id === "string" && poolIds.has(id),
);
}

export function selectCriteriaDependencyPool(
direction: SuggestDirection,
existingCriteria: ExistingCriterion[],
newGates?: GateId[],
): ExistingCriterion[] {
if (!newGates) return existingCriteria;
return direction === "parents"
? existingCriteria.filter((criterion) =>
gatesSatisfyInvariant(criterion.gates, newGates),
)
: existingCriteria.filter((criterion) =>
gatesSatisfyInvariant(newGates, criterion.gates),
);
}

async function chat(
llm: ChatClient,
endpoint: string,
model: string,
systemPrompt: string,
userMessage: string,
request: CriteriaPromptRequest,
): Promise<string> {
const response = await postAdaptiveChatCompletion({
endpoint,
model,
messages: [
{ role: "system", content: systemPrompt },
{ role: "user", content: userMessage },
],
temperature: 0.3,
maxTokens: 512,
model: request.model,
messages: request.messages,
temperature: request.temperature,
maxTokens: request.max_tokens,
send: (body) => llm.path("/chat/completions").post({ body }),
});

Expand Down Expand Up @@ -237,18 +331,9 @@ async function author(
const content = await chat(
llm,
endpoint,
model,
SYSTEM_PROMPT_AUTHOR,
`NEW CRITERION TO CREATE:\n${behavior}${authorGateHint(gates)}`,
buildCriteriaAuthoringRequest(behavior, gates, model),
);
const parsed = parseJson(content);
if (!parsed.prompt) {
throw new Error("Missing required field: prompt");
}
return {
prompt: String(parsed.prompt).trim(),
suggestedId: sanitizeId(parsed.suggestedId),
};
return parseCriteriaAuthoringResponse(content);
}

function buildSuggestMessage(
Expand Down Expand Up @@ -285,18 +370,18 @@ async function suggestDeps(
pool: ExistingCriterion[],
): Promise<string[]> {
if (pool.length === 0) return [];
const poolIds = new Set(pool.map((c) => c.id));
try {
const content = await chat(
llm,
endpoint,
model,
suggestSystemPrompt(direction),
buildSuggestMessage(direction, behavior, pool),
buildCriteriaDependencySuggestionRequest(
direction,
behavior,
pool,
model,
),
);
const parsed = parseJson(content);
const suggestions = Array.isArray(parsed.suggestions) ? parsed.suggestions : [];
return suggestions.filter((sid: unknown): sid is string => typeof sid === "string" && poolIds.has(sid));
return parseCriteriaDependencySuggestionResponse(content, pool);
} catch (err) {
console.warn(`[generate-prompt] ${direction} suggestion call failed, degrading to []:`, err);
return [];
Expand Down Expand Up @@ -332,12 +417,16 @@ export async function generateCriteriaPrompt(
// Priority: explicit arg > key-specific (from Foundry blob) > env > default.
const modelName = model || foundryModel || process.env.LLM_MODEL || "gpt-4.1";

const parentPool = newGates
? existingCriteria.filter((c) => gatesSatisfyInvariant(c.gates, newGates))
: existingCriteria;
const childPool = newGates
? existingCriteria.filter((c) => gatesSatisfyInvariant(newGates, c.gates))
: existingCriteria;
const parentPool = selectCriteriaDependencyPool(
"parents",
existingCriteria,
newGates,
);
const childPool = selectCriteriaDependencyPool(
"children",
existingCriteria,
newGates,
);

const [authored, suggestedParents, suggestedChildrenRaw] = await Promise.all([
author(llm, endpoint, modelName, behavior, newGates),
Expand Down
Loading
Loading