From 822bafe6df182fc110701ca807abe200c1d17d9d Mon Sep 17 00:00:00 2001 From: Dam Minh Tuan Anh Date: Thu, 24 Sep 2026 16:48:07 +0700 Subject: [PATCH 1/7] feat(report): skill that builds an HTML requirement report from JSONL Data lives in JSONL, rendering in a dependency-free Python script that writes one self-contained HTML file. Deterministic: no model involved in generation, and an existing flat reqs.txt imports directly. - config-driven columns, statuses, rules and labels via report.json - status and evidence progress bars, per-group breakdown, summary cards - per-column filters (dropdown for categorical columns) plus free text - requirement table scrolls in its own box with a pinned header - evidence pass: agents write three JSONL files, a second agent reviews the images, rebuild attaches them; the skill asks before starting it - honest-status rules: PASS without a test demotes, E2E FAIL deviates - fails loudly on duplicate ids, undeclared statuses, dangling references --- .agents/skills/report/SKILL.md | 71 +++ .agents/skills/report/assets/report.json | 132 ++++ .agents/skills/report/references/evidence.md | 47 ++ .agents/skills/report/references/schema.md | 107 ++++ .agents/skills/report/scripts/build_report.py | 573 ++++++++++++++++++ .../report/scripts/test_build_report.py | 283 +++++++++ 6 files changed, 1213 insertions(+) create mode 100644 .agents/skills/report/SKILL.md create mode 100644 .agents/skills/report/assets/report.json create mode 100644 .agents/skills/report/references/evidence.md create mode 100644 .agents/skills/report/references/schema.md create mode 100644 .agents/skills/report/scripts/build_report.py create mode 100644 .agents/skills/report/scripts/test_build_report.py diff --git a/.agents/skills/report/SKILL.md b/.agents/skills/report/SKILL.md new file mode 100644 index 0000000..ecfaece --- /dev/null +++ b/.agents/skills/report/SKILL.md @@ -0,0 +1,71 @@ +--- +name: report +description: Use when the user asks for a report, an audit report, a requirement or coverage status report, a traceability matrix or progress dashboard, wants to know how much of a spec is implemented and tested, wants an existing report rebuilt or updated with new results, or asks to attach screenshot evidence to a report. +allowed-tools: Read, Write, Edit, Grep, Glob, Bash, Task, Agent +--- + +# Report + +Data lives in JSONL, rendering lives in a script. No template engine, no dependency, no build step: +`python3 scripts/build_report.py ` writes one self-contained HTML file that opens in a browser. + +Generation is deterministic Python — same data in, same report out, no model involved. What costs +effort is the data: requirements extracted from a spec and checked against code and tests. If that +list already exists as a flat `reqs.txt`, import it and the whole report needs no model at all. + +## Three steps + +### 1. Build from the data you already have + +```bash +mkdir -p /.testcases/report//data +cp assets/report.json /.testcases/report//report.json +# then either import an existing flat list … +python3 scripts/build_report.py /.testcases/report/ --from-reqs /reqs.txt +# … or write data/reqs-*.jsonl yourself and build +python3 scripts/build_report.py /.testcases/report/ +``` + +Read `references/schema.md` before writing the first JSONL line. Adjust `columns` and `status` in +`report.json` to fit the work. The build fails loudly on bad data (duplicate ids, undeclared status, +references to unknown ids) — fix the data, not the script. + +Hand the user the `REPORT.html` path **here**. The report is already usable. + +### 2. Ask + +> Do you want screenshot evidence? + +No → stop. Yes → step 3. Do not start capturing unasked: the evidence pass costs many times more +than step 1, and plenty of reports never need images. + +### 3. Evidence pass + +`references/evidence.md` holds the file contract, how to split the work across agents, and the +mandatory independent review. In short: split requirements by group → each agent captures its group +and writes three JSONL files into `evidence/` → **a different** agent re-checks every image and +writes `review-*.jsonl` → rebuild. + +Rebuilding is idempotent: images attach themselves and the Evidence bar moves. + +## Updating an existing report + +- Human verdicts: append to `data/overrides.jsonl`, never edit `reqs-*.jsonl`. That keeps the line + between what was derived and what a person concluded. +- Fresh E2E results: replace `data/e2e-results.jsonl` and rebuild. +- Progress since last time: fill `baseline` in `report.json` with the previous build's numbers to get + the "Before → Now" table. + +## Honest status + +Two rules in `report.json` stop the report flattering itself: `demote_pass_without_test` downgrades +any requirement marked PASS with no test behind it, and `e2e_fail_status` forces anything with a +failing E2E run to deviation. Do not remove them to make the table look better. + +Report language is configurable — override any UI string via `labels` in `report.json`. + +## Self-check + +```bash +python3 scripts/test_build_report.py +``` diff --git a/.agents/skills/report/assets/report.json b/.agents/skills/report/assets/report.json new file mode 100644 index 0000000..6752900 --- /dev/null +++ b/.agents/skills/report/assets/report.json @@ -0,0 +1,132 @@ +{ + "title": " — requirement audit", + "header": "Scope: … Spec: path/to/spec … Method: one Req per behaviour / condition / validation / edge case, each quoting file:line and checked against code and tests.", + "group_by": "id_prefix", + "findings": [], + "columns": [ + { + "key": "id", + "label": "Req ID", + "nowrap": true, + "width": "112px" + }, + { + "key": "area", + "label": "Area", + "width": "76px" + }, + { + "key": "req", + "label": "Requirement", + "width": "16%" + }, + { + "key": "src", + "label": "Spec source", + "width": "10%" + }, + { + "key": "acc", + "label": "Acceptance", + "width": "70px" + }, + { + "key": "status", + "label": "Status", + "render": "status", + "nowrap": true, + "width": "136px" + }, + { + "key": "evidence", + "label": "Evidence", + "render": "evidence", + "width": "13%" + }, + { + "key": "impl", + "label": "Implementation", + "width": "10%" + }, + { + "key": "unit", + "label": "Unit/IT test", + "width": "8%" + }, + { + "key": "e2e", + "label": "E2E", + "render": "e2e", + "width": "7%" + }, + { + "key": "qa", + "label": "Issue / Q&A", + "width": "8%" + }, + { + "key": "note", + "label": "Note", + "width": "10%" + } + ], + "status": { + "PASS": { + "icon": "✅", + "label": "Implemented and tested", + "color": "#0ca30c" + }, + "NO_TEST": { + "icon": "🟡", + "label": "Implemented, no test", + "color": "#fab219" + }, + "MISSING": { + "icon": "⬜", + "label": "Not implemented", + "color": "#898781" + }, + "DEVIATION": { + "icon": "🔴", + "label": "Deviates from spec", + "color": "#d03b3b" + }, + "CONFIRM": { + "icon": "❓", + "label": "Unclear — needs confirm", + "color": "#ec835a" + }, + "OUT_OF_SCOPE": { + "icon": "⚪", + "label": "Out of scope", + "color": "#c3c2b7" + } + }, + "rules": { + "done": [ + "PASS" + ], + "demote_pass_without_test": { + "from": "PASS", + "to": "NO_TEST", + "when_empty": [ + "unit", + "e2e" + ] + }, + "e2e_fail_status": "DEVIATION", + "attention": [ + "CONFIRM", + "DEVIATION", + "MISSING" + ], + "attention_columns": [ + "qa", + "note" + ] + }, + "emit": { + "goalrun_reqs": false + }, + "labels": {} +} \ No newline at end of file diff --git a/.agents/skills/report/references/evidence.md b/.agents/skills/report/references/evidence.md new file mode 100644 index 0000000..df5fb18 --- /dev/null +++ b/.agents/skills/report/references/evidence.md @@ -0,0 +1,47 @@ +# Evidence pass + +Only run this after the user answers **yes**. The report already exists by then — this pass only +*adds* images. It never touches `data/`, and rebuilding is safe at any point. + +## File contract + +Agents write into `evidence/` only, each agent under its own `` so no two agents write the +same file: + +``` +evidence/.png +evidence/.manifest.jsonl {"file":"EV-A01-login-locked.png","caption":"After 5 failures: 15-minute lock banner","check":"PASS"} +evidence/.reqs.jsonl {"id":"AUTH-001","verdict":"SHOWN","evidence":["EV-A01-login-locked.png"],"note":""} +evidence/review-.jsonl {"id":"AUTH-001","finding":"caption-mismatch","detail":"screenshot shows the signup screen"} +``` + +- `check`: `PASS` when the image really shows what the caption claims; anything else renders as ✖. +- `verdict`: `SHOWN` (proves the requirement) · `SHOWN-PARTIAL` (covers part of it) · + `NOT-SHOWN` (captured, but does not prove it). +- Image names: `EV--.png` — readable without opening the file. +- An id absent from `data/` fails the build. Only capture for requirements that exist. + +## Splitting the work + +Split by requirement group (`group_by`), one agent per group, run in parallel via +`superpowers:dispatching-parallel-agents`. Each agent receives: its requirements (id, requirement +text, acceptance criteria), the `evidence/` path, its `` name, and how to reach the app. + +The capture tool is **not fixed**: `playwright-cdp` is the best option when the app is a web app and +the user's machine can run it, because it can reproduce the exact state. Otherwise manual +screenshots, CI artifacts, or anything else works. The contract is the three files, not the tool. + +## Independent review — do not skip + +An agent that captures its own screenshots and then marks its own `check: PASS` is the single most +likely source of fake evidence: wrong screen, right screen in the wrong state, or a stale image taken +before the fix. Once capture finishes, **a different agent** opens every image, compares it against +`req` and `caption`, and records every mismatch in `review-.jsonl`. The report shows those as +red warnings inside the evidence cell and counts them on the progress bar. + +The reviewing agent must not be the agent that captured that group. + +## Finish + +Re-run `python3 scripts/build_report.py ` and the Evidence bar moves. Running it mid-flight is +fine — whatever has landed shows up. diff --git a/.agents/skills/report/references/schema.md b/.agents/skills/report/references/schema.md new file mode 100644 index 0000000..f1003ec --- /dev/null +++ b/.agents/skills/report/references/schema.md @@ -0,0 +1,107 @@ +# Schema + +Every path is relative to the directory holding `report.json`. +Build: `python3 scripts/build_report.py `. + +``` +/report.json config + meta +/data/reqs-*.jsonl requirements, merged in filename order +/data/overrides.jsonl post-verification corrections (optional) +/data/e2e-results.jsonl E2E outcomes (optional) +/evidence/… see references/evidence.md +/REPORT.html generated +``` + +Rendering is plain Python — deterministic, no network, no model. The judgement lives in the data. + +## reqs-*.jsonl + +One object per line. Required: `id` (unique across all files) and `status` (must be declared in +`report.json.status`). Every other key is free-form: a key shows up only if a column declares it. + +```json +{"id":"AUTH-001","area":"Login","req":"5 wrong passwords -> lock for 15 min","src":"spec.md:88-94","acc":"T-12","impl":"src/auth/lockout.ts:30-61","unit":"src/auth/__tests__/lockout.test.ts","e2e":"tests-e2e/auth-lockout.spec.ts","status":"PASS","qa":"—","note":""} +``` + +Name ids `-`: the default `group_by: "id_prefix"` cuts at the last `-` to group rows. +To group by a field instead, set `"group_by": "area"`. + +### Importing an existing flat list + +If the project already has a flat `reqs.txt` in the +`ID: [STATUS] requirement (src: file:line)` shape (what `emit.goalrun_reqs` writes, and what the +`goalrun` skill keeps), import it instead of retyping — no model involved: + +```bash +python3 scripts/build_report.py --from-reqs /reqs.txt +``` + +It writes `data/reqs-00-imported.jsonl` and builds. It refuses to overwrite an existing import, +so re-running never silently discards hand edits — delete the file or import into a fresh directory. + +## overrides.jsonl + +`{"id": "...", }`, merged onto the requirement with that id. Unknown id fails +the build. This is where a human reviewer records corrections — **do not edit `reqs-*.jsonl`** for +them, or the next run loses the boundary between what was derived and what a person concluded. + +## e2e-results.jsonl + +`{"id":"AUTH-001","result":"PASS|FAIL|BLOCKED|SKIP","date":"2026-01-20","evidence":"run #312"}` +Feeds any column with `"render": "e2e"`. Unknown id fails the build. + +## Column widths + +The main table uses `table-layout: fixed`, so every cell wraps instead of forcing the table +sideways — file paths break mid-token rather than stretching one column to 800px. Give the columns +that carry long text a larger `"width"` (any CSS length or percentage); columns without one share +what is left. + +Widths are a density budget: thirteen columns on a 1240px table leaves ~95px each, which is tight +for prose. If a column is not read in practice, drop it from `columns` — the JSONL keeps the field +either way, and fewer columns is the only real fix for a cramped table. + +## report.json + +| Key | Required | Meaning | +| --- | --- | --- | +| `title` | ✔ | `

` and `` | +| `header` | | one HTML paragraph: scope, spec sources, method | +| `columns` | ✔ | column order. `{"key","label","width"?,"nowrap"?,"render"?}` — `render` is `status`, `evidence` or `e2e`; omitted means plain text | +| `status` | ✔ | `{"KEY": {"icon","label","color"}}`. Declaration order drives the summary columns and the progress-bar segments | +| `group_by` | | `"id_prefix"` (default) or a field name | +| `findings` | | list of HTML strings for the "Key findings" section | +| `baseline` | | `{"label","total","status":{},"e2e":{}}` → the "Before → Now" table | +| `table_height` | | height of the requirement table's scroll box (default `72vh`). The table scrolls inside it so 700 rows do not turn the page into an endless scroll | +| `labels` | | override any UI string, e.g. `{"progress":"Tiến độ","evidence":"Bằng chứng"}`. Report language is config, not code | +| `rules.done` | | which statuses count as done for the headline percentage (default: first status) | +| `rules.demote_pass_without_test` | | `{"from","to","when_empty":[field,…]}` — demote when the row tracks tests at all (at least one listed field is present) and none of them holds a value. An empty field means "looked, found none" and counts toward the demotion; a row carrying none of the fields says nothing about tests, so an imported flat list keeps its statuses | +| `rules.e2e_fail_status` | | an E2E FAIL forces this status and records why in `note` | +| `rules.attention` | | statuses listed in "Needs action / confirmation" | +| `rules.attention_columns` | | extra columns for that section (default `["qa","note"]`) | +| `emit.goalrun_reqs` | | also write a flat `reqs.txt` for the `goalrun` skill | + +## Filtering + +The requirement table carries a free-text box plus one control per column, pinned under the header +while the box scrolls. A column gets a dropdown when its values are categorical — at most 25 distinct +*and* fewer distinct values than rows, so an id column (one value per row) stays a text box. The +evidence column takes no filter. All active filters combine with AND, and the counter beside the +free-text box reports how many rows survived. + +## Responsive behaviour + +One stylesheet, no breakpoints to configure. Wide screens put the scope text beside the progress +bars and the summary cards in one row; narrow screens stack them. Every table that is wider than the +screen scrolls inside its own card or box rather than widening the page, so there is no horizontal +page scroll at any width. Below 700px the requirement box is capped at `80vh` whatever +`table_height` says, and on touch screens the filter controls grow to 36px with 16px text (smaller +text makes iOS zoom on focus). + +## The build fails on + +duplicate Req ID · a status not declared in config · a status named by a rule (`done`, `attention`, +`e2e_fail_status`, `demote_pass_without_test.from`/`.to`) that `status` does not declare · an +override, E2E result or evidence row pointing at an unknown id · malformed JSONL. + +Loudly, never silently: a wrong report is worse than no report. diff --git a/.agents/skills/report/scripts/build_report.py b/.agents/skills/report/scripts/build_report.py new file mode 100644 index 0000000..696c167 --- /dev/null +++ b/.agents/skills/report/scripts/build_report.py @@ -0,0 +1,573 @@ +"""Build REPORT.html from report.json + data/*.jsonl. + +Run: python3 build_report.py <dir> (<dir> holds report.json) + python3 build_report.py <dir> --from-reqs <file> import a flat reqs.txt first + +Layout, all relative to <dir>: + report.json config: columns, statuses, rules, meta + data/reqs-*.jsonl one JSON object per requirement (merged in filename order) + data/overrides.jsonl {"id": ..., <fields to overwrite>} + data/e2e-results.jsonl {"id","result","date","evidence"} + evidence/*.manifest.jsonl {"file","caption","check"} + evidence/*.reqs.jsonl {"id","verdict","evidence":[file,...],"note"} + evidence/review-*.jsonl {"id","finding","detail"} independent reviewer +Rebuilding is idempotent: run it again after evidence lands and the images attach. +""" +import html +import json +import re +import sys +from collections import Counter +from datetime import datetime +from pathlib import Path + +EV_MARK = {"SHOWN": "🟢", "SHOWN-PARTIAL": "🟡", "NOT-SHOWN": "🔴", "NONE": "⬜"} +E2E_MARK = {"PASS": "✅", "FAIL": "🔴", "BLOCKED": "⛔", "SKIP": "—"} +LABELS = { + "progress": "Progress", + "status": "Status", + "evidence": "Evidence", + "evidence_unit": "req with SHOWN evidence", + "evidence_images": "images", + "evidence_findings": "review findings", + "evidence_none": "no evidence yet", + "evidence_partial": "partial", + "no_evidence_bucket": "not covered", + "e2e_ran": "E2E ran on", + "built_at": "Built at", + "baseline": "Compared with", + "baseline_before_after": "Before → Now", + "total_reqs": "Total reqs", + "group": "Group", + "by_group": "Breakdown by group", + "total": "Total", + "findings": "Key findings", + "summary": "Summary", + "attention": "Needs action / confirmation", + "table": "Requirements", + "filter": "filter by Req ID, status, test file …", + "req": "Requirement", + "rows": "rows", + "all": "all", + "filter_col": "…", + "demoted": "[auto] marked {frm} with no test.", + "e2e_failed": "E2E FAIL", +} + +# Reserved status palette (icon + label always accompany the colour; never colour alone). +STATUS_HUE = {"good": "#0ca30c", "warning": "#fab219", "serious": "#ec835a", + "critical": "#d03b3b", "neutral": "#898781", "quiet": "#c3c2b7"} +FALLBACK = list(STATUS_HUE.values()) + +CSS = """:root{ + color-scheme:light; + --plane:#fafaf8; --surface:#fff; --raise:#f2f0ea; + --ink:#1b1b1a; --ink2:#55534e; --muted:#8a877f; + --grid:#e3e1dc; --line:#d5d2c9; --ring:rgba(27,27,26,.10); --track:#e9e6df; + --crit:#b3261e; +} +*{box-sizing:border-box} +.page{max-width:1440px;margin:0 auto} +body{margin:0;padding:24px 20px 64px;background:var(--plane);color:var(--ink); + font:13px/1.55 system-ui,-apple-system,'Segoe UI','Hiragino Sans','Noto Sans JP',sans-serif; + -webkit-font-smoothing:antialiased} +h1{font-size:20px;line-height:1.3;margin:0 0 6px;letter-spacing:-.01em} +h2{font-size:12px;margin:36px 0 12px;text-transform:uppercase;letter-spacing:.07em; + color:var(--muted);font-weight:600} +p.src{margin:0 0 12px;color:var(--ink2);font-size:12px;max-width:110ch} +p.cap{margin:20px 0 6px;font-size:12px;font-weight:600;color:var(--ink)} +code{background:var(--raise);padding:1px 5px;border-radius:4px;font-size:.92em; + font-family:ui-monospace,SFMono-Regular,Menlo,monospace} +a{color:inherit} + +/* progress */ +.bar{display:flex;gap:2px;height:12px;border-radius:6px;overflow:hidden; + background:var(--track);margin:0 0 6px} +.bar span{display:block;min-width:2px} +.barlab{font-size:12px;color:var(--ink2);margin:0 0 18px;max-width:110ch} +.barlab b{color:var(--ink);font-variant-numeric:tabular-nums} +.dot{display:inline-block;width:8px;height:8px;border-radius:2px;margin-right:5px; + vertical-align:baseline;box-shadow:0 0 0 1px var(--ring) inset} + +/* layout */ +/* scope text and progress share the width instead of leaving half the page empty; + the text column keeps a readable measure rather than stretching to 180ch */ +.top{display:flex;flex-wrap:wrap;gap:16px 40px;align-items:flex-start;margin:0 0 8px} +.top>div{flex:1 1 460px;min-width:0} +.top .scope{max-width:82ch} +.top h2{margin-top:0} +.row{display:flex;flex-wrap:wrap;gap:16px;align-items:flex-start;margin:0 0 8px} +/* a card scrolls its own table sideways rather than pushing the page wider than the screen */ +.row section{flex:0 1 auto;min-width:0;max-width:100%;overflow-x:auto} +.row section.find{flex:1 1 320px} +/* a long caption must not decide the card's width: width:0 keeps it out of the + intrinsic size, min-width:100% makes it fill and wrap to the table's width */ +.row .cap{margin-top:0;width:0;min-width:100%} + +/* tables */ +table{border-collapse:separate;border-spacing:0;width:100%;background:var(--surface); + border:1px solid var(--grid);border-radius:8px;overflow:hidden;margin:0 0 28px} +table:last-child{margin-bottom:0} +th,td{border-bottom:1px solid var(--grid);padding:7px 10px;text-align:left;vertical-align:top} +/* file paths and ids offer no break opportunities -- without this a column never wraps */ +td{overflow-wrap:anywhere} +tr:last-child th,tr:last-child td{border-bottom:0} +th{background:var(--raise);font-weight:600;font-size:11px;letter-spacing:.03em; + text-transform:uppercase;color:var(--ink2);position:sticky;top:0;z-index:2; + border-bottom:1px solid var(--line)} +/* per-column filters ride just under the header, both pinned while the box scrolls */ +tr.f td{position:sticky;top:30px;z-index:2;background:var(--raise);padding:4px 6px; + border-bottom:1px solid var(--line)} +tr.f input,tr.f select{width:100%;margin:0;padding:3px 5px;font-size:11px;border-radius:5px} +/* touch: 11px inputs are both hard to hit and trigger focus-zoom on iOS */ +@media (pointer:coarse){ + tr.f input,tr.f select,#q{font-size:16px;min-height:36px;padding:6px 8px} +} +tr.f select{background:var(--surface)} +td{color:var(--ink2)} +/* indicator columns (ids, status) read as one token -- wrapping them mid-word is noise; + content columns keep wrapping */ +td.nw{color:var(--ink);white-space:nowrap} +th{white-space:nowrap;overflow:hidden;text-overflow:ellipsis} +/* the requirement table scrolls inside its own box -- 700+ rows must not turn the + page into an endless scroll, and the sticky header only works against this container */ +.tw{max-height:__TH__;overflow:auto;border:1px solid var(--grid);border-radius:8px; + margin:0 0 28px;background:var(--surface)} +.tw table{border:0;border-radius:0;margin:0} +table#t{table-layout:fixed;min-width:1240px} +.count{font-size:12px;color:var(--muted);margin-left:10px;font-variant-numeric:tabular-nums} +table.sum{width:auto;max-width:100%;font-variant-numeric:tabular-nums} +table.sum td,table.sum th{white-space:nowrap} +table.sum td{text-align:right;color:var(--ink)} +table.sum td:first-child,table.sum th:first-child{text-align:left} +table.sum th{text-align:right} +table.sum th,table.sum td{padding:6px 9px} +/* status headers stack the icon over the label so the column is only as wide as the word */ +table.sum th i{display:block;font-style:normal;line-height:1.2;margin-bottom:1px} + +/* filter */ +input{margin:0 0 12px;padding:7px 11px;width:340px;max-width:100%;font:inherit; + color:var(--ink);background:var(--surface);border:1px solid var(--line);border-radius:7px} +input:focus{outline:2px solid var(--ink2);outline-offset:1px} + +/* evidence */ +img.ev{width:100%;max-width:150px;border:1px solid var(--grid);border-radius:6px;display:block;margin-top:5px} +.evc{font-size:11px;line-height:1.45;color:var(--muted);max-width:220px} +.evr{font-size:11px;line-height:1.45;color:var(--crit);max-width:220px;margin-top:3px} +td.evt{min-width:200px} +ul{margin:0 0 4px;padding-left:18px;color:var(--ink2);max-width:110ch} +/* same for the stand-alone tables outside the card row */ +.scroll{overflow-x:auto;margin:0 0 28px} +.scroll table{margin:0} +li{margin-bottom:4px} + +/* last: these must win over the rules above */ +@media (max-width:700px){.tw{max-height:80vh}}""" + + +def die(msg): + raise SystemExit(f"build_report: {msg}") + + +def esc(v): + return html.escape(str(v if v not in (None, "") else "—")) + + +def load_jsonl(path): + rows = [] + for n, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1): + line = line.strip() + if not line: + continue + try: + rows.append(json.loads(line)) + except json.JSONDecodeError as e: + die(f"{path.name}:{n}: {e}") + return rows + + +def pct(part, total): + return 100.0 * part / total if total else 0.0 + + +def bar(segments, label): + """segments: [(count, color, name)] -> stacked progress bar.""" + total = sum(c for c, _, _ in segments) or 1 + cells = "".join( + f"<span style='width:{pct(c, total):.4f}%;background:{col}' title='{esc(n)}: {c}'></span>" + for c, col, n in segments if c and col) + return f"<div class=\"bar\">{cells}</div><div class='barlab'>{label}</div>" + + +class Report: + def __init__(self, root): + self.root = root + cfg_file = root / "report.json" + if not cfg_file.exists(): + die(f"no report.json in {root}") + self.cfg = json.loads(cfg_file.read_text(encoding="utf-8")) + self.status = self.cfg["status"] + self.columns = self.cfg["columns"] + self.rules = self.cfg.get("rules", {}) + self.data = root / "data" + self.ev_dir = root / "evidence" + self.lab = dict(LABELS, **self.cfg.get("labels", {})) + self.color = {k: v.get("color", FALLBACK[i % len(FALLBACK)]) + for i, (k, v) in enumerate(self.status.items())} + self.check_rule_statuses() + + def check_rule_statuses(self): + """Rules can assign a status; a typo there must die here, not as a KeyError mid-render.""" + named = [("rules.e2e_fail_status", [self.rules.get("e2e_fail_status")]), + ("rules.done", self.rules.get("done") or []), + ("rules.attention", self.rules.get("attention") or [])] + dem = self.rules.get("demote_pass_without_test") or {} + named += [("rules.demote_pass_without_test.%s" % k, [dem.get(k)]) for k in ("from", "to")] + bad = {where: v for where, vals in named for v in vals if v and v not in self.status} + if bad: + die("rules name statuses that report.json does not declare: %s (declared: %s)" + % (bad, ", ".join(self.status))) + + # ---------- load ---------- + + def load_reqs(self): + reqs = [] + for f in sorted(self.data.glob("reqs-*.jsonl")): + reqs += load_jsonl(f) + if not reqs: + die(f"no requirements found in {self.data}/reqs-*.jsonl") + by_id = {} + for r in reqs: + if "id" not in r: + die(f"requirement without id: {r}") + if r["id"] in by_id: + die(f"duplicate Req ID: {r['id']}") + by_id[r["id"]] = r + ov = self.data / "overrides.jsonl" + if ov.exists(): + for o in load_jsonl(ov): + if o.get("id") not in by_id: + die(f"override for unknown Req ID: {o.get('id')}") + by_id[o["id"]].update(o) + bad = {r["id"]: r.get("status") for r in reqs if r.get("status") not in self.status} + if bad: + die(f"unknown status: {bad}") + return reqs, by_id + + def load_e2e(self, by_id): + f = self.data / "e2e-results.jsonl" + if not f.exists(): + return {} + res = {} + for r in load_jsonl(f): + if r.get("id") not in by_id: + die(f"e2e result for unknown Req ID: {r.get('id')}") + res[r["id"]] = r + return res + + def load_evidence(self, by_id): + by_req, caption, findings = {}, {}, {} + if not self.ev_dir.exists(): + return by_req, caption, findings + for f in sorted(self.ev_dir.glob("*.manifest.jsonl")): + for m in load_jsonl(f): + caption[m["file"]] = m + for f in sorted(self.ev_dir.glob("*.reqs.jsonl")): + for r in load_jsonl(f): + if r.get("id") not in by_id: + die(f"evidence for unknown Req ID: {r.get('id')} ({f.name})") + by_req[r["id"]] = r + for f in sorted(self.ev_dir.glob("review-*.jsonl")): + for x in load_jsonl(f): + findings.setdefault(x["id"], []).append(x) + return by_req, caption, findings + + # ---------- rules ---------- + + def apply_rules(self, reqs, e2e): + fail_to = self.rules.get("e2e_fail_status") + for r in reqs: + res = e2e.get(r["id"]) + if not res: + continue + r["_e2e"] = (f"{E2E_MARK.get(res['result'], '?')} {res['result']}" + f" ({res.get('date', '—')}): {res.get('evidence', '—')}") + if res["result"] == "FAIL" and fail_to and r["status"] != fail_to: + r["note"] = (f"{self.lab['e2e_failed']} → {res.get('evidence', '')}. " + + (r.get("note") or "")) + r["status"] = fail_to + dem = self.rules.get("demote_pass_without_test") + if dem: + keys = dem.get("when_empty", []) + for r in reqs: + if r["status"] != dem["from"]: + continue + # Demote only when this dataset tracks tests at all: at least one of the + # fields is present (an empty one means "looked, found none") and none of + # them holds a value. A row carrying none of them -- an imported flat list -- + # says nothing about tests, so it keeps its status. + if not any(k in r or (k == "e2e" and "_e2e" in r) for k in keys): + continue + vals = [r.get("_e2e") if k == "e2e" else r.get(k) for k in keys] + if all(str(v or "—").strip() in ("—", "-", "") for v in vals): + r["status"] = dem["to"] + r["note"] = (self.lab["demoted"].format(frm=dem["from"]) + " " + + (r.get("note") or "")) + + # ---------- render ---------- + + def evidence_cell(self, ev, caption, findings): + if not ev and not findings: + return f"<td class='evt'>⬜ {esc(self.lab['evidence_none'])}</td>" + ev = ev or {"verdict": "NONE", "evidence": []} + parts = [f"{EV_MARK.get(ev['verdict'], '?')} <b>{esc(ev['verdict'])}</b>"] + for f in ev.get("evidence", []): + m = caption.get(f, {}) + ok = "✔" if m.get("check") == "PASS" else "✖" + parts.append( + f"<a href='evidence/{html.escape(f)}' target='_blank' title='{esc(m.get('caption'))}'>" + f"<img src='evidence/{html.escape(f)}' loading='lazy' class='ev'></a>" + f"<div class='evc'>{ok} {esc(f)}<br>{esc(m.get('caption'))}</div>") + if ev.get("note"): + parts.append(f"<div class='evc'>{esc(ev['note'])}</div>") + for x in findings or []: + parts.append(f"<div class='evr'>⚠ review {esc(x.get('finding'))}: {esc(x.get('detail'))}</div>") + return "<td class='evt'>" + "".join(parts) + "</td>" + + def cell(self, r, col, ev_by_req, ev_caption, ev_findings): + key, render = col["key"], col.get("render") + if render == "evidence": + return self.evidence_cell(ev_by_req.get(r["id"]), ev_caption, ev_findings.get(r["id"])) + if render == "status": + st = r["status"] + return (f"<td class='nw'><span class='dot' style='background:{self.color[st]}'></span>" + f"{esc(self.status[st]['icon'])} {esc(st)}</td>") + elif render == "e2e": + v = r.get("_e2e") or r.get(key) + else: + v = r.get(key) + cls = " class='nw'" if col.get("nowrap") else "" + return f"<td{cls}>{esc(v)}</td>" + + FILTER_MAX_OPTIONS = 25 + SHORT_AT = 10 + + def short_label(self, key): + """Compact header label: an explicit `short`, else initials of an OVER_LONG_KEY.""" + cfg = self.status[key].get("short") + if cfg: + return cfg + if len(key) <= self.SHORT_AT: + return key + parts = [p for p in key.split("_") if p] + return "".join(p[0] for p in parts) if len(parts) > 1 else key[:self.SHORT_AT - 1] + "." + + + def filter_row(self, reqs): + """A text box per column, or a dropdown when the column holds few distinct values.""" + cells = [] + for i, c in enumerate(self.columns): + key, render = c["key"], c.get("render") + if render == "evidence": + cells.append("<td></td>") + continue + if render == "status": + vals = list(self.status) + else: + seen = {str(r.get(key)).strip() for r in reqs if str(r.get(key) or "").strip()} + # a dropdown only helps for categorical columns: few values, and repeated. + # an id column has one value per row, so it stays a text box. + vals = (sorted(seen) if 1 < len(seen) <= self.FILTER_MAX_OPTIONS + and len(seen) < len(reqs) else None) + if vals: + opts = "".join("<option>%s</option>" % esc(v) for v in vals) + cells.append("<td><select data-i='%d' onchange='F()'><option value=''>%s</option>%s" + "</select></td>" % (i, esc(self.lab["all"]), opts)) + else: + cells.append("<td><input data-i='%d' oninput='F()' placeholder='%s'></td>" + % (i, esc(self.lab["filter_col"]))) + return "<tr class='f'>" + "".join(cells) + "</tr>" + + def sec(self): + self._sec += 1 + return f"{self._sec}." + + def group_of(self, r): + gb = self.cfg.get("group_by", "id_prefix") + return r["id"].rsplit("-", 1)[0] if gb == "id_prefix" else str(r.get(gb, "—")) + + def build(self): + reqs, by_id = self.load_reqs() + e2e = self.load_e2e(by_id) + self.apply_rules(reqs, e2e) + ev_by_req, ev_caption, ev_findings = self.load_evidence(by_id) + + n = len(reqs) + by_status = Counter(r["status"] for r in reqs) + e2e_count = Counter(v["result"] for v in e2e.values()) + ev_count = Counter(ev_by_req[r["id"]]["verdict"] if r["id"] in ev_by_req else "NONE" for r in reqs) + done = self.rules.get("done") or [next(iter(self.status))] + done_n = sum(by_status.get(s, 0) for s in done) + shown, partial = ev_count.get("SHOWN", 0), ev_count.get("SHOWN-PARTIAL", 0) + + o = [f"<!doctype html><meta charset='utf-8'><title>{esc(self.cfg['title'])}", + "", + "" % CSS.replace("__TH__", self.cfg.get("table_height", "72vh")), + "
", + f"

{self.cfg['title']}

"] + o.append("
") + if self.cfg.get("header"): + o.append(f"

{self.cfg['header']}

") + o.append("
") + + L = self.lab + self._sec = 0 + o.append(f"

{self.sec()} {esc(L['progress'])}

") + o.append(bar([(by_status.get(s, 0), self.color[s], s) for s in self.status], + f"{esc(L['status'])} — {pct(done_n, n):.0f}% {'/'.join(done)} ({done_n}/{n}) · " + + " · ".join( + f"" + f"{esc(self.status[s]['icon'])} {esc(s)}: {by_status.get(s, 0)}" + for s in self.status))) + o.append(bar([(shown, STATUS_HUE["good"], "SHOWN"), + (partial, STATUS_HUE["warning"], "SHOWN-PARTIAL"), + (n - shown - partial, "", L["no_evidence_bucket"])], + f"{esc(L['evidence'])} — {pct(shown, n):.0f}% " + f"({shown}/{n} {esc(L['evidence_unit'])}" + + (f", {partial} {esc(L['evidence_partial'])}" if partial else "") + + f") · {len(ev_caption)} {esc(L['evidence_images'])}" + + (f" · ⚠ {sum(len(v) for v in ev_findings.values())} {esc(L['evidence_findings'])}" + if ev_findings else ""))) + if e2e: + o.append(f"

{esc(L['e2e_ran'])} {len(e2e)} reqs: " + + " · ".join(f"{E2E_MARK.get(k, '?')} {k}: {v}" + for k, v in sorted(e2e_count.items())) + ".

") + o.append(f"

{esc(L['built_at'])} {datetime.now():%Y-%m-%d %H:%M}.

") + o.append("
") + + cards = [] + base = self.cfg.get("baseline") + if base: + def delta(now, before): + d = now - before + return f"{before} → {now}" + (f" ({'+' if d > 0 else ''}{d})" if d else "") + cards.append(f"

{esc(L['baseline'])} {esc(base['label'])}

") + cards.append(f"" + f"" + + f"" + + "".join(f"" + f"" + for s in self.status) + + "".join(f"" + f"" + for k in sorted(base.get("e2e", {}))) + + "
{esc(L['status'])}{esc(L['baseline_before_after'])}
{esc(L['total_reqs'])}{delta(n, base.get('total', 0))}
{esc(self.status[s]['icon'])} {esc(s)}{delta(by_status.get(s, 0), base.get('status', {}).get(s, 0))}
E2E {E2E_MARK.get(k, '?')} {k}{delta(e2e_count.get(k, 0), base.get('e2e', {}).get(k, 0))}
") + + cards = ["
" + "".join(cards) + "
"] if cards else [] + groups = sorted({self.group_of(r) for r in reqs}) + g = [f"

{esc(L['by_group'])}

"] + g.append(f"" + + "".join("" + % (esc(s), esc(self.status[s]["icon"]), esc(self.short_label(s))) + for s in self.status) + + "") + for name in groups: + c = Counter(r["status"] for r in reqs if self.group_of(r) == name) + g.append(f"" + + "".join(f"" for s in self.status) + "") + g.append(f"" + + "".join(f"" for s in self.status) + "
{esc(L['group'])}{esc(L['total'])}%s%s
{esc(name)}{sum(c.values())}{c.get(s, 0)}
{esc(L['total'])}{n}{by_status.get(s, 0)}
") + cards.append("
" + "".join(g) + "
") + if self.cfg.get("findings"): + cards.append("

" + esc(L["findings"]) + "

    " + + "".join(f"
  • {f}
  • " for f in self.cfg["findings"]) + "
") + o.append(f"

{self.sec()} {esc(L['summary'])}

") + o.append("
" + "".join(cards) + "
") + + attention = self.rules.get("attention", []) + if attention: + extra = [c for c in self.columns + if c["key"] in self.rules.get("attention_columns", ["qa", "note"])] + o.append(f"

{self.sec()} {esc(L['attention'])}

" + "
" + f"" + "".join(f"" for c in extra) + "") + for r in reqs: + if r["status"] in attention: + o.append(f"" + f"" + f"" + + "".join(f"" for c in extra) + "") + o.append("
Req ID{esc(L['status'])}{esc(L['req'])}{esc(c['label'])}
{esc(r['id'])}{esc(self.status[r['status']]['icon'])} {esc(r['status'])}{esc(r.get('req'))}{esc(r.get(c['key']))}
") + + o.append(f"

{self.sec()} {esc(L['table'])}

" + f"" + "%d %s" + "" % (n, esc(L["rows"]), n)) + head = [] + for c in self.columns: + cw = c.get("width") + style = " style='width:%s'" % cw if cw else "" + head.append("%s" % (style, esc(c["label"]))) + o.append("
" + "".join(head) + "" + + self.filter_row(reqs)) + for r in reqs: + cells = "".join(self.cell(r, c, ev_by_req, ev_caption, ev_findings) for c in self.columns) + blob = html.escape(" ".join(str(r.get(c["key"], "") or "") for c in self.columns) + + " " + r["status"] + " " + (r.get("_e2e") or ""), quote=True) + o.append(f'{cells}') + o.append("
") + + (self.root / "REPORT.html").write_text("\n".join(o), encoding="utf-8") + if self.cfg.get("emit", {}).get("goalrun_reqs"): + (self.root / "reqs.txt").write_text( + "".join(f"{r['id']}: [{r['status']}] {r.get('req', '')}" + + (f" (src: {r['src']})" if r.get("src") else "") + "\n" for r in reqs), + encoding="utf-8") + print(f"{n} reqs · status={dict(by_status)} · evidence={shown}/{n} SHOWN" + + (f" · e2e={dict(e2e_count)}" if e2e else "")) + + +REQ_LINE = re.compile(r"^(?P\S+):\s*\[(?P[A-Z_]+)\]\s*(?P.*?)" + r"(?:\s*\(src:\s*(?P.*)\))?\s*$") + + +def import_reqs_txt(src, out): + """Flat 'ID: [STATUS] requirement (src: ...)' lines -> reqs jsonl. No AI needed.""" + if out.exists(): + die(f"{out} already exists — delete it first or import into a fresh dir") + rows = [] + for n, line in enumerate(src.read_text(encoding="utf-8").splitlines(), 1): + if not line.strip(): + continue + m = REQ_LINE.match(line.strip()) + if not m: + die(f"{src.name}:{n}: cannot parse: {line.strip()[:80]}") + rows.append({k: v for k, v in m.groupdict().items() if v}) + if not rows: + die(f"{src}: no requirements found") + out.parent.mkdir(parents=True, exist_ok=True) + out.write_text("".join(json.dumps(r, ensure_ascii=False) + "\n" for r in rows), encoding="utf-8") + print(f"imported {len(rows)} reqs from {src} -> {out}") + + +def main(): + args = sys.argv[1:] + root = Path(args[0] if args and not args[0].startswith("-") else ".").resolve() + if "--from-reqs" in args: + src = Path(args[args.index("--from-reqs") + 1]).resolve() + import_reqs_txt(src, root / "data" / "reqs-00-imported.jsonl") + Report(root).build() + + +if __name__ == "__main__": + main() diff --git a/.agents/skills/report/scripts/test_build_report.py b/.agents/skills/report/scripts/test_build_report.py new file mode 100644 index 0000000..d1a9908 --- /dev/null +++ b/.agents/skills/report/scripts/test_build_report.py @@ -0,0 +1,283 @@ +"""Self-check for build_report.py. Run: python3 test_build_report.py""" +import json +import subprocess +import sys +import tempfile +from pathlib import Path + +HERE = Path(__file__).parent +BUILD = HERE / "build_report.py" + +CONFIG = { + "title": "T", + "header": "scope", + "findings": ["f1"], + "columns": [ + {"key": "id", "label": "ID", "nowrap": True}, + {"key": "req", "label": "Requirement"}, + {"key": "status", "label": "Status", "render": "status"}, + {"key": "evidence", "label": "Evidence", "render": "evidence"}, + {"key": "unit", "label": "Unit"}, + {"key": "e2e", "label": "E2E", "render": "e2e"}, + {"key": "note", "label": "Note"}, + ], + "status": { + "PASS": {"icon": "OK", "label": "done", "color": "#16a34a"}, + "NO_TEST": {"icon": "NT", "label": "no test", "color": "#eab308"}, + "MISSING": {"icon": "--", "label": "missing", "color": "#94a3b8"}, + "DEVIATION": {"icon": "XX", "label": "deviation", "color": "#dc2626"}, + }, + "rules": { + "done": ["PASS"], + "demote_pass_without_test": {"from": "PASS", "to": "NO_TEST", "when_empty": ["unit", "e2e"]}, + "e2e_fail_status": "DEVIATION", + "attention": ["DEVIATION", "MISSING"], + }, +} + +REQS = [ + {"id": "A-001", "area": "A", "req": "r1", "status": "PASS", "unit": "t.spec", "note": ""}, + {"id": "A-002", "area": "A", "req": "r2", "status": "PASS", "unit": "", "note": ""}, + {"id": "B-001", "area": "B", "req": "r3", "status": "PASS", "unit": "u.spec", "note": ""}, + {"id": "B-002", "area": "B", "req": "r4", "status": "MISSING", "unit": "", "note": ""}, +] + + +def write(d, rel, rows): + p = d / rel + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text("".join(json.dumps(r, ensure_ascii=False) + "\n" for r in rows), encoding="utf-8") + + +def setup(tmp, config=None, reqs=None): + d = Path(tmp) + (d / "report.json").write_text(json.dumps(config or CONFIG, ensure_ascii=False), encoding="utf-8") + write(d, "data/reqs-main.jsonl", reqs or REQS) + return d + + +def build(d, expect_fail=False): + r = subprocess.run([sys.executable, str(BUILD), str(d)], capture_output=True, text=True) + if expect_fail: + assert r.returncode != 0, f"expected failure, got:\n{r.stdout}" + return r.stderr + r.stdout + assert r.returncode == 0, f"build failed:\n{r.stderr}" + return (d / "REPORT.html").read_text(encoding="utf-8") + + +def test_demote_pass_without_test(): + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp) + out = build(d) + assert "A-002" in out + # A-002 has no unit and no e2e -> demoted + row = [l for l in out.splitlines() if "A-002" in l and "data-r" in l][0] + assert "NO_TEST" in row, row + row1 = [l for l in out.splitlines() if "A-001" in l and "data-r" in l][0] + assert "NO_TEST" not in row1, row1 + + +def test_missing_test_fields_are_unknown_not_demoted(): + """Absent columns (e.g. an imported flat list) must not be read as 'no test'.""" + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp, reqs=[{"id": "A-001", "req": "r", "status": "PASS"}]) + row = [l for l in build(d).splitlines() if "A-001" in l and "data-r" in l][0] + assert "NO_TEST" not in row, row + + +def test_partial_test_columns_still_demote(): + """One tracked-but-empty test column is enough: PASS with no test evidence is not PASS.""" + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp, reqs=[{"id": "A-001", "req": "r", "status": "PASS", "unit": ""}]) + row = [l for l in build(d).splitlines() if "A-001" in l and "data-r" in l][0] + assert "NO_TEST" in row, row + + +def test_rule_statuses_must_be_declared(): + """A typo in a rule's status dies with a message, not a KeyError traceback.""" + for path, value in [("e2e_fail_status", "DEVIATON"), ("done", ["PASSS"]), + ("attention", ["NOPE"])]: + cfg = json.loads(json.dumps(CONFIG)) + cfg["rules"][path] = value + with tempfile.TemporaryDirectory() as tmp: + err = build(setup(tmp, config=cfg), expect_fail=True) + assert "does not declare" in err, (path, err[-300:]) + assert "Traceback" not in err, (path, err[-300:]) + cfg = json.loads(json.dumps(CONFIG)) + cfg["rules"]["demote_pass_without_test"]["to"] = "NOT_A_STATUS" + with tempfile.TemporaryDirectory() as tmp: + assert "does not declare" in build(setup(tmp, config=cfg), expect_fail=True) + + +def test_e2e_fail_forces_deviation(): + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp) + write(d, "data/e2e-results.jsonl", + [{"id": "A-001", "result": "FAIL", "date": "2026-01-01", "evidence": "run#3"}]) + out = build(d) + row = [l for l in out.splitlines() if "A-001" in l and "data-r" in l][0] + assert "DEVIATION" in row, row + assert "run#3" in row, row + + +def test_overrides_apply(): + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp) + write(d, "data/overrides.jsonl", [{"id": "B-002", "status": "PASS", "unit": "fixed.spec"}]) + out = build(d) + row = [l for l in out.splitlines() if "B-002" in l and "data-r" in l][0] + assert "fixed.spec" in row and "MISSING" not in row, row + + +def test_validation_errors(): + checks = [ + ("duplicate", REQS + [dict(REQS[0])], None, "duplicate"), + ("bad status", [dict(REQS[0], status="WAT")], None, "unknown status"), + ] + for name, reqs, _, needle in checks: + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp, reqs=reqs) + assert needle in build(d, expect_fail=True).lower(), name + with tempfile.TemporaryDirectory() as tmp: # e2e for unknown id + d = setup(tmp) + write(d, "data/e2e-results.jsonl", + [{"id": "ZZ-9", "result": "PASS", "date": "x", "evidence": "y"}]) + assert "unknown" in build(d, expect_fail=True).lower() + with tempfile.TemporaryDirectory() as tmp: # override for unknown id + d = setup(tmp) + write(d, "data/overrides.jsonl", [{"id": "ZZ-9", "status": "PASS"}]) + assert "unknown" in build(d, expect_fail=True).lower() + + +def test_progress_bars(): + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp) + out = build(d) + # 4 reqs: A-001 PASS, A-002 -> NO_TEST, B-001 PASS, B-002 MISSING => 50% done + assert 'class="bar"' in out + assert "50%" in out, "status progress percent missing" + assert "0%" in out, "evidence progress percent missing" + + +def test_evidence_attaches_and_review_warns(): + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp) + (d / "evidence").mkdir() + write(d, "evidence/s01.manifest.jsonl", + [{"file": "EV-1.png", "caption": "login empty", "check": "PASS"}]) + write(d, "evidence/s01.reqs.jsonl", + [{"id": "A-001", "verdict": "SHOWN", "evidence": ["EV-1.png"]}]) + write(d, "evidence/review-s01.jsonl", + [{"id": "A-001", "finding": "caption-mismatch", "detail": "other screen"}]) + out = build(d) + row = [l for l in out.splitlines() if "A-001" in l and "data-r" in l][0] + assert "evidence/EV-1.png" in row and "login empty" in row, row + assert "caption-mismatch" in row, "review finding not surfaced" + assert "25%" in out, "evidence bar should be 1/4" + + +def test_baseline_delta(): + cfg = dict(CONFIG, baseline={"label": "prev", "total": 3, "status": {"PASS": 3}}) + with tempfile.TemporaryDirectory() as tmp: + d = setup(tmp, config=cfg) + out = build(d) + assert "prev" in out and "3 →" in out + # both summary tables must sit inside the one flex row, each in its own section + row = out[out.index("
"):out.index("") == 2, row[:200] + assert row.count("
") == 2, "a summary table escaped the row" + assert "
" not in out[:out.index("
")] + + +def test_import_reqs_txt(): + with tempfile.TemporaryDirectory() as tmp: + d = Path(tmp) + (d / "report.json").write_text(json.dumps(CONFIG, ensure_ascii=False), encoding="utf-8") + flat = d / "reqs.txt" + flat.write_text("A-001: [PASS] does a thing (src: spec.md:12)\n" + "A-002: [MISSING] does another\n", encoding="utf-8") + r = subprocess.run([sys.executable, str(BUILD), str(d), "--from-reqs", str(flat)], + capture_output=True, text=True) + assert r.returncode == 0, r.stderr + rows = [json.loads(l) for l in + (d / "data" / "reqs-00-imported.jsonl").read_text(encoding="utf-8").splitlines()] + assert rows[0] == {"id": "A-001", "status": "PASS", "req": "does a thing", "src": "spec.md:12"} + assert rows[1] == {"id": "A-002", "status": "MISSING", "req": "does another"} + out = (d / "REPORT.html").read_text(encoding="utf-8") + assert "does a thing" in out + # re-import must refuse rather than clobber + r2 = subprocess.run([sys.executable, str(BUILD), str(d), "--from-reqs", str(flat)], + capture_output=True, text=True) + assert r2.returncode != 0 and "already exists" in r2.stderr + r2.stdout + + +def test_table_scrolls_in_its_own_box(): + with tempfile.TemporaryDirectory() as tmp: + out = build(setup(tmp)) + assert "
" in out + assert out.rstrip().endswith("
") + assert "max-height:72vh;overflow:auto" in out + cfg = dict(CONFIG, table_height="1800px") + with tempfile.TemporaryDirectory() as t2: + assert "max-height:1800px" in build(setup(t2, config=cfg)) + assert ">4 rows" in out, "row counter missing" + + +def test_scope_and_progress_share_the_width(): + with tempfile.TemporaryDirectory() as tmp: + out = build(setup(tmp)) + top = out[out.index("
"):out.index("
")] + assert top.count("") == 0, "unbalanced top row" + assert "scope" in top and "Progress" in top + + +def test_column_filters(): + """Categorical columns get a dropdown; id-like columns stay a text box.""" + with tempfile.TemporaryDirectory() as tmp: + out = build(setup(tmp)) + frow = out[out.index(""):out.index("", out.index(""))] + assert frow.count(" dropdowns" + assert "" in frow and "" in frow + assert "data-i='0'" in frow and "