diff --git a/.github/agents/wazzup-curator.agent.md b/.github/agents/wazzup-curator.agent.md index 7de6b03..b8524ed 100644 --- a/.github/agents/wazzup-curator.agent.md +++ b/.github/agents/wazzup-curator.agent.md @@ -1,7 +1,7 @@ --- description: "Use when: curating news items for the Wazzup briefing, selecting the most relevant and newsworthy articles from a scored and ranked list." name: "wazzup-curator" -model: "claude-sonnet-4.6" +model: "claude-sonnet-5" tools: [execute, edit] user-invocable: false argument-hint: "Path to curation-input.json and requested output file" diff --git a/.github/agents/wazzup-transparency-reporter.agent.md b/.github/agents/wazzup-transparency-reporter.agent.md index 198709b..d0fbe05 100644 --- a/.github/agents/wazzup-transparency-reporter.agent.md +++ b/.github/agents/wazzup-transparency-reporter.agent.md @@ -1,7 +1,7 @@ --- description: "Use when: writing Wazzup transparency reports that explain scoring, missed news, curation cutoffs, and tuning options from briefing metadata." name: "wazzup-transparency-reporter" -model: "claude-sonnet-4.6" +model: "claude-sonnet-5" tools: [execute, edit] user-invocable: false argument-hint: "Path to transparency-input.json and requested output file" diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index 2810cef..7f332b1 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -119,8 +119,10 @@ Python package [src/wazzup/](src/wazzup): freshness scoring. - [`ai.py`](src/wazzup/ai.py) – AI provider abstraction. Selected by env `AI_PROVIDER` (`copilot-cli` or `fake`). `fake` is the default in tests - and CI. The Copilot CLI provider defaults to `COPILOT_MODEL=claude-sonnet-4.6` - and `COPILOT_AGENT=wazzup-writer`; the agent prompt lives in + and CI. Curator and transparency default to `COPILOT_MODEL=claude-sonnet-5`. + The writer defaults to `claude-opus-4.8`, with override precedence + `COPILOT_WRITER_MODEL` then `COPILOT_MODEL`, and uses + `COPILOT_AGENT=wazzup-writer`; the agent prompt lives in [`.github/agents/wazzup-writer.agent.md`](.github/agents/wazzup-writer.agent.md). - [`publisher.py`](src/wazzup/publisher.py) – writes YAML state and JSON mirrors under `public/data/`. diff --git a/.github/workflows/news.yml b/.github/workflows/news.yml index 66c379a..f9f9474 100644 --- a/.github/workflows/news.yml +++ b/.github/workflows/news.yml @@ -44,7 +44,7 @@ jobs: WAZZUP_TIMEZONE: Europe/Amsterdam WAZZUP_MAX_AI_ITEMS: 30 WAZZUP_MAX_AI_COST_USD: "1.00" - COPILOT_MODEL: claude-sonnet-4.6 + COPILOT_MODEL: claude-sonnet-5 COPILOT_WRITER_MODEL: claude-opus-4.8 REQUESTED_AI_PROVIDER: ${{ inputs.aiProvider || 'copilot-cli' }} steps: diff --git a/docs/adr/0002-ai-execution-strategy.md b/docs/adr/0002-ai-execution-strategy.md index 943874c..6e43311 100644 --- a/docs/adr/0002-ai-execution-strategy.md +++ b/docs/adr/0002-ai-execution-strategy.md @@ -38,7 +38,8 @@ Implemented safety behavior: - [../../.github/workflows/news.yml](../../.github/workflows/news.yml) selects an effective provider before installing Node or Copilot CLI. - If `copilot-cli` is requested without either token secret, the workflow logs a warning and uses `AI_PROVIDER=fake`. - [../../src/wazzup/ai.py](../../src/wazzup/ai.py) checks for `COPILOT_GITHUB_TOKEN` in GitHub Actions and raises an actionable error if the workflow guard is bypassed. -- The provider defaults to model `claude-sonnet-4.6` and the repo-local `wazzup-writer` custom agent, both overridable through environment variables. +- Curator and transparency default to `claude-sonnet-5` via `COPILOT_MODEL`. The writer uses the repo-local `wazzup-writer` custom agent and selects `COPILOT_WRITER_MODEL`, then `COPILOT_MODEL`, then `claude-opus-4.8`; the News workflow explicitly pins both models. +- Pass explicit model IDs with `--model`. An unavailable pinned model is a blocking failure, not a reason to retry with another model or the CLI default. - Copilot CLI stdout/stderr is captured and included in sanitized failure diagnostics when the CLI exits non-zero. ## Consequences diff --git a/docs/architecture.md b/docs/architecture.md index 5fdb2ee..764932b 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -18,7 +18,7 @@ Implementation deviations from the original target are intentional for the curre - Frontend is vanilla HTML/CSS/JavaScript under [../public](../public); there is no frontend build step yet. - JSON is the canonical generated state format consumed directly by the PWA. - Pages state restoration supports tokenless public release-asset downloads because reusable workflow string inputs cannot reliably inject `GH_TOKEN` for nested shell commands. -- The News workflow requests Copilot CLI by default, keeps curator and transparency on Claude Sonnet 4.6 (`claude-sonnet-4.6`), pins the briefing writer to Claude Opus 4.8 (`claude-opus-4.8`) unless `COPILOT_WRITER_MODEL` overrides it, and falls back to the deterministic fake provider if Copilot token secrets are missing. +- The News workflow requests Copilot CLI by default, keeps curator and transparency on Claude Sonnet 5 (`claude-sonnet-5`), pins the briefing writer to Claude Opus 4.8 (`claude-opus-4.8`) unless `COPILOT_WRITER_MODEL` overrides it, and falls back to the deterministic fake provider if Copilot token secrets are missing. Unavailable pinned models block the run rather than triggering model or provider substitution. ## Context diagram diff --git a/docs/github-actions.md b/docs/github-actions.md index e43a702..a42dc8b 100644 --- a/docs/github-actions.md +++ b/docs/github-actions.md @@ -123,7 +123,8 @@ jobs: WAZZUP_TIMEZONE: Europe/Amsterdam WAZZUP_MAX_AI_ITEMS: 30 WAZZUP_MAX_AI_COST_USD: 1.00 - COPILOT_MODEL: claude-sonnet-4.6 + COPILOT_MODEL: claude-sonnet-5 + COPILOT_WRITER_MODEL: claude-opus-4.8 steps: - uses: actions/checkout@v6 - uses: jdx/mise-action@v4 @@ -155,6 +156,7 @@ jobs: AI_PROVIDER: ${{ steps.ai.outputs.provider }} COPILOT_GITHUB_TOKEN: ${{ secrets.COPILOT_REQUESTS_PAT || secrets.COPILOT_GITHUB_TOKEN }} COPILOT_MODEL: ${{ env.COPILOT_MODEL }} + COPILOT_WRITER_MODEL: ${{ env.COPILOT_WRITER_MODEL }} ``` The workflow triggers hourly because GitHub cron is UTC-only and does not understand `Europe/Amsterdam` daylight-saving transitions. A first cadence step computes the local hour and continues only on odd local hours from 07:00 to 21:59. This aligns the first run with the configured morning briefing, keeps AI calls to a daytime two-hour cadence, and skips overnight runs entirely. Manual dispatch always runs. @@ -167,7 +169,9 @@ Operational learning: the site served stale data after manual and watchdog-trigg Operational learning: the site went stale for most of a day because GitHub schedule jitter routinely shifts the hourly `News` cron across hour boundaries. The cadence gate evaluates the local hour at execution time, so a run intended for an odd local hour (07:00, 09:00, …) frequently lands on an even or overnight hour and skips all real work. The original `news-watchdog` could not recover from this: it only checked whether _any_ `News` run existed in the current UTC hour, and since the hourly cron always fires (even when it gate-skips in seconds), it always saw a run and never dispatched a catch-up. The watchdog now runs every hour inside the active local window (07:00–21:59, no odd-hour parity gate so jitter cannot disable it) and decides purely on freshness: it inspects recent `News` runs, treats only a successful `Generate and persist retained news state` step as an effective generation, and dispatches a `workflow_dispatch` catch-up when the newest effective generation is older than the two-hour cadence interval and no `News` run is currently queued or in progress. This keeps the two-hour spacing while making missed scheduled runs self-healing. -The Copilot CLI provider keeps curator and transparency on `COPILOT_MODEL`, defaulting to Claude Sonnet 4.6 via the CLI model ID `claude-sonnet-4.6`, and pins the briefing writer to `COPILOT_WRITER_MODEL`, defaulting to Claude Opus 4.8 via `claude-opus-4.8`. This isolates the more expensive model to the writing step while keeping the model overridable for manual canaries. +The Copilot CLI provider keeps curator and transparency on `COPILOT_MODEL`, defaulting to Claude Sonnet 5 via the CLI model ID `claude-sonnet-5`, and pins the briefing writer to `COPILOT_WRITER_MODEL`, defaulting to Claude Opus 4.8 via `claude-opus-4.8`. This isolates the more expensive model to the writing step while keeping the model overridable for manual canaries. Explicit constructor model overrides take precedence; outside the workflow, the writer uses `COPILOT_WRITER_MODEL`, then `COPILOT_MODEL`, then its Opus default. + +Model pins are intentional: if Copilot CLI reports that the model passed with `--model` is unavailable, the provider surfaces the CLI diagnostics and blocks the run, including transparency reporting. It does not retry with a different model, omit `--model`, or substitute a deterministic provider for this error. Existing missing-token and malformed-output fallback behavior is unchanged. ## Copilot CLI workflow variant @@ -185,7 +189,7 @@ The implementation hides provider-specific commands behind `task news:generate` env: AI_PROVIDER: copilot-cli COPILOT_GITHUB_TOKEN: ${{ secrets.COPILOT_REQUESTS_PAT || secrets.COPILOT_GITHUB_TOKEN }} - COPILOT_MODEL: claude-sonnet-4.6 + COPILOT_MODEL: claude-sonnet-5 COPILOT_WRITER_MODEL: claude-opus-4.8 FORCE_BRIEFING: ${{ inputs.forceBriefing || 'auto' }} run: task news:generate @@ -194,7 +198,7 @@ The implementation hides provider-specific commands behind `task news:generate` Provider adapter requirements: - Run Copilot CLI in programmatic mode with `copilot -p`. -- Pass `--model` from `COPILOT_WRITER_MODEL` for the briefing writer, defaulting to `claude-opus-4.8`; curator and transparency keep using `COPILOT_MODEL`, defaulting to `claude-sonnet-4.6`. +- Pass `--model` from `COPILOT_WRITER_MODEL` for the briefing writer, defaulting to `claude-opus-4.8`; curator and transparency keep using `COPILOT_MODEL`, defaulting to `claude-sonnet-5`. - Pass `--agent wazzup-writer` so the briefing-writing style and JSON contract live in [../.github/agents/wazzup-writer.agent.md](../.github/agents/wazzup-writer.agent.md). - Use `--no-ask-user` so the workflow never blocks for interaction. - Use only narrow `--allow-tool` permissions, such as read-only shell access to generated prompt/input files and write access to a temporary output file. @@ -281,7 +285,7 @@ Expected secrets: | ---------------------------- | ----------------------------------------------------------------------------------------------------------- | | `COPILOT_REQUESTS_PAT` | Preferred repository secret containing a fine-grained PAT for Copilot CLI with Copilot Requests permission. | | `COPILOT_GITHUB_TOKEN` | Alternative secret name accepted by the News workflow. | -| `COPILOT_MODEL` | Optional Copilot CLI model override for curator and transparency; defaults to `claude-sonnet-4.6`. | +| `COPILOT_MODEL` | Optional Copilot CLI model override for curator and transparency; defaults to `claude-sonnet-5`. | | `COPILOT_WRITER_MODEL` | Optional Copilot CLI model override for the briefing writer; defaults to `claude-opus-4.8`. | | `AZURE_OPENAI_ENDPOINT` | Azure OpenAI endpoint. | | `AZURE_OPENAI_API_KEY` | Azure OpenAI API key. | diff --git a/src/wazzup/ai.py b/src/wazzup/ai.py index e4463ec..c54ca01 100644 --- a/src/wazzup/ai.py +++ b/src/wazzup/ai.py @@ -92,7 +92,7 @@ def generate_transparency_report(self, request: TransparencyReportRequest) -> Tr """Generate a structured transparency report for the pipeline run.""" -DEFAULT_COPILOT_MODEL = "claude-sonnet-4.6" +DEFAULT_COPILOT_MODEL = "claude-sonnet-5" DEFAULT_COPILOT_WRITER_MODEL = "claude-opus-4.8" DEFAULT_COPILOT_AGENT = "wazzup-writer" DEFAULT_COPILOT_CURATOR_AGENT = "wazzup-curator" @@ -520,6 +520,8 @@ def generate_transparency_report(self, request: TransparencyReportRequest) -> Tr } return transparency_response_from_payload(payload, provider=provider) except (RuntimeError, ValueError, json.JSONDecodeError) as exc: + if f'Model "{self.model}" from --model flag is not available' in str(exc): + raise fallback = FakeTransparencyReportProvider().generate_transparency_report(request) return TransparencyReportResponse( title=fallback.title, diff --git a/tests/test_ai.py b/tests/test_ai.py index 51e5268..0737e8e 100644 --- a/tests/test_ai.py +++ b/tests/test_ai.py @@ -355,7 +355,7 @@ def test_copilot_cli_writer_model_override_takes_precedence(self, _which, run_mo previous_model = os.environ.get("COPILOT_MODEL") previous_writer_model = os.environ.get("COPILOT_WRITER_MODEL") previous_token = os.environ.get("COPILOT_GITHUB_TOKEN") - os.environ["COPILOT_MODEL"] = "claude-sonnet-4.6" + os.environ["COPILOT_MODEL"] = "claude-sonnet-5" os.environ["COPILOT_WRITER_MODEL"] = "claude-opus-4.8" os.environ["COPILOT_GITHUB_TOKEN"] = "test-token" @@ -399,6 +399,94 @@ def fake_run(command, capture_output, cwd, env, text): # type: ignore[no-untype self.assertEqual("claude-opus-4.8", response.provider["model"]) + @patch("wazzup.ai.subprocess.run") + @patch("wazzup.ai.shutil.which", return_value="/usr/bin/copilot") + def test_copilot_cli_pinned_models_and_overrides_block_when_unavailable(self, _which, run_mock) -> None: # type: ignore[no-untyped-def] + cases = [ + ( + CopilotCliCurationProvider, + "curate_items", + CurationRequest( + kind="hourly", + window_start="2026-05-06T20:00:00Z", + window_end="2026-05-06T21:00:00Z", + generated_at="2026-05-06T21:00:00Z", + timezone="Europe/Amsterdam", + items=[], + max_items=12, + ), + "claude-sonnet-5", + ), + ( + CopilotCliSummaryProvider, + "generate_structured_summary", + SummaryRequest( + kind="hourly", + window_start="2026-05-06T20:00:00Z", + window_end="2026-05-06T21:00:00Z", + generated_at="2026-05-06T21:00:00Z", + timezone="Europe/Amsterdam", + summary_language="en", + items=[], + ), + "claude-opus-4.8", + ), + ( + CopilotCliTransparencyReportProvider, + "generate_transparency_report", + TransparencyReportRequest( + kind="hourly", + window_start="2026-05-06T20:00:00Z", + window_end="2026-05-06T21:00:00Z", + generated_at="2026-05-06T21:00:00Z", + timezone="Europe/Amsterdam", + summary_language="en", + max_items=12, + statuses=[], + ranked_items=[], + selected_items=[], + curation_provider={"type": "fake"}, + summary_provider={"type": "fake"}, + ), + "claude-sonnet-5", + ), + ] + for provider_type, method_name, request, default_model in cases: + overrides = [ + ({}, None, default_model), + ({"COPILOT_MODEL": "shared-model"}, None, "shared-model"), + ( + {"COPILOT_MODEL": "shared-model", "COPILOT_WRITER_MODEL": "writer-model"}, + None, + "writer-model" if provider_type is CopilotCliSummaryProvider else "shared-model", + ), + ( + {"COPILOT_MODEL": "shared-model", "COPILOT_WRITER_MODEL": "writer-model"}, + "explicit-model", + "explicit-model", + ), + ] + for environment, model, expected_model in overrides: + for diagnostic_stream in ("stdout", "stderr"): + with self.subTest(provider=provider_type.__name__, environment=environment, model=model, stream=diagnostic_stream): + error = f'Model "{expected_model}" from --model flag is not available.' + run_mock.reset_mock() + run_mock.return_value = Mock( + returncode=1, + stdout=error if diagnostic_stream == "stdout" else "", + stderr=error if diagnostic_stream == "stderr" else "", + ) + with patch.dict(os.environ, {"COPILOT_GITHUB_TOKEN": "test-token", **environment}, clear=True): + provider = provider_type(model=model) + with self.assertRaises(RuntimeError) as raised: + getattr(provider, method_name)(request) + self.assertIn(error, str(raised.exception)) + self.assertIn("exit code 1", str(raised.exception)) + run_mock.assert_called_once() + command = run_mock.call_args.args[0] + self.assertEqual(1, command.count("--model")) + self.assertEqual(expected_model, command[command.index("--model") + 1]) + @patch("wazzup.ai.shutil.which", return_value="/usr/bin/copilot") def test_copilot_requires_token_in_github_actions(self, _which) -> None: # type: ignore[no-untyped-def] previous_actions = os.environ.get("GITHUB_ACTIONS") @@ -573,8 +661,10 @@ def test_curation_payload_includes_scored_items(self) -> None: @patch("wazzup.ai.subprocess.run") @patch("wazzup.ai.shutil.which", return_value="/usr/bin/copilot") def test_copilot_cli_curation_uses_curator_agent(self, _which, run_mock) -> None: # type: ignore[no-untyped-def] + previous_model = os.environ.get("COPILOT_MODEL") previous_agent = os.environ.get("COPILOT_CURATOR_AGENT") previous_token = os.environ.get("COPILOT_GITHUB_TOKEN") + os.environ.pop("COPILOT_MODEL", None) os.environ.pop("COPILOT_CURATOR_AGENT", None) os.environ["COPILOT_GITHUB_TOKEN"] = "test-token" @@ -612,6 +702,10 @@ def fake_run(command, capture_output, cwd, env, text): # type: ignore[no-untype ) ) finally: + if previous_model is None: + os.environ.pop("COPILOT_MODEL", None) + else: + os.environ["COPILOT_MODEL"] = previous_model if previous_agent is None: os.environ.pop("COPILOT_CURATOR_AGENT", None) else: @@ -772,8 +866,10 @@ def test_transparency_response_from_payload_validates_shape(self) -> None: @patch("wazzup.ai.subprocess.run") @patch("wazzup.ai.shutil.which", return_value="/usr/bin/copilot") def test_copilot_cli_transparency_uses_reporter_agent(self, _which, run_mock) -> None: # type: ignore[no-untyped-def] + previous_model = os.environ.get("COPILOT_MODEL") previous_agent = os.environ.get("COPILOT_TRANSPARENCY_AGENT") previous_token = os.environ.get("COPILOT_GITHUB_TOKEN") + os.environ.pop("COPILOT_MODEL", None) os.environ.pop("COPILOT_TRANSPARENCY_AGENT", None) os.environ["COPILOT_GITHUB_TOKEN"] = "test-token" @@ -811,6 +907,10 @@ def fake_run(command, capture_output, cwd, env, text): # type: ignore[no-untype ) ) finally: + if previous_model is None: + os.environ.pop("COPILOT_MODEL", None) + else: + os.environ["COPILOT_MODEL"] = previous_model if previous_agent is None: os.environ.pop("COPILOT_TRANSPARENCY_AGENT", None) else: diff --git a/tests/test_workflows.py b/tests/test_workflows.py index 89b83ed..c6207c0 100644 --- a/tests/test_workflows.py +++ b/tests/test_workflows.py @@ -12,7 +12,7 @@ def test_news_workflow_uses_active_two_hour_local_cadence_gate(self) -> None: self.assertIn("default: auto", workflow) self.assertIn("- auto", workflow) self.assertIn("Check scheduled cadence", workflow) - self.assertIn("COPILOT_MODEL: claude-sonnet-4.6", workflow) + self.assertIn("COPILOT_MODEL: claude-sonnet-5", workflow) self.assertIn("COPILOT_WRITER_MODEL: claude-opus-4.8", workflow) self.assertIn("- Copilot model: ${COPILOT_MODEL}", workflow) self.assertIn("- Copilot writer model: ${COPILOT_WRITER_MODEL}", workflow) @@ -22,6 +22,16 @@ def test_news_workflow_uses_active_two_hour_local_cadence_gate(self) -> None: self.assertNotIn("overnight two-hour cadence", workflow) self.assertNotIn('cron: "d"', workflow) + def test_copilot_agents_keep_explicit_model_pins(self) -> None: + for agent, model in ( + ("wazzup-curator", "claude-sonnet-5"), + ("wazzup-transparency-reporter", "claude-sonnet-5"), + ("wazzup-writer", "claude-opus-4.8"), + ): + with self.subTest(agent=agent): + definition = Path(f".github/agents/{agent}.agent.md").read_text(encoding="utf-8") + self.assertIn(f'\nmodel: "{model}"\n', definition) + def test_news_workflow_dispatches_pages_deploy_for_every_trigger(self) -> None: workflow = Path(".github/workflows/news.yml").read_text(encoding="utf-8") self.assertIn("actions: write", workflow)