From a4d14de6401ce27408031f5c23fbc67742cd601e Mon Sep 17 00:00:00 2001 From: Jeff Huber Date: Mon, 21 Sep 2026 17:08:48 -0700 Subject: [PATCH 1/4] Harden bounded audit comment ingestion --- .github/workflows/code-mower-gate.yml | 50 ++++- CHANGELOG.md | 7 +- src/code_mower/audit_labeler_lib.py | 211 ++++++++++++++++-- src/code_mower/builder_lineage.py | 44 +++- src/code_mower/builder_lineage_producer.py | 53 ++--- src/code_mower/gate_health.py | 28 ++- src/code_mower/lane_status.py | 5 +- src/code_mower/provider_runners/github_pr.py | 46 +++- src/code_mower/saas_reviewer_labeler.py | 38 ++-- .../workflows/code-mower-gate.yml.j2 | 50 ++++- tests/lineage_producer_fixtures.py | 2 +- tests/test_comment_history.py | 104 +++++++++ tests/test_lineage_consumer_labels.py | 6 +- tests/test_lineage_consumer_projection.py | 10 +- tests/test_lineage_producer_artifacts.py | 6 +- tests/test_lineage_producer_publication.py | 16 +- tests/test_minimal_lineage_transport.py | 16 +- tests/test_provider_runners_github_pr.py | 44 +++- tools/audit_labeler_lib.py | 211 ++++++++++++++++-- tools/builder_lineage.py | 44 +++- 20 files changed, 850 insertions(+), 141 deletions(-) create mode 100644 tests/test_comment_history.py diff --git a/.github/workflows/code-mower-gate.yml b/.github/workflows/code-mower-gate.yml index 7b7a350a..33be42db 100644 --- a/.github/workflows/code-mower-gate.yml +++ b/.github/workflows/code-mower-gate.yml @@ -126,7 +126,49 @@ jobs: workflow_runs_file="$(mktemp)" audit_runs_file="$(mktemp)" gh api "repos/${GITHUB_REPOSITORY}/issues/${PR_NUMBER}/labels" > "${labels_file}" - gh api --paginate --slurp "repos/${GITHUB_REPOSITORY}/issues/${PR_NUMBER}/comments?per_page=100" > "${comments_file}" + python3 - "${comments_file}" <<'PY' + import json + import os + import subprocess + import sys + + try: + from tools.audit_labeler_lib import ( + GITHUB_COMMENT_RESPONSE_BYTES, + GitHubCommentPage, + decode_github_response, + lineage_history, + ) + except ImportError: # pragma: no cover - package fallback + from code_mower.audit_labeler_lib import ( + GITHUB_COMMENT_RESPONSE_BYTES, + GitHubCommentPage, + decode_github_response, + lineage_history, + ) + + def fetch(page, size): + completed = subprocess.run( + ["gh", "api", f"repos/{os.environ['GITHUB_REPOSITORY']}/issues/" + f"{os.environ['PR_NUMBER']}/comments?per_page={size}&page={page}"], + check=False, + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + ) + if completed.returncode != 0: + raise RuntimeError("authenticated GitHub comment request failed") + payload = decode_github_response( + completed.stdout, maximum_bytes=GITHUB_COMMENT_RESPONSE_BYTES) + return GitHubCommentPage(payload, len(completed.stdout)) + + try: + comments = lineage_history(fetch, return_raw=True) + result = [comments] + except Exception as exc: + result = {"code": "bounded_comment_history", "message": str(exc)} + with open(sys.argv[1], "w", encoding="utf-8") as handle: + json.dump(result, handle) + PY gh api --paginate --slurp "repos/${GITHUB_REPOSITORY}/issues/${PR_NUMBER}/timeline?per_page=100" > "${events_file}" gh api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}" > "${pr_file}" if [ -z "${HEAD_SHA:-}" ]; then @@ -351,7 +393,11 @@ jobs: try: history = lineage_core.History.from_pages(comments_payload) except ValueError: - emit("failure", "Code Mower lineage history unreadable") + detail = (comments_payload.get("message") + if isinstance(comments_payload, dict) + and comments_payload.get("code") == "bounded_comment_history" + else "lineage history unreadable") + emit("failure", "Code Mower " + str(detail)) raise SystemExit(0) comments = flatten_paginated_items(comments_payload) events = flatten_paginated_items(events_payload) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7c9dabd8..77b281f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,12 @@ later entries are regular releases. ## Unreleased -No additional changes recorded. +- Audit comment ingestion now uses GitHub-specific response, aggregate-byte, + item, and request-page budgets. Oversized pages reduce `per_page` and restart + safely, complete histories receive a stable reread, and omissions, duplicate + IDs, edits, incomplete terminal proof, and budget exhaustion fail with + actionable diagnostics. Lineage controls are recognized only as standalone + HTML control-comment lines outside inline and fenced examples (#1104). ## 1.5.2 — documentation and repository maintenance diff --git a/src/code_mower/audit_labeler_lib.py b/src/code_mower/audit_labeler_lib.py index f947b17a..c0bfe23b 100644 --- a/src/code_mower/audit_labeler_lib.py +++ b/src/code_mower/audit_labeler_lib.py @@ -38,6 +38,60 @@ import builder_lineage as lineage_core # type: ignore +# GitHub documents a 65,536-character issue-comment body ceiling. Four-byte +# UTF-8 makes one legal body at most 256 KiB before its REST metadata. Keep a +# response comfortably above that single-record maximum, while the separate +# aggregate budgets prevent a large history or adaptive retries from growing +# without bound. +GITHUB_COMMENT_BODY_CHARACTERS = 65_536 +GITHUB_COMMENT_RESPONSE_BYTES = 8 * 1024 * 1024 +GITHUB_COMMENT_HISTORY_BYTES = 64 * 1024 * 1024 +GITHUB_COMMENT_PAGE_SIZE = 100 +GITHUB_COMMENT_HISTORY_ITEMS = 800 + + +class GitHubResponseTooLarge(RuntimeError): + """A bounded GitHub response crossed its explicit transport ceiling.""" + + def __init__(self, response_bytes: int, maximum_bytes: int): + super().__init__(f"GitHub response exceeds {maximum_bytes}-byte budget") + self.response_bytes = response_bytes + self.maximum_bytes = maximum_bytes + + +class CommentHistoryError(lineage_core.ContractError): + """Authenticated GitHub comment history is incomplete or unstable.""" + + +@dataclass(frozen=True) +class GitHubCommentPage: + payload: object + response_bytes: int + + +def decode_github_response(raw: str | bytes, *, maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES) -> Any: + """Decode strict GitHub JSON under a GitHub-specific payload budget. + + This deliberately does not use the 256 KiB private-context decoder. A + valid GitHub comment page can exceed that state-file ceiling. + """ + if type(maximum_bytes) is not int or maximum_bytes <= 0: + raise CommentHistoryError("Invalid GitHub response byte budget") + if not isinstance(raw, (str, bytes)): + raise CommentHistoryError("GitHub response must be JSON text") + encoded = raw.encode("utf-8") if isinstance(raw, str) else raw + if len(encoded) > maximum_bytes: + raise GitHubResponseTooLarge(len(encoded), maximum_bytes) + try: + text = encoded.decode("utf-8") + except UnicodeDecodeError: + raise CommentHistoryError("GitHub response is not valid UTF-8 JSON") from None + try: + return lineage_core._json(text) + except lineage_core.ContractError: + raise CommentHistoryError("GitHub response is not valid unique-key JSON") from None + + def lineage_identity(config): raw = config.get("builder_identity", {}) if not isinstance(raw, dict): @@ -111,23 +165,126 @@ def lineage_snapshot(repo, number, payload): return target, author, labels -def lineage_history(fetch_page, *, page_size=100, max_pages=8): - """Validate each raw page before flattening, including the terminal probe.""" +def lineage_history(fetch_page, *, page_size=GITHUB_COMMENT_PAGE_SIZE, max_pages=8, + max_items=None, max_response_bytes=GITHUB_COMMENT_RESPONSE_BYTES, + max_total_bytes=GITHUB_COMMENT_HISTORY_BYTES, return_raw=False): + """Fetch one complete, stable GitHub comment history within explicit bounds. + + Oversized multi-comment responses halve ``per_page`` and restart at page 1. + Every successful attempt proves a short terminal page. The completed result + is fetched a second time at the accepted page size so duplicate IDs, + omissions, edits, insertions and deletions during pagination fail closed. + """ if type(max_pages) is not int or not 1 <= max_pages <= 8: raise lineage_core.ContractError("Invalid lineage history page budget") if type(page_size) is not int or not 1 <= page_size <= 100: raise lineage_core.ContractError("Invalid lineage history page size") - pages = [] - for page in range(1, max_pages + 2): - raw = fetch_page(page, page_size) - lineage_core.History(raw) - if len(raw) > page_size or (page > max_pages and raw): - raise lineage_core.ContractError("Complete lineage history exceeds page budget") - if page <= max_pages: - pages.append(raw) - if len(raw) < page_size: - return lineage_core.History.from_pages(pages) - raise lineage_core.ContractError("Incomplete lineage history") + item_budget = page_size * max_pages if max_items is None else max_items + if type(item_budget) is not int or item_budget < 1: + raise lineage_core.ContractError("Invalid lineage history item budget") + if (type(max_response_bytes) is not int or max_response_bytes < 1 + or type(max_total_bytes) is not int or max_total_bytes < max_response_bytes): + raise lineage_core.ContractError("Invalid lineage history byte budget") + # Eight data pages historically allowed 800 comments. This request budget + # also covers terminal proofs, one stable reread, and several size restarts. + request_budget = max_pages * 4 + 8 + requests = total_bytes = 0 + observed: dict[int, str] = {} + + def encoded_size(value): + try: + return len(json.dumps(value, ensure_ascii=False, allow_nan=False, + sort_keys=True, separators=(",", ":")).encode("utf-8")) + except (TypeError, ValueError, RecursionError): + raise CommentHistoryError("GitHub comment page is not serializable JSON") from None + + def request(page, size): + nonlocal requests, total_bytes + if requests >= request_budget: + raise CommentHistoryError("GitHub comment history exceeds request-page budget") + requests += 1 + try: + result = fetch_page(page, size) + except GitHubResponseTooLarge as exc: + total_bytes += exc.response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") from None + raise + except CommentHistoryError: + raise + except Exception: + raise CommentHistoryError( + "Authenticated GitHub comment history request failed" + ) from None + if isinstance(result, GitHubCommentPage): + raw, response_bytes = result.payload, result.response_bytes + else: + raw, response_bytes = result, encoded_size(result) + if type(response_bytes) is not int or response_bytes < 0: + raise CommentHistoryError("GitHub comment page has no exact byte count") + if response_bytes > max_response_bytes: + total_bytes += response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") + raise GitHubResponseTooLarge(response_bytes, max_response_bytes) + total_bytes += response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") + return raw + + def attempt(size): + items = [] + fingerprints: dict[int, str] = {} + page = 1 + while True: + raw = request(page, size) + try: + lineage_core.History(raw) + except lineage_core.ContractError: + raise CommentHistoryError("GitHub comment page is unreadable") from None + if len(raw) > size: + raise CommentHistoryError("GitHub comment page exceeds requested item count") + for comment in raw: + comment_id = comment.get("id") + if type(comment_id) is not int or comment_id <= 0: + raise CommentHistoryError("GitHub comment has no stable positive id") + fingerprint = json.dumps(comment, ensure_ascii=False, allow_nan=False, + sort_keys=True, separators=(",", ":")) + if comment_id in fingerprints: + raise CommentHistoryError("GitHub comment history contains duplicate ids") + if comment_id in observed and observed[comment_id] != fingerprint: + raise CommentHistoryError("GitHub comment history changed during pagination") + fingerprints[comment_id] = fingerprint + observed[comment_id] = fingerprint + items.append(comment) + if len(items) > item_budget: + raise CommentHistoryError("GitHub comment history exceeds item budget") + if len(raw) < size: + return items, fingerprints + page += 1 + + accepted_size = page_size + while True: + try: + first, first_fingerprints = attempt(accepted_size) + break + except GitHubResponseTooLarge: + if accepted_size == 1: + raise CommentHistoryError( + "One GitHub comment exceeds the per-response byte budget" + ) from None + accepted_size = max(1, accepted_size // 2) + + if set(observed) - set(first_fingerprints): + raise CommentHistoryError("GitHub comment history omitted ids after adaptive restart") + try: + second, second_fingerprints = attempt(accepted_size) + except GitHubResponseTooLarge: + raise CommentHistoryError("GitHub comment history changed during stable reread") from None + if ([item["id"] for item in first] != [item["id"] for item in second] + or first_fingerprints != second_fingerprints): + raise CommentHistoryError("GitHub comment history changed during stable reread") + return second if return_raw else lineage_core.History(second) def lineage_projection(decision): @@ -408,6 +565,8 @@ def github_request( token: str, body: Optional[Dict[str, Any]] = None, allow_missing: bool = False, + maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes: bool = False, ) -> Any: data = json.dumps(body).encode("utf-8") if body is not None else None request = urllib.request.Request( @@ -423,8 +582,11 @@ def github_request( ) try: with urllib.request.urlopen(request, timeout=30) as response: - response_body = response.read().decode("utf-8") - return lineage_core._json(response_body) + response_body = response.read(maximum_bytes + 1) + payload = decode_github_response(response_body, maximum_bytes=maximum_bytes) + if include_response_bytes: + return GitHubCommentPage(payload, len(response_body)) + return payload except urllib.error.HTTPError as exc: if allow_missing and exc.code == 404: return None @@ -439,6 +601,8 @@ def github_request_with_fallback( tokens: Sequence[GitHubToken], body: Optional[Dict[str, Any]] = None, allow_missing: bool = False, + maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes: bool = False, ) -> Any: """Use the optional PAT first, then fall back to GITHUB_TOKEN on auth errors.""" token_list = tuple(tokens) @@ -451,6 +615,8 @@ def github_request_with_fallback( token=token.value, body=body, allow_missing=allow_missing, + maximum_bytes=maximum_bytes, + include_response_bytes=include_response_bytes, ) except GitHubRequestError as exc: last_error = exc @@ -1078,14 +1244,15 @@ def fetch_issue_comments( tokens: Sequence[GitHubToken], page_cap: int, ) -> list[dict[str, Any]]: - pages = [] def fetch(page, size): - raw = github_request_with_fallback("GET", - f"/repos/{repo}/issues/{issue_number}/comments?per_page={size}&page={page}", tokens=tokens) - pages.append(raw) - return raw - lineage_history(fetch, max_pages=min(page_cap, 8)) - return [comment for page in pages for comment in page] + return github_request_with_fallback( + "GET", + f"/repos/{repo}/issues/{issue_number}/comments?per_page={size}&page={page}", + tokens=tokens, + maximum_bytes=GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes=True, + ) + return lineage_history(fetch, max_pages=min(page_cap, 8), return_raw=True) def apply_label_decision( diff --git a/src/code_mower/builder_lineage.py b/src/code_mower/builder_lineage.py index 64aa60f1..bfb8e636 100644 --- a/src/code_mower/builder_lineage.py +++ b/src/code_mower/builder_lineage.py @@ -21,6 +21,7 @@ MAX_RAW_ARRIVALS = 560 LINEAGE_SCHEMA = "code_mower.builderLineage.v1" LINEAGE_MARKER = "CODE_MOWER_BUILDER_LINEAGE" +LINEAGE_CONTROL_PREFIX = f"", - comment.body, re.DOTALL) + match = re.fullmatch( + re.escape(f"", + controls[0], + ) if match is None: raise ContractError("malformed or unterminated lineage marker") payload = _mapping(_json(match.group(1)), {"schema", "episodes"}) @@ -408,6 +414,36 @@ def _marker_arrivals(history: History, authorities: Authorities) -> Iterable[Map yield from episodes +def lineage_control_comments(body: object) -> tuple[str, ...]: + """Return exact standalone lineage HTML controls outside Markdown fences. + + The marker name is public documentation as well as a reserved control name. + Ordinary prose, inline code and fenced examples therefore cannot announce + lineage. A line beginning with the exact reserved HTML prefix is control + data; once announced by a trusted authority it must parse completely or the + caller fails closed. + """ + if not isinstance(body, str): + return () + controls = [] + fence_character = "" + fence_length = 0 + for raw_line in body.splitlines(): + line = raw_line.strip() + if fence_character: + if re.fullmatch(re.escape(fence_character) + "{" + str(fence_length) + ",}", line): + fence_character, fence_length = "", 0 + continue + fence = re.match(r"^(`{3,}|~{3,})", line) + if fence: + token = fence.group(1) + fence_character, fence_length = token[0], len(token) + continue + if line.startswith(LINEAGE_CONTROL_PREFIX): + controls.append(line) + return tuple(controls) + + def render(chain: Chain) -> str: """Render all episodes of a validated nonempty Chain, without truncation.""" if not isinstance(chain, Chain) or not chain.episodes: diff --git a/src/code_mower/builder_lineage_producer.py b/src/code_mower/builder_lineage_producer.py index 4a06fd4e..c3b79b57 100644 --- a/src/code_mower/builder_lineage_producer.py +++ b/src/code_mower/builder_lineage_producer.py @@ -13,7 +13,7 @@ from .builder_lineage import ( Authorities, Chain, ContractError, Episode, FIRST_EPISODE_KINDS, History, Identity, - LINEAGE_MARKER, Target, parse_markers, render, resolve, + LINEAGE_MARKER, Target, lineage_control_comments, parse_markers, render, resolve, ) from .context_store import ContextStore, strict_json @@ -31,8 +31,11 @@ def _supported(value, kind): """The single compatibility boundary; unsupported history must be re-recorded.""" legacy = False if kind == "history": - legacy = any(LINEAGE_MARKER in c.body and not re.search( - LINEAGE_MARKER + r":", c.body) for c in value.comments) + legacy = any( + any(not control.startswith(f"' duplicate_episode = good.replace('"sequence":1', '"sequence":1,"sequence":1') - invalid = [LINEAGE_MARKER, f"`", + f"```json\n\n```", + f"```json\n```not-a-close\n\n```", + f"~~~~\n\n~~~~", + ) + for text in examples: + with self.subTest(text=text): + self.assertEqual(self.parsed_chain(text).episodes, ()) + def test_marker_target_binding_and_mixed_target_chain(self): good = render(Chain.from_arrivals(target(), [episode()])) for bound in (target(repo="owner/other-repo"), target(pr_number=43), target(branch=BRANCH.lower())): diff --git a/tests/test_provider_runners_github_pr.py b/tests/test_provider_runners_github_pr.py index bff6b352..77dc2ce0 100644 --- a/tests/test_provider_runners_github_pr.py +++ b/tests/test_provider_runners_github_pr.py @@ -1,10 +1,24 @@ import unittest from unittest import mock +import io +import json from code_mower.provider_runners import github_pr class GitHubPrHelperTests(unittest.TestCase): + def test_github_json_decoder_accepts_comment_payload_over_private_state_limit(self) -> None: + comments = [ + {"id": index, "body": "x" * 5_000, "user": {"login": "fixture"}} + for index in range(1, 65) + ] + raw = json.dumps(comments).encode() + self.assertGreater(len(raw), 256 * 1024) + with mock.patch("urllib.request.urlopen", return_value=io.BytesIO(raw)): + self.assertEqual( + github_pr._gh_request("GET", "/fixture", token="ghs_token"), comments + ) + def test_fetch_pull_request_diff_uses_diff_accept(self) -> None: with mock.patch.object(github_pr, "_gh_request", return_value="diff --git a/x b/x") as request: diff = github_pr.fetch_pull_request_diff("owner/repo", 12, token="ghs_token") @@ -87,28 +101,32 @@ def test_fetch_pull_request_files_rejects_non_list_payload(self) -> None: github_pr.fetch_pull_request_files("owner/repo", 12, token="ghs_token") def test_fetch_issue_comments_paginates_until_short_page(self) -> None: - page_one = [{"id": index} for index in range(100)] + page_one = [{"id": index} for index in range(1, 101)] page_two = [{"id": 101}] with mock.patch.object( github_pr, "_gh_request", - side_effect=[page_one, page_two], + side_effect=[page_one, page_two, page_one, page_two], ) as request: comments = github_pr.fetch_issue_comments("owner/repo", 12, token="ghs_token") self.assertEqual(comments, [*page_one, *page_two]) - self.assertEqual(request.call_count, 2) + self.assertEqual(request.call_count, 4) request.assert_has_calls( [ mock.call( "GET", "/repos/owner/repo/issues/12/comments?per_page=100&page=1", token="ghs_token", + maximum_bytes=8388608, + include_response_bytes=True, ), mock.call( "GET", "/repos/owner/repo/issues/12/comments?per_page=100&page=2", token="ghs_token", + maximum_bytes=8388608, + include_response_bytes=True, ), ] ) @@ -118,41 +136,49 @@ def test_fetch_issue_comments_returns_empty_when_first_page_empty(self) -> None: comments = github_pr.fetch_issue_comments("owner/repo", 12, token="ghs_token") self.assertEqual(comments, []) - request.assert_called_once_with( + self.assertEqual(request.call_count, 2) + request.assert_called_with( "GET", "/repos/owner/repo/issues/12/comments?per_page=100&page=1", token="ghs_token", + maximum_bytes=8388608, + include_response_bytes=True, ) def test_fetch_issue_comments_returns_accumulated_comments_on_empty_page(self) -> None: - page_one = [{"id": index} for index in range(100)] + page_one = [{"id": index} for index in range(1, 101)] with mock.patch.object( github_pr, "_gh_request", - side_effect=[page_one, []], + side_effect=[page_one, [], page_one, []], ) as request: comments = github_pr.fetch_issue_comments("owner/repo", 12, token="ghs_token") self.assertEqual(comments, page_one) - self.assertEqual(request.call_count, 2) + self.assertEqual(request.call_count, 4) request.assert_has_calls( [ mock.call( "GET", "/repos/owner/repo/issues/12/comments?per_page=100&page=1", token="ghs_token", + maximum_bytes=8388608, + include_response_bytes=True, ), mock.call( "GET", "/repos/owner/repo/issues/12/comments?per_page=100&page=2", token="ghs_token", + maximum_bytes=8388608, + include_response_bytes=True, ), ] ) def test_fetch_issue_comments_rejects_full_page_cap(self) -> None: - page = [{"id": index} for index in range(100)] - with mock.patch.object(github_pr, "_gh_request", return_value=page): + page = [{"id": index} for index in range(1, 101)] + overflow = [{"id": 101}] + with mock.patch.object(github_pr, "_gh_request", side_effect=[page, overflow]): with self.assertRaisesRegex(RuntimeError, "pagination cap"): github_pr.fetch_issue_comments( "owner/repo", diff --git a/tools/audit_labeler_lib.py b/tools/audit_labeler_lib.py index f947b17a..c0bfe23b 100644 --- a/tools/audit_labeler_lib.py +++ b/tools/audit_labeler_lib.py @@ -38,6 +38,60 @@ import builder_lineage as lineage_core # type: ignore +# GitHub documents a 65,536-character issue-comment body ceiling. Four-byte +# UTF-8 makes one legal body at most 256 KiB before its REST metadata. Keep a +# response comfortably above that single-record maximum, while the separate +# aggregate budgets prevent a large history or adaptive retries from growing +# without bound. +GITHUB_COMMENT_BODY_CHARACTERS = 65_536 +GITHUB_COMMENT_RESPONSE_BYTES = 8 * 1024 * 1024 +GITHUB_COMMENT_HISTORY_BYTES = 64 * 1024 * 1024 +GITHUB_COMMENT_PAGE_SIZE = 100 +GITHUB_COMMENT_HISTORY_ITEMS = 800 + + +class GitHubResponseTooLarge(RuntimeError): + """A bounded GitHub response crossed its explicit transport ceiling.""" + + def __init__(self, response_bytes: int, maximum_bytes: int): + super().__init__(f"GitHub response exceeds {maximum_bytes}-byte budget") + self.response_bytes = response_bytes + self.maximum_bytes = maximum_bytes + + +class CommentHistoryError(lineage_core.ContractError): + """Authenticated GitHub comment history is incomplete or unstable.""" + + +@dataclass(frozen=True) +class GitHubCommentPage: + payload: object + response_bytes: int + + +def decode_github_response(raw: str | bytes, *, maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES) -> Any: + """Decode strict GitHub JSON under a GitHub-specific payload budget. + + This deliberately does not use the 256 KiB private-context decoder. A + valid GitHub comment page can exceed that state-file ceiling. + """ + if type(maximum_bytes) is not int or maximum_bytes <= 0: + raise CommentHistoryError("Invalid GitHub response byte budget") + if not isinstance(raw, (str, bytes)): + raise CommentHistoryError("GitHub response must be JSON text") + encoded = raw.encode("utf-8") if isinstance(raw, str) else raw + if len(encoded) > maximum_bytes: + raise GitHubResponseTooLarge(len(encoded), maximum_bytes) + try: + text = encoded.decode("utf-8") + except UnicodeDecodeError: + raise CommentHistoryError("GitHub response is not valid UTF-8 JSON") from None + try: + return lineage_core._json(text) + except lineage_core.ContractError: + raise CommentHistoryError("GitHub response is not valid unique-key JSON") from None + + def lineage_identity(config): raw = config.get("builder_identity", {}) if not isinstance(raw, dict): @@ -111,23 +165,126 @@ def lineage_snapshot(repo, number, payload): return target, author, labels -def lineage_history(fetch_page, *, page_size=100, max_pages=8): - """Validate each raw page before flattening, including the terminal probe.""" +def lineage_history(fetch_page, *, page_size=GITHUB_COMMENT_PAGE_SIZE, max_pages=8, + max_items=None, max_response_bytes=GITHUB_COMMENT_RESPONSE_BYTES, + max_total_bytes=GITHUB_COMMENT_HISTORY_BYTES, return_raw=False): + """Fetch one complete, stable GitHub comment history within explicit bounds. + + Oversized multi-comment responses halve ``per_page`` and restart at page 1. + Every successful attempt proves a short terminal page. The completed result + is fetched a second time at the accepted page size so duplicate IDs, + omissions, edits, insertions and deletions during pagination fail closed. + """ if type(max_pages) is not int or not 1 <= max_pages <= 8: raise lineage_core.ContractError("Invalid lineage history page budget") if type(page_size) is not int or not 1 <= page_size <= 100: raise lineage_core.ContractError("Invalid lineage history page size") - pages = [] - for page in range(1, max_pages + 2): - raw = fetch_page(page, page_size) - lineage_core.History(raw) - if len(raw) > page_size or (page > max_pages and raw): - raise lineage_core.ContractError("Complete lineage history exceeds page budget") - if page <= max_pages: - pages.append(raw) - if len(raw) < page_size: - return lineage_core.History.from_pages(pages) - raise lineage_core.ContractError("Incomplete lineage history") + item_budget = page_size * max_pages if max_items is None else max_items + if type(item_budget) is not int or item_budget < 1: + raise lineage_core.ContractError("Invalid lineage history item budget") + if (type(max_response_bytes) is not int or max_response_bytes < 1 + or type(max_total_bytes) is not int or max_total_bytes < max_response_bytes): + raise lineage_core.ContractError("Invalid lineage history byte budget") + # Eight data pages historically allowed 800 comments. This request budget + # also covers terminal proofs, one stable reread, and several size restarts. + request_budget = max_pages * 4 + 8 + requests = total_bytes = 0 + observed: dict[int, str] = {} + + def encoded_size(value): + try: + return len(json.dumps(value, ensure_ascii=False, allow_nan=False, + sort_keys=True, separators=(",", ":")).encode("utf-8")) + except (TypeError, ValueError, RecursionError): + raise CommentHistoryError("GitHub comment page is not serializable JSON") from None + + def request(page, size): + nonlocal requests, total_bytes + if requests >= request_budget: + raise CommentHistoryError("GitHub comment history exceeds request-page budget") + requests += 1 + try: + result = fetch_page(page, size) + except GitHubResponseTooLarge as exc: + total_bytes += exc.response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") from None + raise + except CommentHistoryError: + raise + except Exception: + raise CommentHistoryError( + "Authenticated GitHub comment history request failed" + ) from None + if isinstance(result, GitHubCommentPage): + raw, response_bytes = result.payload, result.response_bytes + else: + raw, response_bytes = result, encoded_size(result) + if type(response_bytes) is not int or response_bytes < 0: + raise CommentHistoryError("GitHub comment page has no exact byte count") + if response_bytes > max_response_bytes: + total_bytes += response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") + raise GitHubResponseTooLarge(response_bytes, max_response_bytes) + total_bytes += response_bytes + if total_bytes > max_total_bytes: + raise CommentHistoryError("GitHub comment history exceeds total-byte budget") + return raw + + def attempt(size): + items = [] + fingerprints: dict[int, str] = {} + page = 1 + while True: + raw = request(page, size) + try: + lineage_core.History(raw) + except lineage_core.ContractError: + raise CommentHistoryError("GitHub comment page is unreadable") from None + if len(raw) > size: + raise CommentHistoryError("GitHub comment page exceeds requested item count") + for comment in raw: + comment_id = comment.get("id") + if type(comment_id) is not int or comment_id <= 0: + raise CommentHistoryError("GitHub comment has no stable positive id") + fingerprint = json.dumps(comment, ensure_ascii=False, allow_nan=False, + sort_keys=True, separators=(",", ":")) + if comment_id in fingerprints: + raise CommentHistoryError("GitHub comment history contains duplicate ids") + if comment_id in observed and observed[comment_id] != fingerprint: + raise CommentHistoryError("GitHub comment history changed during pagination") + fingerprints[comment_id] = fingerprint + observed[comment_id] = fingerprint + items.append(comment) + if len(items) > item_budget: + raise CommentHistoryError("GitHub comment history exceeds item budget") + if len(raw) < size: + return items, fingerprints + page += 1 + + accepted_size = page_size + while True: + try: + first, first_fingerprints = attempt(accepted_size) + break + except GitHubResponseTooLarge: + if accepted_size == 1: + raise CommentHistoryError( + "One GitHub comment exceeds the per-response byte budget" + ) from None + accepted_size = max(1, accepted_size // 2) + + if set(observed) - set(first_fingerprints): + raise CommentHistoryError("GitHub comment history omitted ids after adaptive restart") + try: + second, second_fingerprints = attempt(accepted_size) + except GitHubResponseTooLarge: + raise CommentHistoryError("GitHub comment history changed during stable reread") from None + if ([item["id"] for item in first] != [item["id"] for item in second] + or first_fingerprints != second_fingerprints): + raise CommentHistoryError("GitHub comment history changed during stable reread") + return second if return_raw else lineage_core.History(second) def lineage_projection(decision): @@ -408,6 +565,8 @@ def github_request( token: str, body: Optional[Dict[str, Any]] = None, allow_missing: bool = False, + maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes: bool = False, ) -> Any: data = json.dumps(body).encode("utf-8") if body is not None else None request = urllib.request.Request( @@ -423,8 +582,11 @@ def github_request( ) try: with urllib.request.urlopen(request, timeout=30) as response: - response_body = response.read().decode("utf-8") - return lineage_core._json(response_body) + response_body = response.read(maximum_bytes + 1) + payload = decode_github_response(response_body, maximum_bytes=maximum_bytes) + if include_response_bytes: + return GitHubCommentPage(payload, len(response_body)) + return payload except urllib.error.HTTPError as exc: if allow_missing and exc.code == 404: return None @@ -439,6 +601,8 @@ def github_request_with_fallback( tokens: Sequence[GitHubToken], body: Optional[Dict[str, Any]] = None, allow_missing: bool = False, + maximum_bytes: int = GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes: bool = False, ) -> Any: """Use the optional PAT first, then fall back to GITHUB_TOKEN on auth errors.""" token_list = tuple(tokens) @@ -451,6 +615,8 @@ def github_request_with_fallback( token=token.value, body=body, allow_missing=allow_missing, + maximum_bytes=maximum_bytes, + include_response_bytes=include_response_bytes, ) except GitHubRequestError as exc: last_error = exc @@ -1078,14 +1244,15 @@ def fetch_issue_comments( tokens: Sequence[GitHubToken], page_cap: int, ) -> list[dict[str, Any]]: - pages = [] def fetch(page, size): - raw = github_request_with_fallback("GET", - f"/repos/{repo}/issues/{issue_number}/comments?per_page={size}&page={page}", tokens=tokens) - pages.append(raw) - return raw - lineage_history(fetch, max_pages=min(page_cap, 8)) - return [comment for page in pages for comment in page] + return github_request_with_fallback( + "GET", + f"/repos/{repo}/issues/{issue_number}/comments?per_page={size}&page={page}", + tokens=tokens, + maximum_bytes=GITHUB_COMMENT_RESPONSE_BYTES, + include_response_bytes=True, + ) + return lineage_history(fetch, max_pages=min(page_cap, 8), return_raw=True) def apply_label_decision( diff --git a/tools/builder_lineage.py b/tools/builder_lineage.py index 64aa60f1..bfb8e636 100644 --- a/tools/builder_lineage.py +++ b/tools/builder_lineage.py @@ -21,6 +21,7 @@ MAX_RAW_ARRIVALS = 560 LINEAGE_SCHEMA = "code_mower.builderLineage.v1" LINEAGE_MARKER = "CODE_MOWER_BUILDER_LINEAGE" +LINEAGE_CONTROL_PREFIX = f"", - comment.body, re.DOTALL) + match = re.fullmatch( + re.escape(f"", + controls[0], + ) if match is None: raise ContractError("malformed or unterminated lineage marker") payload = _mapping(_json(match.group(1)), {"schema", "episodes"}) @@ -408,6 +414,36 @@ def _marker_arrivals(history: History, authorities: Authorities) -> Iterable[Map yield from episodes +def lineage_control_comments(body: object) -> tuple[str, ...]: + """Return exact standalone lineage HTML controls outside Markdown fences. + + The marker name is public documentation as well as a reserved control name. + Ordinary prose, inline code and fenced examples therefore cannot announce + lineage. A line beginning with the exact reserved HTML prefix is control + data; once announced by a trusted authority it must parse completely or the + caller fails closed. + """ + if not isinstance(body, str): + return () + controls = [] + fence_character = "" + fence_length = 0 + for raw_line in body.splitlines(): + line = raw_line.strip() + if fence_character: + if re.fullmatch(re.escape(fence_character) + "{" + str(fence_length) + ",}", line): + fence_character, fence_length = "", 0 + continue + fence = re.match(r"^(`{3,}|~{3,})", line) + if fence: + token = fence.group(1) + fence_character, fence_length = token[0], len(token) + continue + if line.startswith(LINEAGE_CONTROL_PREFIX): + controls.append(line) + return tuple(controls) + + def render(chain: Chain) -> str: """Render all episodes of a validated nonempty Chain, without truncation.""" if not isinstance(chain, Chain) or not chain.episodes: From 084d64908e6b33ec3960ba93cb848a8cd5d0cb41 Mon Sep 17 00:00:00 2001 From: Jeff Huber Date: Mon, 21 Sep 2026 18:01:51 -0700 Subject: [PATCH 2/4] Preserve transport failures in audit history --- src/code_mower/audit_labeler_lib.py | 4 ---- tests/test_devin_cli_audit_pr.py | 14 ++++++++++---- tests/test_devin_review.py | 5 ++++- tests/test_lane_status.py | 1 + tools/audit_labeler_lib.py | 4 ---- 5 files changed, 15 insertions(+), 13 deletions(-) diff --git a/src/code_mower/audit_labeler_lib.py b/src/code_mower/audit_labeler_lib.py index c0bfe23b..fb218049 100644 --- a/src/code_mower/audit_labeler_lib.py +++ b/src/code_mower/audit_labeler_lib.py @@ -212,10 +212,6 @@ def request(page, size): raise except CommentHistoryError: raise - except Exception: - raise CommentHistoryError( - "Authenticated GitHub comment history request failed" - ) from None if isinstance(result, GitHubCommentPage): raw, response_bytes = result.payload, result.response_bytes else: diff --git a/tests/test_devin_cli_audit_pr.py b/tests/test_devin_cli_audit_pr.py index 08786e37..1cdc0465 100644 --- a/tests/test_devin_cli_audit_pr.py +++ b/tests/test_devin_cli_audit_pr.py @@ -1026,7 +1026,7 @@ def _prior_devin(self, *, include_devin=True): resulting_head=self.head_sha, writer_state='terminated', kind='handoff')] if not include_devin: episodes = [replace(episodes[0], destination_lane='claude', resulting_head=self.head_sha)] - return [{'user': {'login': AUTHORS[0]}, 'body': render(Chain.from_arrivals( + return [{'id': 1, 'user': {'login': AUTHORS[0]}, 'body': render(Chain.from_arrivals( Target('owner/repo', 1, 'codex/topic', self.head_sha), episodes))}] @contextlib.contextmanager @@ -1044,6 +1044,9 @@ def request(method, path, **kwargs): on_history() if isinstance(history, Exception): raise history + if history == '__page_cap__': + page = int(path.rsplit('page=', 1)[1]) + return [{'id': (page - 1) * 100 + item + 1} for item in range(100)] return deepcopy(history) if method == 'POST' and path == '/repos/owner/repo/issues/1/comments': self.assertEqual(len(pr_reads), 2, 'Fresh target must precede the diagnostic effect') @@ -1091,14 +1094,14 @@ def _assert_unknown(self, posts, artifacts, *, legacy=False): def test_public_and_cli_bound_unknown_for_contributor_conflict_and_raw_history(self): from lineage_consumer_fixtures import AUTHORS - raw = [{'user': {'login': AUTHORS[0]}, 'body': ''}] + raw = [{'id': 1, 'user': {'login': AUTHORS[0]}, 'body': ''}] cases = [('contributor', self._snapshot(builder='claude'), self._prior_devin()), ('conflict', self._snapshot(builder='claude'), []), ('marker', self._snapshot(), raw), ('null', self._snapshot(), None), ('object', self._snapshot(), {}), ('mixed', self._snapshot(), [None]), ('unreadable', self._snapshot(), RuntimeError('raw-secret fixture-token')), ('network', self._snapshot(), OSError('raw-secret transport')), - ('cap', self._snapshot(), [{}]*100)] + ('cap', self._snapshot(), '__page_cap__')] for name, pr, history in cases: for cli in (False, True): with self.subTest(case=name, cli=cli), self._boundary(pr, history) as (provider, posts, calls, artifacts): @@ -1112,7 +1115,10 @@ def test_public_and_cli_bound_unknown_for_contributor_conflict_and_raw_history(s provider.assert_not_called() self._assert_unknown(posts, artifacts) reads = [path for method, path in calls if method == 'GET' and '/comments?' in path] - self.assertEqual(len(reads), 9 if name == 'cap' else 1) + expected_reads = 9 if name == 'cap' else 2 if name in { + 'contributor', 'conflict', 'marker' + } else 1 + self.assertEqual(len(reads), expected_reads) def test_public_dry_run_unknown_has_no_effect_and_cli_still_exits_two(self): pr = self._snapshot(builder='claude') diff --git a/tests/test_devin_review.py b/tests/test_devin_review.py index 2a6fc009..0facd0ae 100644 --- a/tests/test_devin_review.py +++ b/tests/test_devin_review.py @@ -584,7 +584,10 @@ def request(*args, requests=requests, mode=mode, **kwargs): requests.append(args) if mode == 'unreadable': raise RuntimeError('authenticated history unavailable') - return [None] if mode == 'malformed' else [{}]*100 + if mode == 'malformed': + return [None] + page = int(args[1].rsplit('page=', 1)[1]) + return [{'id': (page - 1) * 100 + item + 1} for item in range(100)] run = Mock(side_effect=AssertionError('provider must not execute')) count = len(calls) api = Mock(side_effect=AssertionError('provider must not execute')) diff --git a/tests/test_lane_status.py b/tests/test_lane_status.py index bffb1c33..cef45f4c 100644 --- a/tests/test_lane_status.py +++ b/tests/test_lane_status.py @@ -1600,6 +1600,7 @@ def gh_json(args: list[str]) -> object: if args[0] == "api" and "/comments?" in args[1]: return [ { + "id": 1, "user": {"login": "untrusted-user"}, "body": "", "created_at": NOW.isoformat(), diff --git a/tools/audit_labeler_lib.py b/tools/audit_labeler_lib.py index c0bfe23b..fb218049 100644 --- a/tools/audit_labeler_lib.py +++ b/tools/audit_labeler_lib.py @@ -212,10 +212,6 @@ def request(page, size): raise except CommentHistoryError: raise - except Exception: - raise CommentHistoryError( - "Authenticated GitHub comment history request failed" - ) from None if isinstance(result, GitHubCommentPage): raw, response_bytes = result.payload, result.response_bytes else: From d1a19435b1b14eb3b11073221c8acd4f597918cf Mon Sep 17 00:00:00 2001 From: Jeff Huber Date: Mon, 21 Sep 2026 18:27:03 -0700 Subject: [PATCH 3/4] Keep oversized gate-health reads recoverable --- src/code_mower/gate_health.py | 22 +++++++++---- tests/test_gate_health.py | 61 +++++++++++++++++++++++++++++++++++ 2 files changed, 76 insertions(+), 7 deletions(-) diff --git a/src/code_mower/gate_health.py b/src/code_mower/gate_health.py index 5bf0683a..341c0f88 100644 --- a/src/code_mower/gate_health.py +++ b/src/code_mower/gate_health.py @@ -13,12 +13,20 @@ from .audit_labeler_lib import ( GitHubToken, + GitHubResponseTooLarge, decode_github_response, github_actions_comment_attested, lineage_history, ) from .builder_lineage import ContractError +GITHUB_READ_ERRORS = ( + subprocess.CalledProcessError, + ValueError, + ContractError, + GitHubResponseTooLarge, +) + MARKER = "CODE_MOWER_GATE_HEALTH_ALERT" NON_TERMINAL_CHECK_STATUSES = {"queued", "requested", "waiting", "pending", "in_progress"} LOCAL_AUDIT_FAILURE_CONCLUSIONS = { @@ -792,7 +800,7 @@ def enrich_check_runs_with_workflows( "Accept: application/vnd.github+json", ] ) - except (subprocess.CalledProcessError, ValueError) as exc: + except GITHUB_READ_ERRORS as exc: failures.append(f"workflow-run:{run_id}") print(f"warning: failed to fetch workflow run {run_id}: {exc}", flush=True) workflow_run = workflow_runs.get(run_id) or {} @@ -840,7 +848,7 @@ def comment_page(page: int, size: int, if kind == "checks" else items ) - except (subprocess.CalledProcessError, ValueError, ContractError) as exc: + except GITHUB_READ_ERRORS as exc: failures.append(f"{kind}:pr-{number}") print(f"warning: failed to fetch {kind} for PR #{number}: {exc}", flush=True) return out @@ -893,7 +901,7 @@ def fetch_recent_pr_comments( repo, f"issues/comments?per_page=100&since={since}", ) - except (subprocess.CalledProcessError, ValueError) as exc: + except GITHUB_READ_ERRORS as exc: failures.append("comments:recent") print(f"warning: failed to fetch recent issue comments: {exc}", flush=True) return out @@ -912,7 +920,7 @@ def fetch_recent_workflow_runs(repo: str, failures: list[str]) -> list[dict[str, "workflow_runs", paginate=False, ) - except (subprocess.CalledProcessError, ValueError) as exc: + except GITHUB_READ_ERRORS as exc: failures.append("workflow-runs:recent") print(f"warning: failed to fetch recent workflow runs: {exc}", flush=True) return [] @@ -931,7 +939,7 @@ def fetch_workflow_run_jobs( f"actions/runs/{run_id}/jobs?per_page=20", "jobs", ) - except (subprocess.CalledProcessError, ValueError) as exc: + except GITHUB_READ_ERRORS as exc: failures.append(f"workflow-jobs:{run_id}") print(f"warning: failed to fetch workflow jobs for run {run_id}: {exc}", flush=True) return out @@ -965,7 +973,7 @@ def fetch_head_times( ["api", f"repos/{repo}/commits/{sha}", "-H", "Accept: application/vnd.github+json"] ) out[sha] = parse_time(str(data["commit"]["committer"]["date"])) - except (KeyError, subprocess.CalledProcessError, ValueError) as exc: + except (KeyError, *GITHUB_READ_ERRORS) as exc: failures.append(f"head-time:{sha[:12]}") print(f"warning: failed to fetch head time for {sha[:12]}: {exc}", flush=True) return out @@ -1252,7 +1260,7 @@ def main(argv: Sequence[str] | None = None) -> int: if runner_check != "disabled" else [] ) - except subprocess.CalledProcessError as exc: + except GITHUB_READ_ERRORS as exc: if runner_check == "required": print(f"error: required runner check failed: {exc}", flush=True) print( diff --git a/tests/test_gate_health.py b/tests/test_gate_health.py index f2ab5af5..be42f47c 100644 --- a/tests/test_gate_health.py +++ b/tests/test_gate_health.py @@ -17,6 +17,7 @@ recent_alert_keys, workflow_run_jobless_failure_alerts, ) +from code_mower.audit_labeler_lib import GitHubResponseTooLarge NOW = datetime(2026, 8, 17, 5, 0, tzinfo=timezone.utc) SHA = "a" * 40 @@ -730,6 +731,66 @@ def test_fetch_recent_workflow_runs_fetches_one_page(self) -> None: self.assertNotIn("--slurp", calls[0]) self.assertIn("repos/owner/repo/actions/runs?per_page=20", calls[0]) + def test_oversized_github_reads_degrade_through_every_bounded_helper(self) -> None: + import code_mower.gate_health as gate_health + + error = GitHubResponseTooLarge(8 * 1024 * 1024 + 1, 8 * 1024 * 1024) + original_gh_json = gate_health.gh_json + original_gh_api_list = gate_health.gh_api_list + try: + gate_health.gh_json = lambda _args, env=None: (_ for _ in ()).throw(error) + failures: list[str] = [] + enriched = gate_health.enrich_check_runs_with_workflows( + "owner/repo", + [{"details_url": "https://github.com/owner/repo/actions/runs/123/job/456"}], + failures, + ) + self.assertEqual(enriched[0]["run_id"], "123") + self.assertIn("workflow-run:123", failures) + + gate_health.gh_api_list = lambda *_args, **_kwargs: (_ for _ in ()).throw(error) + self.assertEqual( + fetch_per_pr( + "owner/repo", + [{"number": 9, "headRefOid": SHA}], + "checks", + failures, + ), + {}, + ) + self.assertEqual( + gate_health.fetch_recent_pr_comments( + "owner/repo", [{"number": 9}], "2026-08-17T00:00:00Z", failures + ), + {9: []}, + ) + self.assertEqual(fetch_recent_workflow_runs("owner/repo", failures), []) + self.assertEqual( + gate_health.fetch_workflow_run_jobs("owner/repo", ["123"], failures), + {}, + ) + self.assertEqual( + gate_health.fetch_head_times( + "owner/repo", [{"headRefOid": SHA}], failures + ), + {}, + ) + finally: + gate_health.gh_json = original_gh_json + gate_health.gh_api_list = original_gh_api_list + + self.assertEqual( + failures, + [ + "workflow-run:123", + "checks:pr-9", + "comments:recent", + "workflow-runs:recent", + "workflow-jobs:123", + f"head-time:{SHA[:12]}", + ], + ) + def test_fetch_check_runs_uses_pr_numbers_for_duplicate_head_shas(self) -> None: import code_mower.gate_health as gate_health From 0d7967556621b9d3c983b778de47aa23a16e434e Mon Sep 17 00:00:00 2001 From: Jeff Huber Date: Mon, 21 Sep 2026 18:40:55 -0700 Subject: [PATCH 4/4] Preserve lineage publication failure state --- src/code_mower/builder_lineage_producer.py | 3 +- tests/test_lineage_producer_publication.py | 36 +++++++++++++++++++++- 2 files changed, 36 insertions(+), 3 deletions(-) diff --git a/src/code_mower/builder_lineage_producer.py b/src/code_mower/builder_lineage_producer.py index c3b79b57..2b2cb42b 100644 --- a/src/code_mower/builder_lineage_producer.py +++ b/src/code_mower/builder_lineage_producer.py @@ -340,8 +340,7 @@ def snapshot(self, target): return Snapshot(observed, raw["user"]["login"], tuple(item["name"] for item in labels)) def history(self, target): - from .audit_labeler_lib import lineage_history - return lineage_history(lambda page, size: self._json( + return fetch_history(lambda page, size: self._json( f"repos/{target.repo}/issues/{target.pr_number}/comments?per_page={size}&page={page}", include_response_bytes=True)) diff --git a/tests/test_lineage_producer_publication.py b/tests/test_lineage_producer_publication.py index bdbcbd63..e7600d1e 100644 --- a/tests/test_lineage_producer_publication.py +++ b/tests/test_lineage_producer_publication.py @@ -1,11 +1,15 @@ """Raw ingress, shared bounds, public readback and exact effect ordering.""" from pathlib import Path +import subprocess import tempfile import unittest +from unittest import mock +from code_mower.audit_labeler_lib import GitHubCommentPage from code_mower.builder_lineage import Authorities, ContractError, History, Identity from code_mower.builder_lineage_producer import ( - ProducerRefusal, Snapshot, decode_transport, fetch_history, observe, publish, selected_history, + GitHub, ProducerRefusal, Snapshot, decode_transport, fetch_history, observe, publish, + selected_history, ) from code_mower.builder_runs import record_lineage_builder from lineage_producer_fixtures import ( @@ -96,6 +100,36 @@ def test_post_then_missing_untrusted_conflicting_or_malformed_readback(self): self.assertFalse(caught.exception.labels_attempted) self.assertEqual(io.effects, ["snapshot", "history", "snapshot", "post", "history"]) + def test_initial_github_history_transport_failure_has_no_partial_effect(self): + io = GitHub() + effects = GitHubIO() + io.snapshot = effects.snapshot + io.post = effects.post + io.labels = effects.labels + failure = subprocess.CalledProcessError(1, ["gh", "api"]) + with mock.patch.object(io, "_json", side_effect=failure): + with self.assertRaisesRegex( + ProducerRefusal, "Authenticated history request failed") as caught: + self.publish(io) + self.assertFalse(caught.exception.comment_posted) + self.assertFalse(caught.exception.labels_attempted) + self.assertEqual(effects.effects, ["snapshot"]) + + def test_post_then_github_history_transport_failure_preserves_partial_effect(self): + io = GitHub() + effects = GitHubIO() + io.snapshot = effects.snapshot + io.post = effects.post + io.labels = effects.labels + failure = subprocess.CalledProcessError(1, ["gh", "api"]) + pages = [GitHubCommentPage([], 2), GitHubCommentPage([], 2), failure] + with mock.patch.object(io, "_json", side_effect=pages): + with self.assertRaises(ProducerRefusal) as caught: + self.publish(io) + self.assertTrue(caught.exception.comment_posted) + self.assertFalse(caught.exception.labels_attempted) + self.assertEqual(effects.effects, ["snapshot", "snapshot", "post"]) + def test_public_only_missing_final_control(self): io = GitHubIO(2) io.public = comments([episode()])