diff --git a/kun/src/memory/fixtures/kun-memory-feedback-tiebreaker-v1.json b/kun/src/memory/fixtures/kun-memory-feedback-tiebreaker-v1.json new file mode 100644 index 000000000..dade7a4a6 --- /dev/null +++ b/kun/src/memory/fixtures/kun-memory-feedback-tiebreaker-v1.json @@ -0,0 +1,103 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "decisionId": "kun-memory-feedback-tiebreaker-v1", + "status": "final", + "artifactHashes": { + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "manifestSha256": "17ab0ea2186d11bbdfb38e0f76b75f3547144c74412183e8118cb2e525d6124a", + "calibrationSha256": "b34b6d995a2230aac543885d4173c07f8e3024b4fd8603637d5f49e4258a7964", + "decisionPlanSha256": "27dfc5852afa9d86e0ff52bf549fb154e5b788ac00690758c0c8accd15460ab7", + "candidateLockSha256": "03716e77bd081515b098e1844e3bb3e4121194fc2323df8cb74cf5ee9c9cfe8f" + }, + "selectedCandidateId": "foundation-control", + "holdoutRunCount": 1, + "development": { + "foundation": { + "pairAccuracy": 0, + "pairCount": 4, + "recallAtK": 0.4375, + "precisionAtK": 0.4375, + "meanReciprocalRank": 0.4375, + "abstentionAccuracy": 1, + "explicitForbiddenSelections": 6, + "authorizationOrLifecycleViolations": 0, + "rankedCaseCount": 16, + "noResultCaseCount": 2, + "caseCount": 18 + }, + "candidate": { + "pairAccuracy": 0, + "pairCount": 4, + "recallAtK": 0.4375, + "precisionAtK": 0.4375, + "meanReciprocalRank": 0.4375, + "abstentionAccuracy": 1, + "explicitForbiddenSelections": 6, + "authorizationOrLifecycleViolations": 0, + "rankedCaseCount": 16, + "noResultCaseCount": 2, + "caseCount": 18 + }, + "bootstrapLowerBounds": { + "pairAccuracyGain": 0, + "recallGain": 0, + "mrrGain": 0 + }, + "gates": { + "localBenefit": false, + "globalNonRegression": true, + "safety": false, + "privacy": true, + "determinism": true, + "resource": true, + "passed": false + } + }, + "holdout": { + "foundation": { + "pairAccuracy": 0, + "pairCount": 4, + "recallAtK": 0.25, + "precisionAtK": 0.25, + "meanReciprocalRank": 0.25, + "abstentionAccuracy": 1, + "explicitForbiddenSelections": 1, + "authorizationOrLifecycleViolations": 0, + "rankedCaseCount": 16, + "noResultCaseCount": 2, + "caseCount": 18 + }, + "candidate": { + "pairAccuracy": 0, + "pairCount": 4, + "recallAtK": 0.25, + "precisionAtK": 0.25, + "meanReciprocalRank": 0.25, + "abstentionAccuracy": 1, + "explicitForbiddenSelections": 1, + "authorizationOrLifecycleViolations": 0, + "rankedCaseCount": 16, + "noResultCaseCount": 2, + "caseCount": 18 + }, + "bootstrapLowerBounds": { + "pairAccuracyGain": 0, + "recallGain": 0, + "mrrGain": 0 + }, + "gates": { + "localBenefit": false, + "globalNonRegression": true, + "safety": false, + "privacy": true, + "determinism": true, + "resource": true, + "passed": false + } + }, + "decision": "no-go", + "reasons": [ + "foundation-control is the pre-registered fallback after all development candidates failed the safety gate." + ] +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-calibration.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-calibration.v1.json new file mode 100644 index 000000000..8cc08b0c3 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-calibration.v1.json @@ -0,0 +1,34 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "foundationVersion": "post-1308-foundation-v1", + "procedure": { + "source": "development-foundation-only", + "gapFormula": "higher-score-minus-lower-score", + "quantiles": [0.25, 0.5, 0.75], + "quantileMethod": "floor-index", + "maximumGridSize": 4, + "boundaryInclusive": true, + "grouping": "leader-relative", + "maximumReorderWindow": 3, + "exactTieControl": true, + "noRerankControl": true, + "stableFallback": ["foundation-score", "memory-id"] + }, + "observedGaps": [ + {"caseId":"case_dev_confirmed_01","higherId":"mem_tie_list","lowerId":"mem_tie_table","gap":0.00375}, + {"caseId":"case_dev_corrected_01","higherId":"mem_corr_distractor","lowerId":"mem_corr_new","gap":0.0075}, + {"caseId":"case_dev_corrected_02","higherId":"mem_corr_distractor","lowerId":"mem_corr_new","gap":0.0075}, + {"caseId":"case_dev_frequent_01","higherId":"mem_freq_wrong","lowerId":"mem_freq_right","gap":0.0075}, + {"caseId":"case_dev_misleading_01","higherId":"mem_misleading_confirmed","lowerId":"mem_misleading_expected","gap":0.01125}, + {"caseId":"case_dev_misleading_02","higherId":"mem_misleading_confirmed","lowerId":"mem_misleading_expected","gap":0.01125}, + {"caseId":"case_dev_neutral_01","higherId":"mem_neutral_blue","lowerId":"mem_neutral_green","gap":0.0075}, + {"caseId":"case_dev_wide_01","higherId":"mem_wide_primary","lowerId":"mem_wide_confirmed","gap":0.2} + ], + "boundaryGrid": [ + {"id":"gap_0","maximumGap":0,"inclusive":true}, + {"id":"gap_1","maximumGap":0.0075,"inclusive":true}, + {"id":"gap_2","maximumGap":0.01125,"inclusive":true} + ] +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-checksums.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-checksums.v1.json new file mode 100644 index 000000000..39721a389 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-checksums.v1.json @@ -0,0 +1,9 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "algorithm": "sha256", + "files": { + "fixture": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "manifest": "17ab0ea2186d11bbdfb38e0f76b75f3547144c74412183e8118cb2e525d6124a" + } +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-development.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-development.v1.json new file mode 100644 index 000000000..52a9cc689 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-development.v1.json @@ -0,0 +1,67 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "partition": "development", + "artifactHashes": { + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "manifestSha256": "17ab0ea2186d11bbdfb38e0f76b75f3547144c74412183e8118cb2e525d6124a", + "calibrationSha256": "b34b6d995a2230aac543885d4173c07f8e3024b4fd8603637d5f49e4258a7964", + "decisionPlanSha256": "27dfc5852afa9d86e0ff52bf549fb154e5b788ac00690758c0c8accd15460ab7" + }, + "foundation": { + "pairAccuracy":0,"pairCount":4,"recallAtK":0.4375,"precisionAtK":0.4375, + "meanReciprocalRank":0.4375,"abstentionAccuracy":1,"explicitForbiddenSelections":6, + "authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18 + }, + "configurations": [ + { + "candidateId":"foundation-control","boundaryId":null,"maximumGap":0,"signalRule":"foundation-only", + "metrics":{"pairAccuracy":0,"pairCount":4,"recallAtK":0.4375,"precisionAtK":0.4375,"meanReciprocalRank":0.4375,"abstentionAccuracy":1,"explicitForbiddenSelections":6,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0,"recallGain":0,"mrrGain":0}, + "gates":{"localBenefit":false,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"867715da1bd988bdf02bc50eef39d6440e7d633ee12f5018056a386402455540" + }, + { + "candidateId":"confirmation-gap_0","boundaryId":"gap_0","maximumGap":0,"signalRule":"confirmation", + "metrics":{"pairAccuracy":0,"pairCount":4,"recallAtK":0.4375,"precisionAtK":0.4375,"meanReciprocalRank":0.4375,"abstentionAccuracy":1,"explicitForbiddenSelections":6,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0,"recallGain":0,"mrrGain":0}, + "gates":{"localBenefit":false,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"867715da1bd988bdf02bc50eef39d6440e7d633ee12f5018056a386402455540" + }, + { + "candidateId":"confirmation-correction-gap_0","boundaryId":"gap_0","maximumGap":0,"signalRule":"confirmation-correction", + "metrics":{"pairAccuracy":0,"pairCount":4,"recallAtK":0.4375,"precisionAtK":0.4375,"meanReciprocalRank":0.4375,"abstentionAccuracy":1,"explicitForbiddenSelections":6,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0,"recallGain":0,"mrrGain":0}, + "gates":{"localBenefit":false,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"867715da1bd988bdf02bc50eef39d6440e7d633ee12f5018056a386402455540" + }, + { + "candidateId":"confirmation-gap_1","boundaryId":"gap_1","maximumGap":0.0075,"signalRule":"confirmation", + "metrics":{"pairAccuracy":0.25,"pairCount":4,"recallAtK":0.5,"precisionAtK":0.5,"meanReciprocalRank":0.5,"abstentionAccuracy":1,"explicitForbiddenSelections":5,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0,"recallGain":0,"mrrGain":0}, + "gates":{"localBenefit":false,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"839628f204c80ed5e5b39ee20066e70b92126dfe08c5e934890eb1aefa19c1fc" + }, + { + "candidateId":"confirmation-correction-gap_1","boundaryId":"gap_1","maximumGap":0.0075,"signalRule":"confirmation-correction", + "metrics":{"pairAccuracy":0.75,"pairCount":4,"recallAtK":0.625,"precisionAtK":0.625,"meanReciprocalRank":0.625,"abstentionAccuracy":1,"explicitForbiddenSelections":3,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0.25,"recallGain":0.0625,"mrrGain":0.0625}, + "gates":{"localBenefit":true,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"3e386fe6b11bdc8c7dd68c43e210533bdc70557f25901c168c3d2ca78f5fbc57" + }, + { + "candidateId":"confirmation-gap_2","boundaryId":"gap_2","maximumGap":0.01125,"signalRule":"confirmation", + "metrics":{"pairAccuracy":0.25,"pairCount":4,"recallAtK":0.5,"precisionAtK":0.5,"meanReciprocalRank":0.5,"abstentionAccuracy":1,"explicitForbiddenSelections":5,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0,"recallGain":0,"mrrGain":0}, + "gates":{"localBenefit":false,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"839628f204c80ed5e5b39ee20066e70b92126dfe08c5e934890eb1aefa19c1fc" + }, + { + "candidateId":"confirmation-correction-gap_2","boundaryId":"gap_2","maximumGap":0.01125,"signalRule":"confirmation-correction", + "metrics":{"pairAccuracy":0.75,"pairCount":4,"recallAtK":0.625,"precisionAtK":0.625,"meanReciprocalRank":0.625,"abstentionAccuracy":1,"explicitForbiddenSelections":3,"authorizationOrLifecycleViolations":0,"rankedCaseCount":16,"noResultCaseCount":2,"caseCount":18}, + "bootstrapLowerBounds":{"pairAccuracyGain":0.25,"recallGain":0.0625,"mrrGain":0.0625}, + "gates":{"localBenefit":true,"globalNonRegression":true,"safety":false,"passed":false}, + "selectedIdsSha256":"3e386fe6b11bdc8c7dd68c43e210533bdc70557f25901c168c3d2ca78f5fbc57" + } + ] +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-fixtures.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-fixtures.v1.json new file mode 100644 index 000000000..a911e5081 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-fixtures.v1.json @@ -0,0 +1,80 @@ +{ + "schemaVersion": 1, + "datasetId": "kun-memory-feedback-tiebreaker-anonymous-v1", + "status": "frozen", + "evaluationNow": "2026-09-16T00:00:00.000Z", + "records": [ + {"schemaVersion":2,"id":"mem_tie_table","content":"Use compact tables for deployment status reports.","scope":"workspace","workspace":"fixture-workspace-a","tags":["deployment-status"],"confidence":0.8,"type":"preference","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_tie_list","content":"Use compact lists for deployment status reports.","scope":"workspace","workspace":"fixture-workspace-a","tags":["deployment-status"],"confidence":0.8,"type":"preference","authority":"reference","importance":0.55,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-02T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_corr_old","content":"Use a twenty second service timeout for batch requests.","scope":"workspace","workspace":"fixture-workspace-a","tags":["service-timeout"],"confidence":0.9,"type":"fact","authority":"reference","importance":0.6,"observedAt":"2026-07-01T00:00:00.000Z","createdAt":"2026-07-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","supersededAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_corr_new","content":"Use a forty five second service timeout for batch requests.","scope":"workspace","workspace":"fixture-workspace-a","tags":["service-timeout"],"confidence":0.9,"type":"fact","authority":"reference","importance":0.6,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","supersedes":"mem_corr_old","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_corr_distractor","content":"Use a forty second service timeout for batch requests.","scope":"workspace","workspace":"fixture-workspace-a","tags":["service-timeout"],"confidence":0.9,"type":"fact","authority":"reference","importance":0.7,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-03T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_freq_wrong","content":"Run package lint before publishing a desktop build.","scope":"workspace","workspace":"fixture-workspace-a","tags":["package-check"],"confidence":0.8,"type":"fact","authority":"reference","importance":0.6,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-03T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_freq_right","content":"Run package validation before publishing a desktop build.","scope":"workspace","workspace":"fixture-workspace-a","tags":["package-check"],"confidence":0.8,"type":"fact","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_neutral_blue","content":"Use a blue label for preview releases.","scope":"workspace","workspace":"fixture-workspace-a","tags":["preview-label"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.6,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-03T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_neutral_green","content":"Use a green label for preview releases.","scope":"workspace","workspace":"fixture-workspace-a","tags":["preview-label"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_wide_primary","content":"Archive completed reports after ninety days.","scope":"workspace","workspace":"fixture-workspace-a","tags":["report-retention"],"confidence":0.9,"type":"decision","authority":"reference","importance":0.7,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_wide_confirmed","content":"Archive temporary images after seven days.","scope":"workspace","workspace":"fixture-workspace-a","tags":["image-retention"],"confidence":0.9,"type":"decision","authority":"reference","importance":0.7,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_noise_keyboard","content":"Use quiet keyboard switches in the shared lab.","scope":"workspace","workspace":"fixture-workspace-a","tags":["lab-equipment"],"confidence":0.8,"type":"preference","authority":"reference","importance":0.4,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_noise_zoom","content":"Keep diagram zoom at one hundred percent.","scope":"workspace","workspace":"fixture-workspace-a","tags":["diagram-view"],"confidence":0.8,"type":"preference","authority":"reference","importance":0.4,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_scope_other","content":"Use an amber badge for staging incidents.","scope":"workspace","workspace":"fixture-workspace-b","tags":["staging-badge"],"confidence":1,"type":"decision","authority":"reference","importance":0.8,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_scope_local","content":"Use an amber badge for staging notices.","scope":"workspace","workspace":"fixture-workspace-a","tags":["staging-badge"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_disabled_timeline","content":"Enable a compact timeline for incident review.","scope":"workspace","workspace":"fixture-workspace-a","tags":["incident-timeline"],"confidence":1,"type":"decision","authority":"reference","importance":0.8,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-05T00:00:00.000Z","disabledAt":"2026-08-10T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_active_timeline","content":"Enable a compact timeline for incident summaries.","scope":"workspace","workspace":"fixture-workspace-a","tags":["incident-timeline"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_misleading_expected","content":"Use three review passes for release notes.","scope":"workspace","workspace":"fixture-workspace-a","tags":["release-review"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.5,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-01T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]}, + {"schemaVersion":2,"id":"mem_misleading_confirmed","content":"Use two review passes for release notes.","scope":"workspace","workspace":"fixture-workspace-a","tags":["release-review"],"confidence":0.8,"type":"decision","authority":"reference","importance":0.65,"observedAt":"2026-08-01T00:00:00.000Z","createdAt":"2026-08-01T00:00:00.000Z","updatedAt":"2026-08-02T00:00:00.000Z","sources":[{"id":"fixture-source","kind":"user","trust":"explicit-user"}]} + ], + "events": [ + {"schemaVersion":1,"id":"evt_tie_confirmed","kind":"confirmed","memoryId":"mem_tie_table","occurredAt":"2026-09-01T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_corrected","kind":"corrected","memoryId":"mem_corr_old","replacementMemoryId":"mem_corr_new","occurredAt":"2026-09-02T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_freq_01","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-01T00:00:00.000Z","threadId":"thread-fixture-1","turnId":"turn-fixture-1"}, + {"schemaVersion":1,"id":"evt_freq_02","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-02T00:00:00.000Z","threadId":"thread-fixture-1","turnId":"turn-fixture-2"}, + {"schemaVersion":1,"id":"evt_freq_03","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-03T00:00:00.000Z","threadId":"thread-fixture-1","turnId":"turn-fixture-3"}, + {"schemaVersion":1,"id":"evt_freq_04","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-04T00:00:00.000Z","threadId":"thread-fixture-2","turnId":"turn-fixture-4"}, + {"schemaVersion":1,"id":"evt_freq_05","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-05T00:00:00.000Z","threadId":"thread-fixture-2","turnId":"turn-fixture-5"}, + {"schemaVersion":1,"id":"evt_freq_06","kind":"retrieved","memoryId":"mem_freq_wrong","occurredAt":"2026-09-06T00:00:00.000Z","threadId":"thread-fixture-2","turnId":"turn-fixture-6"}, + {"schemaVersion":1,"id":"evt_wide_confirmed","kind":"confirmed","memoryId":"mem_wide_confirmed","occurredAt":"2026-09-03T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_noise_confirmed","kind":"confirmed","memoryId":"mem_noise_keyboard","occurredAt":"2026-09-04T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_scope_confirmed","kind":"confirmed","memoryId":"mem_scope_other","occurredAt":"2026-09-05T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_disabled_confirmed","kind":"confirmed","memoryId":"mem_disabled_timeline","occurredAt":"2026-09-06T00:00:00.000Z"}, + {"schemaVersion":1,"id":"evt_misleading_confirmed","kind":"confirmed","memoryId":"mem_misleading_confirmed","occurredAt":"2026-09-07T00:00:00.000Z"} + ], + "cases": [ + {"id":"case_dev_confirmed_01","partition":"development","stratum":"confirmed-near-tie","query":"Which compact structure should deployment status reports use?","workspace":"fixture-workspace-a","candidateIds":["mem_tie_table","mem_tie_list"],"expectedIds":["mem_tie_table"],"forbiddenIds":["mem_tie_list"],"preferredPairs":[{"preferredId":"mem_tie_table","otherId":"mem_tie_list","rationale":"The explicit confirmation identifies tables as the retained preference."}],"rationale":"Tests confirmation inside an otherwise symmetric lexical pair.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_confirmed_02","partition":"development","stratum":"confirmed-near-tie","query":"How should concise rollout progress be formatted?","workspace":"fixture-workspace-a","candidateIds":["mem_tie_table","mem_tie_list"],"expectedIds":["mem_tie_table"],"forbiddenIds":["mem_tie_list"],"preferredPairs":[{"preferredId":"mem_tie_table","otherId":"mem_tie_list","rationale":"The confirmed table preference is the expected answer."}],"rationale":"Second development phrasing for confirmation-aware ordering.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_confirmed_01","partition":"holdout","stratum":"confirmed-near-tie","query":"Select the approved layout for brief deployment summaries.","workspace":"fixture-workspace-a","candidateIds":["mem_tie_table","mem_tie_list"],"expectedIds":["mem_tie_table"],"forbiddenIds":["mem_tie_list"],"preferredPairs":[{"preferredId":"mem_tie_table","otherId":"mem_tie_list","rationale":"The explicit confirmation favors the table layout."}],"rationale":"Holdout wording avoids the development query vocabulary.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_confirmed_02","partition":"holdout","stratum":"confirmed-near-tie","query":"What presentation was retained for short release status updates?","workspace":"fixture-workspace-a","candidateIds":["mem_tie_table","mem_tie_list"],"expectedIds":["mem_tie_table"],"forbiddenIds":["mem_tie_list"],"preferredPairs":[{"preferredId":"mem_tie_table","otherId":"mem_tie_list","rationale":"Confirmed tables are preferred over the newer distractor."}],"rationale":"Independent holdout expression of the same product preference.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_corrected_01","partition":"development","stratum":"corrected-replacement-near-tie","query":"What service timeout applies to batch requests?","workspace":"fixture-workspace-a","candidateIds":["mem_corr_old","mem_corr_new","mem_corr_distractor"],"expectedIds":["mem_corr_new"],"forbiddenIds":["mem_corr_old","mem_corr_distractor"],"preferredPairs":[{"preferredId":"mem_corr_new","otherId":"mem_corr_distractor","rationale":"The correction event identifies the active forty five second replacement."}],"rationale":"Tests explicit correction evidence between close active candidates.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_corrected_02","partition":"development","stratum":"corrected-replacement-near-tie","query":"Recall the current batch request service limit.","workspace":"fixture-workspace-a","candidateIds":["mem_corr_old","mem_corr_new","mem_corr_distractor"],"expectedIds":["mem_corr_new"],"forbiddenIds":["mem_corr_old","mem_corr_distractor"],"preferredPairs":[{"preferredId":"mem_corr_new","otherId":"mem_corr_distractor","rationale":"The corrected replacement is the current value."}],"rationale":"Alternative development query for correction-aware ordering.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_corrected_01","partition":"holdout","stratum":"corrected-replacement-near-tie","query":"Which duration now governs batched service calls?","workspace":"fixture-workspace-a","candidateIds":["mem_corr_old","mem_corr_new","mem_corr_distractor"],"expectedIds":["mem_corr_new"],"forbiddenIds":["mem_corr_old","mem_corr_distractor"],"preferredPairs":[{"preferredId":"mem_corr_new","otherId":"mem_corr_distractor","rationale":"The versioned correction makes the replacement current."}],"rationale":"Holdout paraphrase for corrected replacement selection.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_corrected_02","partition":"holdout","stratum":"corrected-replacement-near-tie","query":"Find the revised timeout used by grouped requests.","workspace":"fixture-workspace-a","candidateIds":["mem_corr_old","mem_corr_new","mem_corr_distractor"],"expectedIds":["mem_corr_new"],"forbiddenIds":["mem_corr_old","mem_corr_distractor"],"preferredPairs":[{"preferredId":"mem_corr_new","otherId":"mem_corr_distractor","rationale":"Only the explicit replacement carries correction evidence."}],"rationale":"Second holdout wording for the revision scenario.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_frequent_01","partition":"development","stratum":"frequent-unconfirmed-control","query":"Which package check runs before publishing a desktop build?","workspace":"fixture-workspace-a","candidateIds":["mem_freq_wrong","mem_freq_right"],"expectedIds":["mem_freq_right"],"forbiddenIds":["mem_freq_wrong"],"preferredPairs":[{"preferredId":"mem_freq_right","otherId":"mem_freq_wrong","rationale":"Retrieval frequency is not evidence that lint is the intended validation step."}],"rationale":"Confirms repeated exposure cannot alter the comparator.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_frequent_02","partition":"development","stratum":"frequent-unconfirmed-control","query":"Name the package verification performed ahead of desktop publication.","workspace":"fixture-workspace-a","candidateIds":["mem_freq_wrong","mem_freq_right"],"expectedIds":["mem_freq_right"],"forbiddenIds":["mem_freq_wrong"],"preferredPairs":[{"preferredId":"mem_freq_right","otherId":"mem_freq_wrong","rationale":"Only semantic relevance, not impressions, supports validation."}],"rationale":"Development control with a high-frequency distractor.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_frequent_01","partition":"holdout","stratum":"frequent-unconfirmed-control","query":"What verification precedes shipment of the desktop package?","workspace":"fixture-workspace-a","candidateIds":["mem_freq_wrong","mem_freq_right"],"expectedIds":["mem_freq_right"],"forbiddenIds":["mem_freq_wrong"],"preferredPairs":[{"preferredId":"mem_freq_right","otherId":"mem_freq_wrong","rationale":"Six retrievals must not convert lint into confirmation."}],"rationale":"Holdout frequency control with different terminology.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_frequent_02","partition":"holdout","stratum":"frequent-unconfirmed-control","query":"Recall the required check before the desktop artifact is released.","workspace":"fixture-workspace-a","candidateIds":["mem_freq_wrong","mem_freq_right"],"expectedIds":["mem_freq_right"],"forbiddenIds":["mem_freq_wrong"],"preferredPairs":[{"preferredId":"mem_freq_right","otherId":"mem_freq_wrong","rationale":"The popular distractor has no explicit user evidence."}],"rationale":"Second holdout frequency-only negative.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_neutral_01","partition":"development","stratum":"no-feedback-control","query":"Which color label was chosen for preview releases?","workspace":"fixture-workspace-a","candidateIds":["mem_neutral_blue","mem_neutral_green"],"expectedIds":["mem_neutral_blue"],"forbiddenIds":["mem_neutral_green"],"preferredPairs":[{"preferredId":"mem_neutral_blue","otherId":"mem_neutral_green","rationale":"The synthetic decision labels blue as current; neither record has feedback."}],"rationale":"No-feedback cases must preserve foundation ordering.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_neutral_02","partition":"development","stratum":"no-feedback-control","query":"Recall the selected preview release marker color.","workspace":"fixture-workspace-a","candidateIds":["mem_neutral_blue","mem_neutral_green"],"expectedIds":["mem_neutral_blue"],"forbiddenIds":["mem_neutral_green"],"preferredPairs":[{"preferredId":"mem_neutral_blue","otherId":"mem_neutral_green","rationale":"No feedback exists to justify changing the baseline choice."}],"rationale":"Second development stability control.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_neutral_01","partition":"holdout","stratum":"no-feedback-control","query":"What hue identifies a pre-release preview?","workspace":"fixture-workspace-a","candidateIds":["mem_neutral_blue","mem_neutral_green"],"expectedIds":["mem_neutral_blue"],"forbiddenIds":["mem_neutral_green"],"preferredPairs":[{"preferredId":"mem_neutral_blue","otherId":"mem_neutral_green","rationale":"Without explicit evidence the candidate must retain foundation order."}],"rationale":"Holdout no-feedback control using a distinct phrasing.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_neutral_02","partition":"holdout","stratum":"no-feedback-control","query":"Find the decision about the preview badge color.","workspace":"fixture-workspace-a","candidateIds":["mem_neutral_blue","mem_neutral_green"],"expectedIds":["mem_neutral_blue"],"forbiddenIds":["mem_neutral_green"],"preferredPairs":[{"preferredId":"mem_neutral_blue","otherId":"mem_neutral_green","rationale":"The newer foundation record is the labeled synthetic decision."}],"rationale":"Second holdout baseline-preservation case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_wide_01","partition":"development","stratum":"wide-gap-control","query":"After how many days are completed reports archived?","workspace":"fixture-workspace-a","candidateIds":["mem_wide_primary","mem_wide_confirmed"],"expectedIds":["mem_wide_primary"],"forbiddenIds":["mem_wide_confirmed"],"preferredPairs":[{"preferredId":"mem_wide_primary","otherId":"mem_wide_confirmed","rationale":"The confirmed image policy is irrelevant to completed reports."}],"rationale":"Explicit feedback must not cross a clear foundation gap.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_wide_02","partition":"development","stratum":"wide-gap-control","query":"State the retention period for finished reporting artifacts.","workspace":"fixture-workspace-a","candidateIds":["mem_wide_primary","mem_wide_confirmed"],"expectedIds":["mem_wide_primary"],"forbiddenIds":["mem_wide_confirmed"],"preferredPairs":[{"preferredId":"mem_wide_primary","otherId":"mem_wide_confirmed","rationale":"Report retention has stronger lexical evidence than image retention."}],"rationale":"Second development wide-gap control.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_wide_01","partition":"holdout","stratum":"wide-gap-control","query":"When should finalized reports move into the archive?","workspace":"fixture-workspace-a","candidateIds":["mem_wide_primary","mem_wide_confirmed"],"expectedIds":["mem_wide_primary"],"forbiddenIds":["mem_wide_confirmed"],"preferredPairs":[{"preferredId":"mem_wide_primary","otherId":"mem_wide_confirmed","rationale":"Image confirmation cannot displace a report-specific match."}],"rationale":"Holdout wide-gap protection case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_wide_02","partition":"holdout","stratum":"wide-gap-control","query":"Retrieve the archive rule for completed report material.","workspace":"fixture-workspace-a","candidateIds":["mem_wide_primary","mem_wide_confirmed"],"expectedIds":["mem_wide_primary"],"forbiddenIds":["mem_wide_confirmed"],"preferredPairs":[{"preferredId":"mem_wide_primary","otherId":"mem_wide_confirmed","rationale":"The unrelated confirmed record must remain below the direct match."}],"rationale":"Second holdout wide-gap protection case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_empty_01","partition":"development","stratum":"no-result-control","query":"Which lunch menu is scheduled for the studio?","workspace":"fixture-workspace-a","candidateIds":["mem_noise_keyboard","mem_noise_zoom"],"expectedIds":[],"forbiddenIds":["mem_noise_keyboard","mem_noise_zoom"],"preferredPairs":[],"rationale":"No candidate contains positive evidence about meals or schedules.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_empty_02","partition":"development","stratum":"no-result-control","query":"What bicycle route was approved for the weekend?","workspace":"fixture-workspace-a","candidateIds":["mem_noise_keyboard","mem_noise_zoom"],"expectedIds":[],"forbiddenIds":["mem_noise_keyboard","mem_noise_zoom"],"preferredPairs":[],"rationale":"The evaluator must abstain despite historical confirmation noise.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_empty_01","partition":"holdout","stratum":"no-result-control","query":"Where is the indoor garden watering schedule documented?","workspace":"fixture-workspace-a","candidateIds":["mem_noise_keyboard","mem_noise_zoom"],"expectedIds":[],"forbiddenIds":["mem_noise_keyboard","mem_noise_zoom"],"preferredPairs":[],"rationale":"Unrelated synthetic candidates cannot fill an empty result.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_empty_02","partition":"holdout","stratum":"no-result-control","query":"Which music playlist accompanies the evening workshop?","workspace":"fixture-workspace-a","candidateIds":["mem_noise_keyboard","mem_noise_zoom"],"expectedIds":[],"forbiddenIds":["mem_noise_keyboard","mem_noise_zoom"],"preferredPairs":[],"rationale":"Feedback must not defeat lexical abstention on a novel topic.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_scope_01","partition":"development","stratum":"scope-safety","query":"Which amber badge is used for staging status?","workspace":"fixture-workspace-a","candidateIds":["mem_scope_other","mem_scope_local"],"expectedIds":["mem_scope_local"],"forbiddenIds":["mem_scope_other"],"preferredPairs":[{"preferredId":"mem_scope_local","otherId":"mem_scope_other","rationale":"The stronger confirmed record belongs to a different workspace."}],"rationale":"Scope authorization must precede feedback evaluation.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_scope_02","partition":"development","stratum":"scope-safety","query":"Find the staging notice badge for this workspace.","workspace":"fixture-workspace-a","candidateIds":["mem_scope_other","mem_scope_local"],"expectedIds":["mem_scope_local"],"forbiddenIds":["mem_scope_other"],"preferredPairs":[{"preferredId":"mem_scope_local","otherId":"mem_scope_other","rationale":"Only the local workspace record is authorized."}],"rationale":"Second development scope isolation case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_scope_01","partition":"holdout","stratum":"scope-safety","query":"What colored staging indicator applies locally?","workspace":"fixture-workspace-a","candidateIds":["mem_scope_other","mem_scope_local"],"expectedIds":["mem_scope_local"],"forbiddenIds":["mem_scope_other"],"preferredPairs":[{"preferredId":"mem_scope_local","otherId":"mem_scope_other","rationale":"Cross-workspace confirmation cannot grant visibility."}],"rationale":"Holdout scope case with different wording.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_scope_02","partition":"holdout","stratum":"scope-safety","query":"Recall this workspace staging notification marker.","workspace":"fixture-workspace-a","candidateIds":["mem_scope_other","mem_scope_local"],"expectedIds":["mem_scope_local"],"forbiddenIds":["mem_scope_other"],"preferredPairs":[{"preferredId":"mem_scope_local","otherId":"mem_scope_other","rationale":"Authorization is independent of feedback strength."}],"rationale":"Second holdout scope isolation case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_lifecycle_01","partition":"development","stratum":"lifecycle-safety","query":"Which compact timeline remains active for incident summaries?","workspace":"fixture-workspace-a","candidateIds":["mem_disabled_timeline","mem_active_timeline"],"expectedIds":["mem_active_timeline"],"forbiddenIds":["mem_disabled_timeline"],"preferredPairs":[{"preferredId":"mem_active_timeline","otherId":"mem_disabled_timeline","rationale":"The exact but disabled record must remain ineligible."}],"rationale":"Lifecycle filtering must run before feedback ordering.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_lifecycle_02","partition":"development","stratum":"lifecycle-safety","query":"Retrieve the enabled incident timeline guidance.","workspace":"fixture-workspace-a","candidateIds":["mem_disabled_timeline","mem_active_timeline"],"expectedIds":["mem_active_timeline"],"forbiddenIds":["mem_disabled_timeline"],"preferredPairs":[{"preferredId":"mem_active_timeline","otherId":"mem_disabled_timeline","rationale":"Historical confirmation cannot reactivate a disabled Memory."}],"rationale":"Second development lifecycle case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_lifecycle_01","partition":"holdout","stratum":"lifecycle-safety","query":"What timeline format can still be used for incident recaps?","workspace":"fixture-workspace-a","candidateIds":["mem_disabled_timeline","mem_active_timeline"],"expectedIds":["mem_active_timeline"],"forbiddenIds":["mem_disabled_timeline"],"preferredPairs":[{"preferredId":"mem_active_timeline","otherId":"mem_disabled_timeline","rationale":"Only the active summary record is eligible."}],"rationale":"Holdout lifecycle isolation case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_lifecycle_02","partition":"holdout","stratum":"lifecycle-safety","query":"Locate the current compact chronology rule for incidents.","workspace":"fixture-workspace-a","candidateIds":["mem_disabled_timeline","mem_active_timeline"],"expectedIds":["mem_active_timeline"],"forbiddenIds":["mem_disabled_timeline"],"preferredPairs":[{"preferredId":"mem_active_timeline","otherId":"mem_disabled_timeline","rationale":"Disabled content stays excluded regardless of confirmation."}],"rationale":"Second holdout lifecycle isolation case.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_misleading_01","partition":"development","stratum":"misleading-feedback-control","query":"How many review passes are required for release notes?","workspace":"fixture-workspace-a","candidateIds":["mem_misleading_expected","mem_misleading_confirmed"],"expectedIds":["mem_misleading_expected"],"forbiddenIds":["mem_misleading_confirmed"],"preferredPairs":[{"preferredId":"mem_misleading_expected","otherId":"mem_misleading_confirmed","rationale":"The synthetic label marks three passes current even though the distractor was confirmed."}],"rationale":"Exposes the risk of stale or mistaken explicit feedback.","limit":1,"promptCharacterBudget":512}, + {"id":"case_dev_misleading_02","partition":"development","stratum":"misleading-feedback-control","query":"Recall the required release note review count.","workspace":"fixture-workspace-a","candidateIds":["mem_misleading_expected","mem_misleading_confirmed"],"expectedIds":["mem_misleading_expected"],"forbiddenIds":["mem_misleading_confirmed"],"preferredPairs":[{"preferredId":"mem_misleading_expected","otherId":"mem_misleading_confirmed","rationale":"Confirmation is evidence, not authority over the labeled current fact."}],"rationale":"Second development misleading-feedback control.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_misleading_01","partition":"holdout","stratum":"misleading-feedback-control","query":"What is the mandated number of checks for release notes?","workspace":"fixture-workspace-a","candidateIds":["mem_misleading_expected","mem_misleading_confirmed"],"expectedIds":["mem_misleading_expected"],"forbiddenIds":["mem_misleading_confirmed"],"preferredPairs":[{"preferredId":"mem_misleading_expected","otherId":"mem_misleading_confirmed","rationale":"A confirmed distractor must be visible as a quality regression."}],"rationale":"Holdout case prevents treating all confirmation as correct.","limit":1,"promptCharacterBudget":512}, + {"id":"case_hold_misleading_02","partition":"holdout","stratum":"misleading-feedback-control","query":"Find the review-pass policy governing release documentation.","workspace":"fixture-workspace-a","candidateIds":["mem_misleading_expected","mem_misleading_confirmed"],"expectedIds":["mem_misleading_expected"],"forbiddenIds":["mem_misleading_confirmed"],"preferredPairs":[{"preferredId":"mem_misleading_expected","otherId":"mem_misleading_confirmed","rationale":"The evaluator must count an evidence-driven wrong order against the candidate."}],"rationale":"Second holdout misleading-feedback regression case.","limit":1,"promptCharacterBudget":512} + ] +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-lock.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-lock.v1.json new file mode 100644 index 000000000..f27862658 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-lock.v1.json @@ -0,0 +1,26 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "decisionId": "kun-memory-feedback-tiebreaker-v1", + "status": "locked", + "artifactHashes": { + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "manifestSha256": "17ab0ea2186d11bbdfb38e0f76b75f3547144c74412183e8118cb2e525d6124a", + "calibrationSha256": "b34b6d995a2230aac543885d4173c07f8e3024b4fd8603637d5f49e4258a7964", + "decisionPlanSha256": "27dfc5852afa9d86e0ff52bf549fb154e5b788ac00690758c0c8accd15460ab7", + "developmentReportSha256": "c66eecdd069a6507bf158a980e25a556f87fb39f69b6674bede77470d70b1542" + }, + "selectedCandidateId": "foundation-control", + "gates": { + "localBenefit": {"minimumPairAccuracyGain":0.25,"minimumPairAccuracyGainLowerBound":0.001}, + "globalNonRegression": {"minimumRecallGainLowerBound":0,"minimumMrrGainLowerBound":0,"maximumPrecisionDecline":0,"maximumAbstentionDecline":0}, + "safety": {"maximumExplicitForbiddenSelections":0,"maximumAuthorizationOrLifecycleViolations":0,"maximumProductionRankingChanges":0}, + "privacy": {"queryTextInTrace":false,"memoryContentInTrace":false,"sourceExcerptInTrace":false,"machinePathInTrace":false,"credentialInTrace":false}, + "determinism": {"repeatedRuns":3,"numericTolerance":0.000001}, + "resource": {"maximumEvaluationMilliseconds":5000,"maximumTraceRankings":64,"maximumEvidenceBytes":1000000} + }, + "evaluatorIdentity": "memory-feedback-tiebreaker-v1|foundation=post-1308-foundation-v1|grouping=leader-relative", + "bootstrapSeed": 20260916, + "holdoutRunLimit": 1, + "lockedAt": "2026-09-16T00:00:00.000Z" +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-manifest.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-manifest.v1.json new file mode 100644 index 000000000..445f17f37 --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-manifest.v1.json @@ -0,0 +1,29 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "datasetId": "kun-memory-feedback-tiebreaker-anonymous-v1", + "status": "frozen", + "evaluationNow": "2026-09-16T00:00:00.000Z", + "normalization": { + "unicode": "NFKC", + "caseFold": "unicode-lowercase", + "punctuation": "strip", + "nearDuplicateTokenJaccard": 0.75 + }, + "splitCounts": { + "development": 18, + "holdout": 18 + }, + "stratumQuotas": [ + {"stratum":"confirmed-near-tie","development":2,"holdout":2}, + {"stratum":"corrected-replacement-near-tie","development":2,"holdout":2}, + {"stratum":"frequent-unconfirmed-control","development":2,"holdout":2}, + {"stratum":"no-feedback-control","development":2,"holdout":2}, + {"stratum":"wide-gap-control","development":2,"holdout":2}, + {"stratum":"no-result-control","development":2,"holdout":2}, + {"stratum":"scope-safety","development":2,"holdout":2}, + {"stratum":"lifecycle-safety","development":2,"holdout":2}, + {"stratum":"misleading-feedback-control","development":2,"holdout":2} + ], + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259" +} diff --git a/kun/src/memory/fixtures/memory-feedback-tiebreaker-plan.v1.json b/kun/src/memory/fixtures/memory-feedback-tiebreaker-plan.v1.json new file mode 100644 index 000000000..328b02caf --- /dev/null +++ b/kun/src/memory/fixtures/memory-feedback-tiebreaker-plan.v1.json @@ -0,0 +1,88 @@ +{ + "schemaVersion": 1, + "evaluationVersion": "p3-feedback-tiebreaker-v1", + "decisionId": "kun-memory-feedback-tiebreaker-v1", + "status": "pre-registered", + "artifactHashes": { + "fixtureSha256": "a6a37e35a5da6afdfbdae42b9abe8a9f182dcad19167fc82361f662873d7e259", + "manifestSha256": "17ab0ea2186d11bbdfb38e0f76b75f3547144c74412183e8118cb2e525d6124a", + "calibrationSha256": "b34b6d995a2230aac543885d4173c07f8e3024b4fd8603637d5f49e4258a7964" + }, + "partitions": { + "development": [ + "case_dev_confirmed_01", "case_dev_confirmed_02", "case_dev_corrected_01", + "case_dev_corrected_02", "case_dev_empty_01", "case_dev_empty_02", + "case_dev_frequent_01", "case_dev_frequent_02", "case_dev_lifecycle_01", + "case_dev_lifecycle_02", "case_dev_misleading_01", "case_dev_misleading_02", + "case_dev_neutral_01", "case_dev_neutral_02", "case_dev_scope_01", + "case_dev_scope_02", "case_dev_wide_01", "case_dev_wide_02" + ], + "holdout": [ + "case_hold_confirmed_01", "case_hold_confirmed_02", "case_hold_corrected_01", + "case_hold_corrected_02", "case_hold_empty_01", "case_hold_empty_02", + "case_hold_frequent_01", "case_hold_frequent_02", "case_hold_lifecycle_01", + "case_hold_lifecycle_02", "case_hold_misleading_01", "case_hold_misleading_02", + "case_hold_neutral_01", "case_hold_neutral_02", "case_hold_scope_01", + "case_hold_scope_02", "case_hold_wide_01", "case_hold_wide_02" + ] + }, + "candidates": [ + {"id":"foundation-control","kind":"foundation-control","boundaryId":null,"signalRule":"foundation-only","orderingSignals":[],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-gap_0","kind":"near-tie-rerank","boundaryId":"gap_0","signalRule":"confirmation","orderingSignals":["confirmation"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-correction-gap_0","kind":"near-tie-rerank","boundaryId":"gap_0","signalRule":"confirmation-correction","orderingSignals":["confirmation","correction"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-gap_1","kind":"near-tie-rerank","boundaryId":"gap_1","signalRule":"confirmation","orderingSignals":["confirmation"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-correction-gap_1","kind":"near-tie-rerank","boundaryId":"gap_1","signalRule":"confirmation-correction","orderingSignals":["confirmation","correction"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-gap_2","kind":"near-tie-rerank","boundaryId":"gap_2","signalRule":"confirmation","orderingSignals":["confirmation"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]}, + {"id":"confirmation-correction-gap_2","kind":"near-tie-rerank","boundaryId":"gap_2","signalRule":"confirmation-correction","orderingSignals":["confirmation","correction"],"shadowSignals":["retrieval-frequency","last-retrieved-at"]} + ], + "selectionRule": { + "orderBy": [ + "all-required-gates", "pair-accuracy-lower-bound", "pair-accuracy-point-estimate", + "smallest-boundary", "candidate-id" + ], + "fallback": "foundation-control" + }, + "gates": { + "localBenefit": { + "minimumPairAccuracyGain": 0.25, + "minimumPairAccuracyGainLowerBound": 0.001 + }, + "globalNonRegression": { + "minimumRecallGainLowerBound": 0, + "minimumMrrGainLowerBound": 0, + "maximumPrecisionDecline": 0, + "maximumAbstentionDecline": 0 + }, + "safety": { + "maximumExplicitForbiddenSelections": 0, + "maximumAuthorizationOrLifecycleViolations": 0, + "maximumProductionRankingChanges": 0 + }, + "privacy": { + "queryTextInTrace": false, + "memoryContentInTrace": false, + "sourceExcerptInTrace": false, + "machinePathInTrace": false, + "credentialInTrace": false + }, + "determinism": {"repeatedRuns":3,"numericTolerance":0.000001}, + "resource": { + "maximumEvaluationMilliseconds": 5000, + "maximumTraceRankings": 64, + "maximumEvidenceBytes": 1000000 + } + }, + "bootstrap": { + "method": "paired-percentile-lower-bound", + "seed": 20260916, + "resamples": 10000, + "confidenceLevel": 0.95, + "unit": "case", + "holdoutRuns": 1 + }, + "production": { + "rankingChanged": false, + "dormantFeatureFlagAdded": false, + "evaluatorOnly": true + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-bootstrap.test.ts b/kun/src/memory/memory-feedback-tiebreaker-bootstrap.test.ts new file mode 100644 index 000000000..05f037b5a --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-bootstrap.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, it } from 'vitest' +import { memoryFeedbackPairedBootstrapLowerBound } from './memory-feedback-tiebreaker-bootstrap.js' + +describe('memory feedback paired bootstrap', () => { + it('produces deterministic paired lower bounds', () => { + const input = { + baseline: [0, 1, 0, 1], + candidate: [1, 1, 0, 1], + seed: 20260916, + resamples: 2_000, + confidenceLevel: 0.95 + } + const first = memoryFeedbackPairedBootstrapLowerBound(input) + const second = memoryFeedbackPairedBootstrapLowerBound(input) + + expect(first).toEqual(second) + expect(first.pointEstimate).toBe(0.25) + expect(first.lowerBound).toBeLessThanOrEqual(first.pointEstimate) + }) + + it('retains an exact lower bound for a constant paired gain', () => { + expect(memoryFeedbackPairedBootstrapLowerBound({ + baseline: [0, 0.25, 0.5, 0.75], + candidate: [0.25, 0.5, 0.75, 1], + seed: 0, + resamples: 1_000, + confidenceLevel: 0.95 + })).toMatchObject({ pointEstimate: 0.25, lowerBound: 0.25 }) + }) + + it('rejects unpaired or invalid bootstrap requests', () => { + expect(() => memoryFeedbackPairedBootstrapLowerBound({ + baseline: [1], candidate: [], seed: 1, resamples: 1_000, confidenceLevel: 0.95 + })).toThrow(/equal non-empty/u) + expect(() => memoryFeedbackPairedBootstrapLowerBound({ + baseline: [1], candidate: [1], seed: 1, resamples: 0, confidenceLevel: 0.95 + })).toThrow(/positive integer/u) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-bootstrap.ts b/kun/src/memory/memory-feedback-tiebreaker-bootstrap.ts new file mode 100644 index 000000000..74d8f8479 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-bootstrap.ts @@ -0,0 +1,65 @@ +export type MemoryFeedbackPairedBootstrapResult = { + pointEstimate: number + lowerBound: number + seed: number + resamples: number + confidenceLevel: number +} + +export function memoryFeedbackPairedBootstrapLowerBound(input: { + baseline: readonly number[] + candidate: readonly number[] + seed: number + resamples: number + confidenceLevel: number +}): MemoryFeedbackPairedBootstrapResult { + if (input.baseline.length === 0 || input.baseline.length !== input.candidate.length) { + throw new Error('paired bootstrap requires equal non-empty samples') + } + if (!Number.isInteger(input.resamples) || input.resamples < 1) { + throw new Error('paired bootstrap resamples must be a positive integer') + } + if (input.confidenceLevel <= 0.5 || input.confidenceLevel >= 1) { + throw new Error('paired bootstrap confidence level must be between 0.5 and 1') + } + const random = seededRandom(input.seed) + const deltas = input.candidate.map((value, index) => value - input.baseline[index]!) + const samples: number[] = [] + for (let sample = 0; sample < input.resamples; sample += 1) { + let total = 0 + for (let index = 0; index < deltas.length; index += 1) { + total += deltas[Math.floor(random() * deltas.length)]! + } + samples.push(total / deltas.length) + } + samples.sort((left, right) => left - right) + const lowerIndex = Math.min( + samples.length - 1, + Math.floor((1 - input.confidenceLevel) * samples.length) + ) + return { + pointEstimate: round(average(deltas)), + lowerBound: round(samples[lowerIndex]!), + seed: input.seed, + resamples: input.resamples, + confidenceLevel: input.confidenceLevel + } +} + +function seededRandom(seed: number): () => number { + let state = (seed >>> 0) || 0x9e3779b9 + return () => { + state ^= state << 13 + state ^= state >>> 17 + state ^= state << 5 + return (state >>> 0) / 0x1_0000_0000 + } +} + +function average(values: readonly number[]): number { + return values.reduce((total, value) => total + value, 0) / values.length +} + +function round(value: number): number { + return Math.round(value * 1_000_000) / 1_000_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-calibration-report.test.ts b/kun/src/memory/memory-feedback-tiebreaker-calibration-report.test.ts new file mode 100644 index 000000000..ff2dff027 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-calibration-report.test.ts @@ -0,0 +1,46 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + memoryFeedbackTiebreakerArtifactSha256 +} from './memory-feedback-tiebreaker-contracts.js' +import { createMemoryFeedbackTiebreakerCalibrationReport } from './memory-feedback-tiebreaker-calibration-report.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' + +describe('memory feedback tiebreaker calibration report', () => { + it('derives a deterministic finite boundary grid from development foundation gaps only', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const first = createMemoryFeedbackTiebreakerCalibrationReport({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture + }) + const second = createMemoryFeedbackTiebreakerCalibrationReport({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture + }) + + expect(first).toEqual(second) + expect(first.boundaryGrid[0]?.maximumGap).toBe(0) + expect(first.boundaryGrid.length).toBeLessThanOrEqual(4) + expect(first.observedGaps.length).toBeGreaterThan(0) + expect(first.observedGaps.every((gap) => gap.caseId.startsWith('case_dev_'))).toBe(true) + expect(memoryFeedbackTiebreakerArtifactSha256(first)) + .toBe(memoryFeedbackTiebreakerArtifactSha256(second)) + }) + + it('matches the frozen baseline-only calibration artifact', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const generated = createMemoryFeedbackTiebreakerCalibrationReport({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture + }) + const path = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-calibration.v1.json', + import.meta.url + )) + const frozen = MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(await readFile(path, 'utf8'))) + + expect(frozen).toEqual(generated) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-calibration-report.ts b/kun/src/memory/memory-feedback-tiebreaker-calibration-report.ts new file mode 100644 index 000000000..8b38e255f --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-calibration-report.ts @@ -0,0 +1,32 @@ +import { + MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE, + deriveMemoryFeedbackTiebreakerBoundaries +} from './memory-feedback-tiebreaker-calibration.js' +import { + MemoryFeedbackTiebreakerCalibrationReport, + type MemoryFeedbackTiebreakerCalibration, + type MemoryFeedbackTiebreakerFixture +} from './memory-feedback-tiebreaker-contracts.js' +import { rankMemoryFeedbackTiebreakerDevelopment } from './memory-feedback-tiebreaker-foundation.js' + +export const MEMORY_FEEDBACK_TIEBREAKER_FOUNDATION_VERSION = 'post-1308-foundation-v1' + +export function createMemoryFeedbackTiebreakerCalibrationReport(input: { + fixture: MemoryFeedbackTiebreakerFixture + fixtureSha256: string +}): MemoryFeedbackTiebreakerCalibration { + const observedGaps = rankMemoryFeedbackTiebreakerDevelopment(input.fixture) + .flatMap((result) => result.gaps.map((gap) => ({ caseId: result.caseId, ...gap }))) + .sort((left, right) => left.caseId.localeCompare(right.caseId) || + left.higherId.localeCompare(right.higherId) || left.lowerId.localeCompare(right.lowerId)) + + return MemoryFeedbackTiebreakerCalibrationReport.parse({ + schemaVersion: 1, + evaluationVersion: 'p3-feedback-tiebreaker-v1', + fixtureSha256: input.fixtureSha256, + foundationVersion: MEMORY_FEEDBACK_TIEBREAKER_FOUNDATION_VERSION, + procedure: MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE, + observedGaps, + boundaryGrid: deriveMemoryFeedbackTiebreakerBoundaries(observedGaps.map((item) => item.gap)) + }) +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-calibration.ts b/kun/src/memory/memory-feedback-tiebreaker-calibration.ts new file mode 100644 index 000000000..638045844 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-calibration.ts @@ -0,0 +1,107 @@ +import { + MemoryFeedbackTiebreakerCalibrationProcedure, + MemoryFeedbackTiebreakerCandidate, + type MemoryFeedbackTiebreakerCandidateValue, + type MemoryFeedbackTiebreakerSignalRuleValue +} from './memory-feedback-tiebreaker-contracts.js' + +export const MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE = + MemoryFeedbackTiebreakerCalibrationProcedure.parse({ + source: 'development-foundation-only', + gapFormula: 'higher-score-minus-lower-score', + quantiles: [0.25, 0.5, 0.75], + quantileMethod: 'floor-index', + maximumGridSize: 4, + boundaryInclusive: true, + grouping: 'leader-relative', + maximumReorderWindow: 3, + exactTieControl: true, + noRerankControl: true, + stableFallback: ['foundation-score', 'memory-id'] + }) + +export type MemoryFeedbackTiebreakerItem = { + memoryId: string + foundationScore: number + confirmationCount: number + correctionCount: number + retrievalCount: number + lastRetrievedAt?: string +} + +export function deriveMemoryFeedbackTiebreakerBoundaries(gaps: readonly number[]): Array<{ + id: string + maximumGap: number + inclusive: true +}> { + const sorted = gaps + .filter((gap) => Number.isFinite(gap) && gap >= 0 && gap <= 1) + .map((gap) => round(gap)) + .sort((left, right) => left - right) + const values = [0, ...MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE.quantiles.map((quantile) => + sorted[Math.floor(quantile * Math.max(0, sorted.length - 1))] ?? 0)] + return [...new Set(values)] + .slice(0, MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE.maximumGridSize) + .map((maximumGap, index) => ({ id: `gap_${index}`, maximumGap, inclusive: true })) +} + +export function buildMemoryFeedbackTiebreakerCandidates( + boundaries: ReadonlyArray<{ id: string }> +): MemoryFeedbackTiebreakerCandidateValue[] { + const shadowSignals = ['retrieval-frequency', 'last-retrieved-at'] as const + const candidates: unknown[] = [{ + id: 'foundation-control', + kind: 'foundation-control', + boundaryId: null, + signalRule: 'foundation-only', + orderingSignals: [], + shadowSignals + }] + for (const boundary of boundaries) { + candidates.push({ + id: `confirmation-${boundary.id}`, + kind: 'near-tie-rerank', + boundaryId: boundary.id, + signalRule: 'confirmation', + orderingSignals: ['confirmation'], + shadowSignals + }, { + id: `confirmation-correction-${boundary.id}`, + kind: 'near-tie-rerank', + boundaryId: boundary.id, + signalRule: 'confirmation-correction', + orderingSignals: ['confirmation', 'correction'], + shadowSignals + }) + } + return candidates.map((candidate) => MemoryFeedbackTiebreakerCandidate.parse(candidate)) +} + +export function compareMemoryFeedbackTiebreakerItems( + left: MemoryFeedbackTiebreakerItem, + right: MemoryFeedbackTiebreakerItem, + signalRule: MemoryFeedbackTiebreakerSignalRuleValue +): number { + const evidence = evidencePriority(right, signalRule) - evidencePriority(left, signalRule) + if (evidence !== 0) return evidence + const foundation = finiteScore(right.foundationScore) - finiteScore(left.foundationScore) + if (foundation !== 0) return foundation + return left.memoryId < right.memoryId ? -1 : left.memoryId > right.memoryId ? 1 : 0 +} + +function evidencePriority( + item: MemoryFeedbackTiebreakerItem, + signalRule: MemoryFeedbackTiebreakerSignalRuleValue +): number { + if (signalRule === 'confirmation-correction' && item.correctionCount > 0) return 2 + if (signalRule !== 'foundation-only' && item.confirmationCount > 0) return 1 + return 0 +} + +function finiteScore(value: number): number { + return Number.isFinite(value) ? value : 0 +} + +function round(value: number): number { + return Math.round(value * 1_000_000) / 1_000_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-contracts.test.ts b/kun/src/memory/memory-feedback-tiebreaker-contracts.test.ts new file mode 100644 index 000000000..144a01199 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-contracts.test.ts @@ -0,0 +1,326 @@ +import { describe, expect, it } from 'vitest' +import { + MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID, + MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerCandidateLock, + MemoryFeedbackTiebreakerDecisionPlan, + MemoryFeedbackTiebreakerEvidence, + MemoryFeedbackTiebreakerFixtureFile, + MemoryFeedbackTiebreakerManifest, + memoryFeedbackTiebreakerArtifactSha256 +} from './memory-feedback-tiebreaker-contracts.js' +import { + MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE, + buildMemoryFeedbackTiebreakerCandidates, + compareMemoryFeedbackTiebreakerItems, + deriveMemoryFeedbackTiebreakerBoundaries +} from './memory-feedback-tiebreaker-calibration.js' + +const HASH_A = 'a'.repeat(64) +const HASH_B = 'b'.repeat(64) +const EVALUATION_NOW = '2026-09-16T00:00:00.000Z' + +describe('memory feedback tie-breaker contracts', () => { + it('strictly validates every versioned artifact shape', () => { + const fixture = makeFixture() + const manifest = makeManifest() + const calibration = makeCalibration() + const plan = makePlan() + const lock = makeLock() + const evidence = makeEvidence() + + expect(MemoryFeedbackTiebreakerFixtureFile.parse(fixture).cases).toHaveLength(2) + expect(MemoryFeedbackTiebreakerManifest.parse(manifest).splitCounts).toEqual({ development: 1, holdout: 1 }) + expect(MemoryFeedbackTiebreakerCalibrationReport.parse(calibration).boundaryGrid[0]?.maximumGap).toBe(0) + expect(MemoryFeedbackTiebreakerDecisionPlan.parse(plan).production.evaluatorOnly).toBe(true) + expect(MemoryFeedbackTiebreakerCandidateLock.parse(lock).holdoutRunLimit).toBe(1) + expect(MemoryFeedbackTiebreakerEvidence.parse(evidence).decision).toBe('no-go') + + expect(() => MemoryFeedbackTiebreakerFixtureFile.parse({ ...fixture, unexpected: true })).toThrow() + expect(() => MemoryFeedbackTiebreakerFixtureFile.parse({ ...fixture, schemaVersion: 2 })).toThrow() + expect(() => MemoryFeedbackTiebreakerCandidateLock.parse({ + ...lock, + artifactHashes: { ...lock.artifactHashes, fixtureSha256: 'not-a-hash' } + })).toThrow() + }) + + it('rejects duplicate identities and invalid references before evaluation', () => { + const fixture = makeFixture() + expect(() => MemoryFeedbackTiebreakerFixtureFile.parse({ + ...fixture, + records: [...fixture.records, fixture.records[0]] + })).toThrow(/record ids must be unique/u) + expect(() => MemoryFeedbackTiebreakerFixtureFile.parse({ + ...fixture, + cases: [{ ...fixture.cases[0], expectedIds: ['missing'] }, fixture.cases[1]] + })).toThrow(/unknown memory/u) + }) + + it('derives a finite deterministic boundary grid with exact-tie control', () => { + const boundaries = deriveMemoryFeedbackTiebreakerBoundaries([0.04, 0.01, 0.03, 0.02]) + expect(boundaries).toEqual([ + { id: 'gap_0', maximumGap: 0, inclusive: true }, + { id: 'gap_1', maximumGap: 0.01, inclusive: true }, + { id: 'gap_2', maximumGap: 0.02, inclusive: true }, + { id: 'gap_3', maximumGap: 0.03, inclusive: true } + ]) + expect(deriveMemoryFeedbackTiebreakerBoundaries([Number.NaN, -1, 2])).toEqual([ + { id: 'gap_0', maximumGap: 0, inclusive: true } + ]) + expect(MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE.maximumReorderWindow).toBe(3) + }) + + it('keeps retrieval frequency shadow-only in candidates and comparison', () => { + const [boundary] = deriveMemoryFeedbackTiebreakerBoundaries([0.02]) + const candidates = buildMemoryFeedbackTiebreakerCandidates([boundary!]) + expect(candidates).toHaveLength(3) + expect(candidates.every((candidate) => !candidate.orderingSignals.includes('retrieval-frequency' as never))).toBe(true) + expect(candidates.every((candidate) => candidate.shadowSignals[0] === 'retrieval-frequency')).toBe(true) + + const base = { + memoryId: 'memory-a', + foundationScore: 0.5, + confirmationCount: 0, + correctionCount: 0, + retrievalCount: 0 + } + const other = { ...base, memoryId: 'memory-b', foundationScore: 0.49, retrievalCount: 1 } + const frequent = { ...other, retrievalCount: 1_000_000 } + expect(compareMemoryFeedbackTiebreakerItems(base, other, 'confirmation')).toBe( + compareMemoryFeedbackTiebreakerItems(base, frequent, 'confirmation') + ) + }) + + it('changes the plan hash when a gate or candidate rule changes', () => { + const plan = MemoryFeedbackTiebreakerDecisionPlan.parse(makePlan()) + const changedGate = { + ...plan, + gates: { + ...plan.gates, + localBenefit: { ...plan.gates.localBenefit, minimumPairAccuracyGain: 0.2 } + } + } + const changedCandidate = { + ...plan, + candidates: plan.candidates.map((candidate, index) => index === 1 + ? { ...candidate, signalRule: 'confirmation-correction', orderingSignals: ['confirmation', 'correction'] } + : candidate) + } + const originalHash = memoryFeedbackTiebreakerArtifactSha256(plan) + expect(memoryFeedbackTiebreakerArtifactSha256(changedGate)).not.toBe(originalHash) + expect(memoryFeedbackTiebreakerArtifactSha256(changedCandidate)).not.toBe(originalHash) + }) +}) + +function makeFixture() { + const records = [makeRecord('memory-a'), makeRecord('memory-b')] + return { + schemaVersion: 1, + datasetId: MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID, + status: 'frozen', + evaluationNow: EVALUATION_NOW, + records, + events: [{ + schemaVersion: 1, + id: 'event-confirm-a', + kind: 'confirmed', + memoryId: 'memory-a', + occurredAt: '2026-09-01T00:00:00.000Z' + }], + cases: [makeCase('case_development', 'development'), makeCase('case_holdout', 'holdout')] + } +} + +function makeRecord(id: string) { + return { + schemaVersion: 2, + id, + content: `Synthetic content for ${id}.`, + scope: 'workspace', + workspace: 'fixture-workspace', + tags: ['synthetic'], + confidence: 0.8, + type: 'fact', + authority: 'reference', + importance: 0.5, + observedAt: '2026-01-01T00:00:00.000Z', + createdAt: '2026-01-01T00:00:00.000Z', + updatedAt: '2026-01-01T00:00:00.000Z', + sources: [{ id: 'synthetic-source', kind: 'user', trust: 'explicit-user' }] + } +} + +function makeCase(id: string, partition: 'development' | 'holdout') { + return { + id, + partition, + stratum: 'confirmed-near-tie', + query: `Synthetic query for ${partition}.`, + workspace: 'fixture-workspace', + candidateIds: ['memory-a', 'memory-b'], + expectedIds: ['memory-a'], + forbiddenIds: [], + preferredPairs: [{ preferredId: 'memory-a', otherId: 'memory-b', rationale: 'Synthetic preference.' }], + rationale: 'Synthetic contract case.', + limit: 1, + promptCharacterBudget: 1_000 + } +} + +function makeManifest() { + return { + schemaVersion: 1, + evaluationVersion: MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + datasetId: MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID, + status: 'frozen', + evaluationNow: EVALUATION_NOW, + normalization: { + unicode: 'NFKC', + caseFold: 'unicode-lowercase', + punctuation: 'strip', + nearDuplicateTokenJaccard: 0.75 + }, + splitCounts: { development: 1, holdout: 1 }, + stratumQuotas: [{ stratum: 'confirmed-near-tie', development: 1, holdout: 1 }], + fixtureSha256: HASH_A + } +} + +function makeCalibration() { + return { + schemaVersion: 1, + evaluationVersion: MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + fixtureSha256: HASH_A, + foundationVersion: 'post-1308-lexical-foundation', + procedure: MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE, + observedGaps: [], + boundaryGrid: [{ id: 'gap_0', maximumGap: 0, inclusive: true }] + } +} + +function makePlan() { + const boundaries = [{ id: 'gap_0', maximumGap: 0, inclusive: true }] + return { + schemaVersion: 1, + evaluationVersion: MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + decisionId: 'kun-memory-feedback-tiebreaker-v1', + status: 'pre-registered', + artifactHashes: { fixtureSha256: HASH_A, manifestSha256: HASH_B, calibrationSha256: HASH_A }, + partitions: { development: ['case_development'], holdout: ['case_holdout'] }, + candidates: buildMemoryFeedbackTiebreakerCandidates(boundaries), + selectionRule: { + orderBy: [ + 'all-required-gates', + 'pair-accuracy-lower-bound', + 'pair-accuracy-point-estimate', + 'smallest-boundary', + 'candidate-id' + ], + fallback: 'foundation-control' + }, + gates: { + localBenefit: { minimumPairAccuracyGain: 0.1, minimumPairAccuracyGainLowerBound: 0 }, + globalNonRegression: { + minimumRecallGainLowerBound: -0.01, + minimumMrrGainLowerBound: -0.01, + maximumPrecisionDecline: 0.01, + maximumAbstentionDecline: 0 + }, + safety: { + maximumExplicitForbiddenSelections: 0, + maximumAuthorizationOrLifecycleViolations: 0, + maximumProductionRankingChanges: 0 + }, + privacy: { + queryTextInTrace: false, + memoryContentInTrace: false, + sourceExcerptInTrace: false, + machinePathInTrace: false, + credentialInTrace: false + }, + determinism: { repeatedRuns: 3, numericTolerance: 0.000001 }, + resource: { maximumEvaluationMilliseconds: 1_000, maximumTraceRankings: 64, maximumEvidenceBytes: 1_000_000 } + }, + bootstrap: { + method: 'paired-percentile-lower-bound', + seed: 20260916, + resamples: 2_000, + confidenceLevel: 0.95, + unit: 'case', + holdoutRuns: 1 + }, + production: { rankingChanged: false, dormantFeatureFlagAdded: false, evaluatorOnly: true } + } +} + +function makeLock() { + return { + schemaVersion: 1, + evaluationVersion: MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + decisionId: 'kun-memory-feedback-tiebreaker-v1', + status: 'locked', + artifactHashes: { + fixtureSha256: HASH_A, + manifestSha256: HASH_B, + calibrationSha256: HASH_A, + decisionPlanSha256: HASH_B, + developmentReportSha256: HASH_A + }, + selectedCandidateId: 'foundation-control', + gates: makePlan().gates, + evaluatorIdentity: 'memory-feedback-tiebreaker-v1', + bootstrapSeed: 20260916, + holdoutRunLimit: 1, + lockedAt: EVALUATION_NOW + } +} + +function makeEvidence() { + const metrics = { + pairAccuracy: 0.5, + pairCount: 1, + recallAtK: 1, + precisionAtK: 1, + meanReciprocalRank: 1, + abstentionAccuracy: 1, + explicitForbiddenSelections: 0, + authorizationOrLifecycleViolations: 0, + rankedCaseCount: 1, + noResultCaseCount: 0, + caseCount: 1 + } + const partition = { + foundation: metrics, + candidate: metrics, + bootstrapLowerBounds: { pairAccuracyGain: 0, recallGain: 0, mrrGain: 0 }, + gates: { + localBenefit: false, + globalNonRegression: true, + safety: true, + privacy: true, + determinism: true, + resource: true, + passed: false + } + } + return { + schemaVersion: 1, + evaluationVersion: MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + decisionId: 'kun-memory-feedback-tiebreaker-v1', + status: 'final', + artifactHashes: { + fixtureSha256: HASH_A, + manifestSha256: HASH_B, + calibrationSha256: HASH_A, + decisionPlanSha256: HASH_B, + candidateLockSha256: HASH_A + }, + selectedCandidateId: 'foundation-control', + holdoutRunCount: 1, + development: partition, + holdout: partition, + decision: 'no-go', + reasons: ['Synthetic contract evidence.'] + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-contracts.ts b/kun/src/memory/memory-feedback-tiebreaker-contracts.ts new file mode 100644 index 000000000..06d6fbcfc --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-contracts.ts @@ -0,0 +1,412 @@ +import { createHash } from 'node:crypto' +import { z } from 'zod' +import { MemoryFeedbackEvent } from '../contracts/memory-feedback.js' +import { MemoryRecord } from '../contracts/memory.js' + +export const MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID = 'kun-memory-feedback-tiebreaker-anonymous-v1' +export const MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION = 'p3-feedback-tiebreaker-v1' + +const Hash = z.string().regex(/^[a-f0-9]{64}$/u) +const ArtifactId = z.string().regex(/^[a-z0-9][a-z0-9_-]*$/u) +const Partition = z.enum(['development', 'holdout']) +const Stratum = z.enum([ + 'confirmed-near-tie', + 'corrected-replacement-near-tie', + 'frequent-unconfirmed-control', + 'no-feedback-control', + 'wide-gap-control', + 'no-result-control', + 'scope-safety', + 'lifecycle-safety', + 'misleading-feedback-control' +]) + +const PreferredPair = z.object({ + preferredId: z.string().min(1), + otherId: z.string().min(1), + rationale: z.string().min(1).max(512) +}).strict() + +const FixtureCase = z.object({ + id: z.string().regex(/^case_[a-z0-9_]+$/u), + partition: Partition, + stratum: Stratum, + query: z.string().min(1).max(512), + workspace: z.string().min(1).max(256), + candidateIds: z.array(z.string().min(1)).min(1).max(64), + expectedIds: z.array(z.string().min(1)).max(64), + forbiddenIds: z.array(z.string().min(1)).max(64), + preferredPairs: z.array(PreferredPair).max(64), + rationale: z.string().min(1).max(1_024), + limit: z.number().int().positive().max(64), + promptCharacterBudget: z.number().int().positive().max(64_000) +}).strict() + +export const MemoryFeedbackTiebreakerFixtureFile = z.object({ + schemaVersion: z.literal(1), + datasetId: z.literal(MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID), + status: z.literal('frozen'), + evaluationNow: z.string().datetime(), + records: z.array(MemoryRecord).min(1), + events: z.array(MemoryFeedbackEvent), + cases: z.array(FixtureCase).min(2) +}).strict().superRefine((fixture, context) => { + reportDuplicates(fixture.records.map((record) => record.id), 'record id', context) + reportDuplicates(fixture.events.map((event) => event.id), 'event id', context) + reportDuplicates(fixture.cases.map((item) => item.id), 'case id', context) + const records = new Map(fixture.records.map((record) => [record.id, record])) + for (const event of fixture.events) { + if (!records.has(event.memoryId)) addIssue(context, `unknown feedback memory: ${event.memoryId}`) + if (event.kind === 'corrected') { + const replacement = records.get(event.replacementMemoryId) + if (!replacement || replacement.supersedes !== event.memoryId) { + addIssue(context, `invalid correction replacement: ${event.id}`) + } + } + } + for (const item of fixture.cases) validateCaseReferences(item, records, context) + if (!fixture.cases.some((item) => item.partition === 'development') || + !fixture.cases.some((item) => item.partition === 'holdout')) { + addIssue(context, 'fixture requires both development and holdout cases') + } +}) + +const StratumQuota = z.object({ + stratum: Stratum, + development: z.number().int().nonnegative(), + holdout: z.number().int().nonnegative() +}).strict() + +export const MemoryFeedbackTiebreakerManifest = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + datasetId: z.literal(MEMORY_FEEDBACK_TIEBREAKER_DATASET_ID), + status: z.literal('frozen'), + evaluationNow: z.string().datetime(), + normalization: z.object({ + unicode: z.literal('NFKC'), + caseFold: z.literal('unicode-lowercase'), + punctuation: z.literal('strip'), + nearDuplicateTokenJaccard: z.literal(0.75) + }).strict(), + splitCounts: z.object({ + development: z.number().int().positive(), + holdout: z.number().int().positive() + }).strict(), + stratumQuotas: z.array(StratumQuota).min(1), + fixtureSha256: Hash +}).strict().superRefine((manifest, context) => { + reportDuplicates(manifest.stratumQuotas.map((quota) => quota.stratum), 'stratum quota', context) +}) + +export const MemoryFeedbackTiebreakerCalibrationProcedure = z.object({ + source: z.literal('development-foundation-only'), + gapFormula: z.literal('higher-score-minus-lower-score'), + quantiles: z.tuple([z.literal(0.25), z.literal(0.5), z.literal(0.75)]), + quantileMethod: z.literal('floor-index'), + maximumGridSize: z.literal(4), + boundaryInclusive: z.literal(true), + grouping: z.literal('leader-relative'), + maximumReorderWindow: z.number().int().min(2).max(8), + exactTieControl: z.literal(true), + noRerankControl: z.literal(true), + stableFallback: z.tuple([z.literal('foundation-score'), z.literal('memory-id')]) +}).strict() + +export const MemoryFeedbackTiebreakerBoundary = z.object({ + id: z.string().regex(/^gap_[0-9]+$/u), + maximumGap: z.number().min(0).max(1), + inclusive: z.literal(true) +}).strict() + +export const MemoryFeedbackTiebreakerCalibrationReport = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + fixtureSha256: Hash, + foundationVersion: z.string().min(1), + procedure: MemoryFeedbackTiebreakerCalibrationProcedure, + observedGaps: z.array(z.object({ + caseId: z.string().min(1), + higherId: z.string().min(1), + lowerId: z.string().min(1), + gap: z.number().min(0).max(1) + }).strict()), + boundaryGrid: z.array(MemoryFeedbackTiebreakerBoundary).min(1).max(4) +}).strict().superRefine((report, context) => { + reportDuplicates(report.boundaryGrid.map((boundary) => boundary.id), 'boundary id', context) + if (report.boundaryGrid[0]?.maximumGap !== 0) addIssue(context, 'boundary grid must start with exact tie') + for (let index = 1; index < report.boundaryGrid.length; index += 1) { + if (report.boundaryGrid[index]!.maximumGap <= report.boundaryGrid[index - 1]!.maximumGap) { + addIssue(context, 'boundary grid must be strictly increasing') + } + } +}) + +export const MemoryFeedbackTiebreakerSignalRule = z.enum([ + 'foundation-only', + 'confirmation', + 'confirmation-correction' +]) + +export const MemoryFeedbackTiebreakerCandidate = z.object({ + id: ArtifactId, + kind: z.enum(['foundation-control', 'near-tie-rerank']), + boundaryId: z.string().regex(/^gap_[0-9]+$/u).nullable(), + signalRule: MemoryFeedbackTiebreakerSignalRule, + orderingSignals: z.array(z.enum(['confirmation', 'correction'])).max(2), + shadowSignals: z.tuple([z.literal('retrieval-frequency'), z.literal('last-retrieved-at')]) +}).strict().superRefine((candidate, context) => { + if (candidate.kind === 'foundation-control' && + (candidate.boundaryId !== null || candidate.signalRule !== 'foundation-only' || candidate.orderingSignals.length > 0)) { + addIssue(context, 'foundation control cannot use a boundary or feedback signal') + } + if (candidate.kind === 'near-tie-rerank' && candidate.boundaryId === null) { + addIssue(context, 'near-tie candidate requires a boundary') + } + const expected = candidate.signalRule === 'confirmation' + ? ['confirmation'] + : candidate.signalRule === 'confirmation-correction' + ? ['confirmation', 'correction'] + : [] + if (candidate.orderingSignals.join(',') !== expected.join(',')) { + addIssue(context, 'candidate ordering signals do not match its signal rule') + } +}) + +const ArtifactHashes = z.object({ + fixtureSha256: Hash, + manifestSha256: Hash, + calibrationSha256: Hash +}).strict() + +const Bootstrap = z.object({ + method: z.literal('paired-percentile-lower-bound'), + seed: z.number().int().nonnegative().max(0xffffffff), + resamples: z.number().int().min(1_000), + confidenceLevel: z.number().gt(0.5).lt(1), + unit: z.literal('case'), + holdoutRuns: z.literal(1) +}).strict() + +const DecisionGates = z.object({ + localBenefit: z.object({ + minimumPairAccuracyGain: z.number().min(-1).max(1), + minimumPairAccuracyGainLowerBound: z.number().min(-1).max(1) + }).strict(), + globalNonRegression: z.object({ + minimumRecallGainLowerBound: z.number().min(-1).max(1), + minimumMrrGainLowerBound: z.number().min(-1).max(1), + maximumPrecisionDecline: z.number().min(0).max(1), + maximumAbstentionDecline: z.number().min(0).max(1) + }).strict(), + safety: z.object({ + maximumExplicitForbiddenSelections: z.literal(0), + maximumAuthorizationOrLifecycleViolations: z.literal(0), + maximumProductionRankingChanges: z.literal(0) + }).strict(), + privacy: z.object({ + queryTextInTrace: z.literal(false), + memoryContentInTrace: z.literal(false), + sourceExcerptInTrace: z.literal(false), + machinePathInTrace: z.literal(false), + credentialInTrace: z.literal(false) + }).strict(), + determinism: z.object({ + repeatedRuns: z.number().int().min(2).max(10), + numericTolerance: z.number().min(0).max(1e-6) + }).strict(), + resource: z.object({ + maximumEvaluationMilliseconds: z.number().int().positive(), + maximumTraceRankings: z.number().int().positive().max(64), + maximumEvidenceBytes: z.number().int().positive() + }).strict() +}).strict() + +export const MemoryFeedbackTiebreakerDecisionPlan = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + decisionId: z.string().regex(/^kun-memory-[a-z0-9-]+$/u), + status: z.literal('pre-registered'), + artifactHashes: ArtifactHashes, + partitions: z.object({ + development: z.array(z.string().min(1)).min(1), + holdout: z.array(z.string().min(1)).min(1) + }).strict(), + candidates: z.array(MemoryFeedbackTiebreakerCandidate).min(3).max(16), + selectionRule: z.object({ + orderBy: z.tuple([ + z.literal('all-required-gates'), + z.literal('pair-accuracy-lower-bound'), + z.literal('pair-accuracy-point-estimate'), + z.literal('smallest-boundary'), + z.literal('candidate-id') + ]), + fallback: z.literal('foundation-control') + }).strict(), + gates: DecisionGates, + bootstrap: Bootstrap, + production: z.object({ + rankingChanged: z.literal(false), + dormantFeatureFlagAdded: z.literal(false), + evaluatorOnly: z.literal(true) + }).strict() +}).strict().superRefine((plan, context) => { + reportDuplicates([...plan.partitions.development, ...plan.partitions.holdout], 'partition case id', context) + reportDuplicates(plan.candidates.map((candidate) => candidate.id), 'candidate id', context) +}) + +export const MemoryFeedbackTiebreakerCandidateLock = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + decisionId: z.string().regex(/^kun-memory-[a-z0-9-]+$/u), + status: z.literal('locked'), + artifactHashes: ArtifactHashes.extend({ + decisionPlanSha256: Hash, + developmentReportSha256: Hash + }).strict(), + selectedCandidateId: ArtifactId, + gates: DecisionGates, + evaluatorIdentity: z.string().min(1), + bootstrapSeed: z.number().int().nonnegative().max(0xffffffff), + holdoutRunLimit: z.literal(1), + lockedAt: z.string().datetime() +}).strict() + +export const MemoryFeedbackTiebreakerMetricSet = z.object({ + pairAccuracy: z.number().min(0).max(1), + pairCount: z.number().int().nonnegative(), + recallAtK: z.number().min(0).max(1), + precisionAtK: z.number().min(0).max(1), + meanReciprocalRank: z.number().min(0).max(1), + abstentionAccuracy: z.number().min(0).max(1), + explicitForbiddenSelections: z.number().int().nonnegative(), + authorizationOrLifecycleViolations: z.number().int().nonnegative(), + rankedCaseCount: z.number().int().nonnegative(), + noResultCaseCount: z.number().int().nonnegative(), + caseCount: z.number().int().positive() +}).strict() + +export const MemoryFeedbackTiebreakerDevelopmentArtifact = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + partition: z.literal('development'), + artifactHashes: ArtifactHashes.extend({ decisionPlanSha256: Hash }).strict(), + foundation: MemoryFeedbackTiebreakerMetricSet, + configurations: z.array(z.object({ + candidateId: ArtifactId, + boundaryId: z.string().regex(/^gap_[0-9]+$/u).nullable(), + maximumGap: z.number().min(0).max(1), + signalRule: MemoryFeedbackTiebreakerSignalRule, + metrics: MemoryFeedbackTiebreakerMetricSet, + bootstrapLowerBounds: z.object({ + pairAccuracyGain: z.number().min(-1).max(1), + recallGain: z.number().min(-1).max(1), + mrrGain: z.number().min(-1).max(1) + }).strict(), + gates: z.object({ + localBenefit: z.boolean(), + globalNonRegression: z.boolean(), + safety: z.boolean(), + passed: z.boolean() + }).strict(), + selectedIdsSha256: Hash + }).strict()).min(3).max(16) +}).strict() +export type MemoryFeedbackTiebreakerDevelopmentArtifactValue = z.infer< + typeof MemoryFeedbackTiebreakerDevelopmentArtifact +> + +const PartitionEvidence = z.object({ + foundation: MemoryFeedbackTiebreakerMetricSet, + candidate: MemoryFeedbackTiebreakerMetricSet, + bootstrapLowerBounds: z.object({ + pairAccuracyGain: z.number().min(-1).max(1), + recallGain: z.number().min(-1).max(1), + mrrGain: z.number().min(-1).max(1) + }).strict(), + gates: z.object({ + localBenefit: z.boolean(), + globalNonRegression: z.boolean(), + safety: z.boolean(), + privacy: z.boolean(), + determinism: z.boolean(), + resource: z.boolean(), + passed: z.boolean() + }).strict() +}).strict() + +export const MemoryFeedbackTiebreakerEvidence = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + decisionId: z.string().regex(/^kun-memory-[a-z0-9-]+$/u), + status: z.literal('final'), + artifactHashes: ArtifactHashes.extend({ + decisionPlanSha256: Hash, + candidateLockSha256: Hash + }).strict(), + selectedCandidateId: ArtifactId, + holdoutRunCount: z.literal(1), + development: PartitionEvidence, + holdout: PartitionEvidence, + decision: z.enum(['go', 'no-go']), + reasons: z.array(z.string().min(1)).max(32) +}).strict() + +export type MemoryFeedbackTiebreakerFixture = z.infer +export type MemoryFeedbackTiebreakerManifestValue = z.infer +export type MemoryFeedbackTiebreakerCalibration = z.infer +export type MemoryFeedbackTiebreakerCandidateValue = z.infer +export type MemoryFeedbackTiebreakerLock = z.infer +export type MemoryFeedbackTiebreakerSignalRuleValue = z.infer +export type MemoryFeedbackTiebreakerPlan = z.infer +export type MemoryFeedbackTiebreakerEvidenceValue = z.infer + +export function memoryFeedbackTiebreakerArtifactSha256(value: unknown): string { + return createHash('sha256').update(canonicalJson(value)).digest('hex') +} + +function validateCaseReferences( + item: z.infer, + records: ReadonlyMap, + context: z.RefinementCtx +): void { + reportDuplicates(item.candidateIds, `${item.id} candidate id`, context) + reportDuplicates(item.expectedIds, `${item.id} expected id`, context) + reportDuplicates(item.forbiddenIds, `${item.id} forbidden id`, context) + const candidates = new Set(item.candidateIds) + for (const id of [...item.candidateIds, ...item.expectedIds, ...item.forbiddenIds]) { + if (!records.has(id)) addIssue(context, `${item.id} references unknown memory: ${id}`) + } + for (const id of item.expectedIds) { + if (!candidates.has(id)) addIssue(context, `${item.id} expected memory is not a candidate: ${id}`) + if (item.forbiddenIds.includes(id)) addIssue(context, `${item.id} memory is both expected and forbidden: ${id}`) + } + const pairKeys = item.preferredPairs.map((pair) => `${pair.preferredId}->${pair.otherId}`) + reportDuplicates(pairKeys, `${item.id} preference pair`, context) + for (const pair of item.preferredPairs) { + if (pair.preferredId === pair.otherId || !candidates.has(pair.preferredId) || !candidates.has(pair.otherId)) { + addIssue(context, `${item.id} has an invalid preference pair`) + } + if (!item.expectedIds.includes(pair.preferredId)) { + addIssue(context, `${item.id} preferred memory must be expected: ${pair.preferredId}`) + } + } +} + +function reportDuplicates(values: readonly string[], label: string, context: z.RefinementCtx): void { + if (new Set(values).size !== values.length) addIssue(context, `${label}s must be unique`) +} + +function addIssue(context: z.RefinementCtx, message: string): void { + context.addIssue({ code: z.ZodIssueCode.custom, message }) +} + +function canonicalJson(value: unknown): string { + if (Array.isArray(value)) return `[${value.map(canonicalJson).join(',')}]` + if (value && typeof value === 'object') { + const entries = Object.entries(value as Record) + .sort(([left], [right]) => left < right ? -1 : left > right ? 1 : 0) + return `{${entries.map(([key, item]) => `${JSON.stringify(key)}:${canonicalJson(item)}`).join(',')}}` + } + return JSON.stringify(value) +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-correction.test.ts b/kun/src/memory/memory-feedback-tiebreaker-correction.test.ts new file mode 100644 index 000000000..94d3de5a2 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-correction.test.ts @@ -0,0 +1,81 @@ +import { mkdtemp, readFile, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, expect, it } from 'vitest' +import { MemoryCapabilityConfig } from '../contracts/capabilities.js' +import { MemoryFeedbackCheckpoint, MemoryFeedbackConfig, MemoryFeedbackEvent } from '../contracts/memory-feedback.js' +import { FileMemoryStore } from './memory-store.js' +import { MemoryFeedbackService } from './memory-feedback-service.js' +import { FileMemoryFeedbackStore } from './memory-feedback-store.js' +import type { MemoryFeedbackTiebreakerFixture } from './memory-feedback-tiebreaker-contracts.js' +import { evaluateMemoryFeedbackTiebreakerCase } from './memory-feedback-tiebreaker-evaluator.js' + +const roots: string[] = [] +const NOW = '2026-09-16T00:00:00.000Z' +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +it('preserves replacement evidence through real correction, replay, compaction and restart', async () => { + const root = await mkdtemp(join(tmpdir(), 'kun-tiebreaker-correction-')) + roots.push(root) + // Small internal test segment forces compaction without thousands of writes. + const config = { ...MemoryFeedbackConfig.parse({ enabled: true }), maxSegmentBytes: 1000 } + const memories = new FileMemoryStore({ + rootDir: join(root, 'memory'), config: MemoryCapabilityConfig.parse({ enabled: true }), nowIso: () => NOW + }) + const feedback = new FileMemoryFeedbackStore({ dataDir: root, config, nowIso: () => NOW }) + const service = new MemoryFeedbackService({ + dataDir: root, memoryStore: memories, feedbackStore: feedback, config, nowIso: () => NOW + }) + await memories.createWithId('mem_old', { + content: 'Use twenty second batch timeout', scope: 'workspace', workspace: 'fixture-workspace-a' + }) + const request = { + operationId: 'fixture-correction', memoryId: 'mem_old', access: { workspace: 'fixture-workspace-a' }, + replacement: { content: 'Use forty five second batch timeout' } + } + const result = await service.correct(request) + expect((await service.correct(request)).replayed).toBe(true) + const event = MemoryFeedbackEvent.parse(await feedback.event(result.eventId)) + expect(event).toMatchObject({ kind: 'corrected', memoryId: 'mem_old', replacementMemoryId: result.replacementMemoryId }) + expect(await feedback.aggregate('mem_old')).toMatchObject({ correctionCount: 1 }) + expect(await feedback.aggregate(result.replacementMemoryId)).toBeUndefined() + + const records = await memories.list({ workspace: request.access.workspace, includeDeleted: true }) + const item: MemoryFeedbackTiebreakerFixture['cases'][number] = { + id: 'case_dev_correction_replay', partition: 'development', stratum: 'corrected-replacement-near-tie', + query: 'batch timeout', workspace: request.access.workspace, candidateIds: records.map((record) => record.id), + expectedIds: [result.replacementMemoryId], forbiddenIds: ['mem_old'], preferredPairs: [], + rationale: 'Integration-only replay probe, not frozen decision data.', limit: 1, promptCharacterBudget: 512 + } + const fixture: MemoryFeedbackTiebreakerFixture = { + schemaVersion: 1, datasetId: 'kun-memory-feedback-tiebreaker-anonymous-v1', status: 'frozen', + evaluationNow: NOW, records, events: [event], cases: [item] + } + const before = evaluateMemoryFeedbackTiebreakerCase({ fixture, item, signalRule: 'confirmation-correction', maximumGap: 0 }) + expect(before.selectedIds).toEqual([result.replacementMemoryId]) + expect(before.rankings).toHaveLength(1) + expect(before.rankings[0]).toMatchObject({ memoryId: result.replacementMemoryId, correctionCount: 1 }) + + for (let index = 0; index < 8; index += 1) { + await feedback.append(MemoryFeedbackEvent.parse({ + schemaVersion: 1, id: `fixture-retrieved-${index}`, kind: 'retrieved', memoryId: 'mem_old', + occurredAt: NOW, threadId: 'fixture-thread', turnId: `fixture-turn-${index}` + })) + } + const checkpoint = MemoryFeedbackCheckpoint.parse(JSON.parse( + await readFile(join(root, 'memory-feedback', 'checkpoint.json'), 'utf8') + )) + expect(checkpoint.explicitEvents).toContainEqual(event) + const restarted = new FileMemoryFeedbackStore({ dataDir: root, config }) + await restarted.ready() + expect(await restarted.append(event)).toBe('replayed') + expect(await restarted.aggregate('mem_old')).toMatchObject({ correctionCount: 1 }) + expect(await restarted.aggregate(result.replacementMemoryId)).toBeUndefined() + const replayed = MemoryFeedbackEvent.parse(await restarted.event(result.eventId)) + const after = evaluateMemoryFeedbackTiebreakerCase({ + fixture: { ...fixture, events: [replayed] }, item, signalRule: 'confirmation-correction', maximumGap: 0 + }) + expect(after).toEqual(before) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-determinism.test.ts b/kun/src/memory/memory-feedback-tiebreaker-determinism.test.ts new file mode 100644 index 000000000..02f4c6b01 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-determinism.test.ts @@ -0,0 +1,65 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerDecisionPlan, + MemoryFeedbackTiebreakerDevelopmentArtifact, + memoryFeedbackTiebreakerArtifactSha256 +} from './memory-feedback-tiebreaker-contracts.js' +import { + compactMemoryFeedbackTiebreakerDevelopmentReport, + evaluateMemoryFeedbackTiebreakerDevelopment +} from './memory-feedback-tiebreaker-development.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' + +const fixturePath = (name: string) => fileURLToPath(new URL(`./fixtures/${name}`, import.meta.url)) + +describe('memory feedback tiebreaker deterministic replay', () => { + it('keeps scores, selections, metrics, and compact hashes stable after replay', async () => { + const input = await loadInput() + const first = evaluateMemoryFeedbackTiebreakerDevelopment(input) + const replay = evaluateMemoryFeedbackTiebreakerDevelopment({ + ...input, + fixture: { + ...input.fixture, + records: [...input.fixture.records].reverse(), + events: [...input.fixture.events].reverse() + } + }) + + expect(replay).toEqual(first) + expect(compactMemoryFeedbackTiebreakerDevelopmentReport(replay)) + .toEqual(compactMemoryFeedbackTiebreakerDevelopmentReport(first)) + }) + + it('preserves canonical hashes through JSON serialization and schema parsing', async () => { + const input = await loadInput() + const report = compactMemoryFeedbackTiebreakerDevelopmentReport( + evaluateMemoryFeedbackTiebreakerDevelopment(input) + ) + const serialized = JSON.stringify(report) + const parsed = MemoryFeedbackTiebreakerDevelopmentArtifact.parse(JSON.parse(serialized)) + + expect(parsed).toEqual(report) + expect(memoryFeedbackTiebreakerArtifactSha256(parsed)) + .toBe(memoryFeedbackTiebreakerArtifactSha256(report)) + expect(memoryFeedbackTiebreakerArtifactSha256(JSON.parse(JSON.stringify(parsed)))) + .toBe(memoryFeedbackTiebreakerArtifactSha256(parsed)) + }) +}) + +async function loadInput() { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const [calibrationText, planText] = await Promise.all([ + readFile(fixturePath('memory-feedback-tiebreaker-calibration.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-plan.v1.json'), 'utf8') + ]) + return { + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration: MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(calibrationText)), + plan: MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(planText)) + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-development.test.ts b/kun/src/memory/memory-feedback-tiebreaker-development.test.ts new file mode 100644 index 000000000..286cd9784 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-development.test.ts @@ -0,0 +1,109 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerDevelopmentArtifact, + MemoryFeedbackTiebreakerDecisionPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { + compactMemoryFeedbackTiebreakerDevelopmentReport, + evaluateMemoryFeedbackTiebreakerDevelopment +} from './memory-feedback-tiebreaker-development.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' + +const calibrationPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-calibration.v1.json', + import.meta.url +)) +const planPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-plan.v1.json', + import.meta.url +)) +const developmentPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-development.v1.json', + import.meta.url +)) + +describe('memory feedback tiebreaker development evaluation', () => { + it('evaluates every frozen candidate deterministically without holdout results', async () => { + const input = await loadInput() + const first = evaluateMemoryFeedbackTiebreakerDevelopment(input) + const second = evaluateMemoryFeedbackTiebreakerDevelopment(input) + + expect(first).toEqual(second) + expect(first.candidates).toHaveLength(7) + expect(first.candidates.map((candidate) => candidate.candidateId)) + .toEqual(input.plan.candidates.map((candidate) => candidate.id)) + expect(first.candidates.every((candidate) => candidate.cases.length === 18)).toBe(true) + expect(first.candidates.flatMap((candidate) => candidate.cases) + .every((item) => item.caseId.startsWith('case_dev_'))).toBe(true) + }) + + it('keeps traces bounded to synthetic ids and numeric features', async () => { + const report = evaluateMemoryFeedbackTiebreakerDevelopment(await loadInput()) + const serialized = JSON.stringify(report) + + expect(serialized).not.toContain('"query"') + expect(serialized).not.toContain('"content"') + expect(serialized).not.toContain('"sources"') + expect(report.candidates.every((candidate) => candidate.cases.every((item) => + item.rankings.length <= 64))).toBe(true) + }) + + it('compacts every configuration with a deterministic selected-id digest', async () => { + const input = await loadInput() + const report = evaluateMemoryFeedbackTiebreakerDevelopment(input) + const compact = compactMemoryFeedbackTiebreakerDevelopmentReport(report) + + expect(compact.configurations).toHaveLength(input.plan.candidates.length) + expect(compact.configurations.map((item) => item.candidateId)) + .toEqual(input.plan.candidates.map((item) => item.id)) + expect(compact.configurations.every((item) => /^[a-f0-9]{64}$/u.test(item.selectedIdsSha256))).toBe(true) + }) + + it('matches the frozen development artifact', async () => { + const report = compactMemoryFeedbackTiebreakerDevelopmentReport( + evaluateMemoryFeedbackTiebreakerDevelopment(await loadInput()) + ) + const frozen = MemoryFeedbackTiebreakerDevelopmentArtifact.parse( + JSON.parse(await readFile(developmentPath, 'utf8')) + ) + + expect(frozen).toEqual(report) + }) + + it('does not qualify local gains while explicit forbidden selections remain', async () => { + const report = evaluateMemoryFeedbackTiebreakerDevelopment(await loadInput()) + const compact = compactMemoryFeedbackTiebreakerDevelopmentReport(report) + const improved = compact.configurations.filter((candidate) => candidate.gates.localBenefit) + + expect(improved.map((candidate) => candidate.candidateId)).toEqual([ + 'confirmation-correction-gap_1', + 'confirmation-correction-gap_2' + ]) + for (const candidate of improved) { + expect(candidate.gates.globalNonRegression).toBe(true) + expect(candidate.metrics.explicitForbiddenSelections).toBe(3) + expect(candidate.metrics.authorizationOrLifecycleViolations).toBe(0) + expect(candidate.gates.safety).toBe(false) + expect(candidate.gates.passed).toBe(false) + } + expect(compact.configurations.some((candidate) => candidate.gates.passed)).toBe(false) + }) +}) + +async function loadInput() { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const [calibrationText, planText] = await Promise.all([ + readFile(calibrationPath, 'utf8'), + readFile(planPath, 'utf8') + ]) + return { + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration: MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(calibrationText)), + plan: MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(planText)) + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-development.ts b/kun/src/memory/memory-feedback-tiebreaker-development.ts new file mode 100644 index 000000000..e3510fd26 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-development.ts @@ -0,0 +1,209 @@ +import { memoryFeedbackPairedBootstrapLowerBound } from './memory-feedback-tiebreaker-bootstrap.js' +import { evaluateMemoryFeedbackTiebreakerGates } from './memory-feedback-tiebreaker-gates.js' +import type { + MemoryFeedbackTiebreakerCalibration, + MemoryFeedbackTiebreakerCandidateValue, + MemoryFeedbackTiebreakerDevelopmentArtifactValue, + MemoryFeedbackTiebreakerFixture, + MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { memoryFeedbackTiebreakerArtifactSha256 } from './memory-feedback-tiebreaker-contracts.js' +import { + evaluateMemoryFeedbackTiebreakerCase, + type MemoryFeedbackTiebreakerCaseResult +} from './memory-feedback-tiebreaker-evaluator.js' +import { rankMemoryFeedbackTiebreakerFoundationCase } from './memory-feedback-tiebreaker-foundation.js' +import { + scoreMemoryFeedbackTiebreakerCases, + type MemoryFeedbackTiebreakerCaseMetrics, + type MemoryFeedbackTiebreakerMetrics +} from './memory-feedback-tiebreaker-metrics.js' + +type LowerBounds = { + pairAccuracyGain: number + recallGain: number + mrrGain: number +} + +export type MemoryFeedbackTiebreakerDevelopmentCandidate = { + candidateId: string + boundaryId: string | null + maximumGap: number + signalRule: MemoryFeedbackTiebreakerCandidateValue['signalRule'] + metrics: MemoryFeedbackTiebreakerMetrics + bootstrapLowerBounds: LowerBounds + gates: { + localBenefit: boolean + globalNonRegression: boolean + safety: boolean + passed: boolean + } + cases: MemoryFeedbackTiebreakerCaseResult[] +} + +export type MemoryFeedbackTiebreakerDevelopmentReport = { + schemaVersion: 1 + evaluationVersion: 'p3-feedback-tiebreaker-v1' + partition: 'development' + artifactHashes: { + fixtureSha256: string + manifestSha256: string + calibrationSha256: string + decisionPlanSha256: string + } + foundation: { + metrics: MemoryFeedbackTiebreakerMetrics + cases: MemoryFeedbackTiebreakerCaseResult[] + } + candidates: MemoryFeedbackTiebreakerDevelopmentCandidate[] +} + +export function evaluateMemoryFeedbackTiebreakerDevelopment(input: { + fixture: MemoryFeedbackTiebreakerFixture + fixtureSha256: string + manifestSha256: string + calibration: MemoryFeedbackTiebreakerCalibration + plan: MemoryFeedbackTiebreakerPlan +}): MemoryFeedbackTiebreakerDevelopmentReport { + const cases = input.fixture.cases.filter((item) => item.partition === 'development') + const foundationCases = cases.map((item) => { + const foundation = rankMemoryFeedbackTiebreakerFoundationCase(input.fixture, item) + return { + caseId: item.id, + signalRule: 'foundation-only' as const, + maximumGap: 0, + rankings: foundation.rankings.map((ranking) => ({ + memoryId: ranking.memoryId, + foundationScore: ranking.foundationScore, + confirmationCount: 0, + correctionCount: 0, + retrievalCount: 0, + selected: ranking.selected + })), + selectedIds: foundation.selectedIds, + forbiddenSelectedIds: foundation.selectedIds.filter((id) => item.forbiddenIds.includes(id)), + selectedCharacters: foundation.trace.selectedCharacters + } + }) + const foundationScored = scoreMemoryFeedbackTiebreakerCases({ fixture: input.fixture, results: foundationCases }) + const candidates = input.plan.candidates.map((candidate) => evaluateCandidate({ + fixture: input.fixture, + cases, + candidate, + calibration: input.calibration, + plan: input.plan, + foundationMetrics: foundationScored.metrics, + foundationCases: foundationScored.cases + })) + + return { + schemaVersion: 1, + evaluationVersion: 'p3-feedback-tiebreaker-v1', + partition: 'development', + artifactHashes: { + fixtureSha256: input.fixtureSha256, + manifestSha256: input.manifestSha256, + calibrationSha256: memoryFeedbackTiebreakerArtifactSha256(input.calibration), + decisionPlanSha256: memoryFeedbackTiebreakerArtifactSha256(input.plan) + }, + foundation: { metrics: foundationScored.metrics, cases: foundationCases }, + candidates + } +} + +export function compactMemoryFeedbackTiebreakerDevelopmentReport( + report: MemoryFeedbackTiebreakerDevelopmentReport +): MemoryFeedbackTiebreakerDevelopmentArtifactValue { + return { + schemaVersion: 1, + evaluationVersion: report.evaluationVersion, + partition: 'development', + artifactHashes: report.artifactHashes, + foundation: report.foundation.metrics, + configurations: report.candidates.map((candidate) => ({ + candidateId: candidate.candidateId, + boundaryId: candidate.boundaryId, + maximumGap: candidate.maximumGap, + signalRule: candidate.signalRule, + metrics: candidate.metrics, + bootstrapLowerBounds: candidate.bootstrapLowerBounds, + gates: candidate.gates, + selectedIdsSha256: memoryFeedbackTiebreakerArtifactSha256(candidate.cases.map((item) => ({ + caseId: item.caseId, + selectedIds: item.selectedIds + }))) + })) + } +} + +function evaluateCandidate(input: { + fixture: MemoryFeedbackTiebreakerFixture + cases: MemoryFeedbackTiebreakerFixture['cases'] + candidate: MemoryFeedbackTiebreakerCandidateValue + calibration: MemoryFeedbackTiebreakerCalibration + plan: MemoryFeedbackTiebreakerPlan + foundationMetrics: MemoryFeedbackTiebreakerMetrics + foundationCases: MemoryFeedbackTiebreakerCaseMetrics[] +}): MemoryFeedbackTiebreakerDevelopmentCandidate { + const maximumGap = input.candidate.boundaryId === null + ? 0 + : input.calibration.boundaryGrid.find((boundary) => boundary.id === input.candidate.boundaryId)?.maximumGap + if (maximumGap === undefined) throw new Error(`unknown tiebreaker boundary: ${input.candidate.boundaryId}`) + const cases = input.cases.map((item) => evaluateMemoryFeedbackTiebreakerCase({ + fixture: input.fixture, + item, + signalRule: input.candidate.signalRule, + maximumGap + })) + const scored = scoreMemoryFeedbackTiebreakerCases({ fixture: input.fixture, results: cases }) + const lowerBounds = bootstrapLowerBounds(input.foundationCases, scored.cases, input.plan) + const gates = evaluateMemoryFeedbackTiebreakerGates(input.foundationMetrics, { + metrics: scored.metrics, bootstrapLowerBounds: lowerBounds + }, input.plan) + + return { + candidateId: input.candidate.id, + boundaryId: input.candidate.boundaryId, + maximumGap, + signalRule: input.candidate.signalRule, + metrics: scored.metrics, + bootstrapLowerBounds: lowerBounds, + gates, + cases + } +} + +function bootstrapLowerBounds( + foundation: readonly MemoryFeedbackTiebreakerCaseMetrics[], + candidate: readonly MemoryFeedbackTiebreakerCaseMetrics[], + plan: MemoryFeedbackTiebreakerPlan +): LowerBounds { + return { + pairAccuracyGain: bootstrapMetric(foundation, candidate, 'pairCorrect', plan), + recallGain: bootstrapMetric(foundation, candidate, 'recallAtK', plan), + mrrGain: bootstrapMetric(foundation, candidate, 'reciprocalRank', plan) + } +} + +function bootstrapMetric( + foundation: readonly MemoryFeedbackTiebreakerCaseMetrics[], + candidate: readonly MemoryFeedbackTiebreakerCaseMetrics[], + key: 'pairCorrect' | 'recallAtK' | 'reciprocalRank', + plan: MemoryFeedbackTiebreakerPlan +): number { + const candidateById = new Map(candidate.map((item) => [item.caseId, item])) + const pairs = foundation.flatMap((item) => { + const right = candidateById.get(item.caseId) + const leftValue = item[key] + const rightValue = right?.[key] + return leftValue === undefined || rightValue === undefined ? [] : [[leftValue, rightValue] as const] + }) + if (pairs.length === 0) return 0 + return memoryFeedbackPairedBootstrapLowerBound({ + baseline: pairs.map(([left]) => left), + candidate: pairs.map(([, right]) => right), + seed: plan.bootstrap.seed, + resamples: plan.bootstrap.resamples, + confidenceLevel: plan.bootstrap.confidenceLevel + }).lowerBound +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-evaluator.test.ts b/kun/src/memory/memory-feedback-tiebreaker-evaluator.test.ts new file mode 100644 index 000000000..ed6085454 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-evaluator.test.ts @@ -0,0 +1,99 @@ +import { describe, expect, it } from 'vitest' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { + evaluateMemoryFeedbackTiebreakerCase, + groupMemoryFeedbackTiebreakerNearTies +} from './memory-feedback-tiebreaker-evaluator.js' +import { rankMemoryFeedbackTiebreakerFoundationCase } from './memory-feedback-tiebreaker-foundation.js' + +describe('memory feedback near-tie evaluator', () => { + it('uses an inclusive leader-relative boundary and bounded groups', () => { + const rankings = [ + { id: 'a', foundationScore: 0.8 }, + { id: 'b', foundationScore: 0.7 }, + { id: 'c', foundationScore: 0.69 }, + { id: 'd', foundationScore: 0.68 } + ] + + expect(groupMemoryFeedbackTiebreakerNearTies(rankings, 0.1, 3)) + .toEqual([[rankings[0], rankings[1]], [rankings[2], rankings[3]]]) + expect(groupMemoryFeedbackTiebreakerNearTies(rankings, 1, 2)) + .toEqual([[rankings[0], rankings[1]], [rankings[2], rankings[3]]]) + expect(groupMemoryFeedbackTiebreakerNearTies(rankings, 0, 3)) + .toEqual(rankings.map((item) => [item])) + }) + + it('reorders only admitted candidates and preserves abstention, scope, lifecycle, and budgets', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const cases = fixture.cases.filter((item) => item.partition === 'development') + + for (const item of cases) { + const foundation = rankMemoryFeedbackTiebreakerFoundationCase(fixture, item) + const candidate = evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item, + signalRule: 'confirmation-correction', + maximumGap: 1 + }) + + expect(new Set(candidate.rankings.map((ranking) => ranking.memoryId))) + .toEqual(new Set(foundation.rankings.map((ranking) => ranking.memoryId))) + expect(candidate.selectedIds.length).toBeLessThanOrEqual(item.limit) + expect(candidate.selectedCharacters).toBeLessThanOrEqual(item.promptCharacterBudget) + if (item.stratum === 'no-result-control') expect(candidate.selectedIds).toEqual([]) + if (item.stratum === 'scope-safety' || item.stratum === 'lifecycle-safety') { + expect(candidate.forbiddenSelectedIds).toEqual([]) + } + } + }) + + it('applies explicit confirmation and correction only inside the declared window', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + for (const stratum of ['confirmed-near-tie', 'corrected-replacement-near-tie'] as const) { + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === stratum)! + const signalRule = stratum === 'confirmed-near-tie' ? 'confirmation' : 'confirmation-correction' + const candidate = evaluateMemoryFeedbackTiebreakerCase({ fixture, item, signalRule, maximumGap: 1 }) + + expect(candidate.selectedIds).toEqual(item.expectedIds) + } + + for (const stratum of ['no-feedback-control', 'wide-gap-control'] as const) { + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === stratum)! + const foundation = rankMemoryFeedbackTiebreakerFoundationCase(fixture, item) + const candidate = evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item, + signalRule: 'confirmation-correction', + maximumGap: 0 + }) + + expect(candidate.selectedIds).toEqual(foundation.selectedIds) + } + }) + + it('keeps retrieval frequency shadow-only', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === 'frequent-unconfirmed-control')! + const withFrequency = evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item, + signalRule: 'confirmation', + maximumGap: 1 + }) + const withoutFrequency = evaluateMemoryFeedbackTiebreakerCase({ + fixture: { ...fixture, events: fixture.events.filter((event) => event.kind !== 'retrieved') }, + item, + signalRule: 'confirmation', + maximumGap: 1 + }) + + expect(withFrequency.selectedIds).toEqual(withoutFrequency.selectedIds) + expect(withFrequency.rankings.map((ranking) => ranking.memoryId)) + .toEqual(withoutFrequency.rankings.map((ranking) => ranking.memoryId)) + expect(withFrequency.rankings.some((ranking) => ranking.retrievalCount > 0)).toBe(true) + expect(withoutFrequency.rankings.every((ranking) => ranking.retrievalCount === 0)).toBe(true) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-evaluator.ts b/kun/src/memory/memory-feedback-tiebreaker-evaluator.ts new file mode 100644 index 000000000..6c5975ddc --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-evaluator.ts @@ -0,0 +1,145 @@ +import type { MemoryFeedbackEvent } from '../contracts/memory-feedback.js' +import type { MemoryRecord } from '../contracts/memory.js' +import { applyMemoryContextBudget } from './memory-retrieval-trace.js' +import { + MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE, + compareMemoryFeedbackTiebreakerItems, + type MemoryFeedbackTiebreakerItem +} from './memory-feedback-tiebreaker-calibration.js' +import type { + MemoryFeedbackTiebreakerFixture, + MemoryFeedbackTiebreakerSignalRuleValue +} from './memory-feedback-tiebreaker-contracts.js' +import { rankMemoryFeedbackTiebreakerFoundationCase } from './memory-feedback-tiebreaker-foundation.js' + +type FixtureCase = MemoryFeedbackTiebreakerFixture['cases'][number] + +export type MemoryFeedbackTiebreakerRanking = MemoryFeedbackTiebreakerItem & { + selected: boolean +} + +export type MemoryFeedbackTiebreakerCaseResult = { + caseId: string + signalRule: MemoryFeedbackTiebreakerSignalRuleValue + maximumGap: number + rankings: MemoryFeedbackTiebreakerRanking[] + selectedIds: string[] + forbiddenSelectedIds: string[] + selectedCharacters: number +} + +export function evaluateMemoryFeedbackTiebreakerCase(input: { + fixture: MemoryFeedbackTiebreakerFixture + item: FixtureCase + signalRule: MemoryFeedbackTiebreakerSignalRuleValue + maximumGap: number +}): MemoryFeedbackTiebreakerCaseResult { + const foundation = rankMemoryFeedbackTiebreakerFoundationCase(input.fixture, input.item) + const records = new Map(input.fixture.records.map((record) => [record.id, record])) + const signals = buildSignals(input.fixture.events) + const items = foundation.rankings.map((ranking) => ({ + memoryId: ranking.memoryId, + foundationScore: ranking.foundationScore, + ...signals.get(ranking.memoryId) ?? emptySignals() + })) + const ordered = input.signalRule === 'foundation-only' + ? items + : groupMemoryFeedbackTiebreakerNearTies( + items, + input.maximumGap, + MEMORY_FEEDBACK_TIEBREAKER_CALIBRATION_PROCEDURE.maximumReorderWindow + ).flatMap((group) => [...group].sort((left, right) => + compareMemoryFeedbackTiebreakerItems(left, right, input.signalRule))) + const ranked = ordered.map((item) => toRankedMemory(item, foundation.trace, records)) + const selected = applyMemoryContextBudget( + ranked, + input.item.limit, + input.item.promptCharacterBudget, + Date.parse(input.fixture.evaluationNow) + ) + const selectedIds = selected.records.map((record) => record.id) + const selectedSet = new Set(selectedIds) + + return { + caseId: input.item.id, + signalRule: input.signalRule, + maximumGap: input.maximumGap, + rankings: ordered.map((item) => ({ ...item, selected: selectedSet.has(item.memoryId) })), + selectedIds, + forbiddenSelectedIds: selectedIds.filter((id) => input.item.forbiddenIds.includes(id)), + selectedCharacters: selected.selectedCharacters + } +} + +export function groupMemoryFeedbackTiebreakerNearTies(rankings: readonly T[], maximumGap: number, maximumWindow: number): T[][] { + const groups: T[][] = [] + let index = 0 + while (index < rankings.length) { + const leader = rankings[index]! + const group = [leader] + index += 1 + while (index < rankings.length && group.length < maximumWindow) { + const candidate = rankings[index]! + if (round(leader.foundationScore - candidate.foundationScore) > maximumGap) break + group.push(candidate) + index += 1 + } + groups.push(group) + } + return groups +} + +function buildSignals(events: readonly MemoryFeedbackEvent[]): Map> { + const signals = new Map>() + const get = (id: string): ReturnType => { + const existing = signals.get(id) + if (existing) return existing + const created = emptySignals() + signals.set(id, created) + return created + } + for (const event of events) { + if (event.kind === 'retrieved') { + const signal = get(event.memoryId) + signal.retrievalCount += 1 + if (!signal.lastRetrievedAt || signal.lastRetrievedAt < event.occurredAt) { + signal.lastRetrievedAt = event.occurredAt + } + } + if (event.kind === 'confirmed') get(event.memoryId).confirmationCount += 1 + // Evaluator-only replacement evidence, NOT the production aggregate's + // correctionCount (which counts corrections of the old event.memoryId). + // Keep the frozen trace field name for v1 evidence compatibility. + if (event.kind === 'corrected') get(event.replacementMemoryId).correctionCount += 1 + } + return signals +} + +function emptySignals(): { + confirmationCount: number + correctionCount: number + retrievalCount: number + lastRetrievedAt?: string +} { + return { confirmationCount: 0, correctionCount: 0, retrievalCount: 0 } +} + +function toRankedMemory( + item: MemoryFeedbackTiebreakerItem, + trace: ReturnType['trace'], + records: ReadonlyMap +) { + const ranking = trace.rankings.find((candidate) => candidate.memoryId === item.memoryId) + const record = records.get(item.memoryId) + if (!ranking || !record) throw new Error(`missing admitted tiebreaker memory: ${item.memoryId}`) + return { record, channel: ranking.channel, features: ranking.features } +} + +function round(value: number): number { + return Math.round(value * 1_000_000) / 1_000_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.test.ts b/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.test.ts new file mode 100644 index 000000000..769ba2adf --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.test.ts @@ -0,0 +1,87 @@ +import { readFile } from 'node:fs/promises' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerFixtureFile, + MemoryFeedbackTiebreakerManifest +} from './memory-feedback-tiebreaker-contracts.js' +import { + DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS, + loadMemoryFeedbackTiebreakerDataset, + memoryFeedbackTiebreakerFileSha256, + parseMemoryFeedbackTiebreakerDataset +} from './memory-feedback-tiebreaker-fixture-loader.js' + +describe('memory feedback tiebreaker fixture loader', () => { + it('validates frozen hashes, quotas, labels, privacy, and split isolation', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + + expect(dataset.fixture.cases.filter((item) => item.partition === 'development')).toHaveLength(18) + expect(dataset.fixture.cases.filter((item) => item.partition === 'holdout')).toHaveLength(18) + expect(dataset.sourceHashes.fixture).toBe(dataset.manifest.fixtureSha256) + }) + + it('invalidates the manifest after a one-byte fixture change', async () => { + const input = await loadTexts() + + expect(() => parseMemoryFeedbackTiebreakerDataset({ + ...input, + fixtureText: input.fixtureText.replace('compact tables', 'compact tablez') + })).toThrow(/fixture checksum mismatch/u) + }) + + it('rejects normalized duplicates across development and holdout', async () => { + const input = await loadTexts() + const fixture = MemoryFeedbackTiebreakerFixtureFile.parse(JSON.parse(input.fixtureText)) + const manifest = MemoryFeedbackTiebreakerManifest.parse(JSON.parse(input.manifestText)) + const development = fixture.cases.find((item) => item.partition === 'development')! + const holdout = fixture.cases.find((item) => item.partition === 'holdout')! + holdout.query = development.query.toUpperCase() + + expect(() => parseMemoryFeedbackTiebreakerDataset(seal(fixture, manifest))) + .toThrow(/cross-split near-duplicates/u) + }) + + it('rejects ambiguous labels before retrieval can execute', async () => { + const input = await loadTexts() + const fixture = MemoryFeedbackTiebreakerFixtureFile.parse(JSON.parse(input.fixtureText)) + const item = fixture.cases.find((candidate) => candidate.stratum !== 'no-result-control')! + item.forbiddenIds.push(item.expectedIds[0]!) + + expect(() => parseMemoryFeedbackTiebreakerDataset({ + ...input, + fixtureText: JSON.stringify(fixture) + })).toThrow(/both expected and forbidden/u) + }) +}) + +async function loadTexts(): Promise<{ + fixtureText: string + manifestText: string + checksumsText: string +}> { + const [fixtureText, manifestText, checksumsText] = await Promise.all([ + readFile(DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS.fixture, 'utf8'), + readFile(DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS.manifest, 'utf8'), + readFile(DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS.checksums, 'utf8') + ]) + return { fixtureText, manifestText, checksumsText } +} + +function seal( + fixture: ReturnType, + manifest: ReturnType +): { fixtureText: string; manifestText: string; checksumsText: string } { + const fixtureText = `${JSON.stringify(fixture, null, 2)}\n` + manifest.fixtureSha256 = memoryFeedbackTiebreakerFileSha256(fixtureText) + const manifestText = `${JSON.stringify(manifest, null, 2)}\n` + const checksumsText = JSON.stringify({ + schemaVersion: 1, + evaluationVersion: 'p3-feedback-tiebreaker-v1', + algorithm: 'sha256', + files: { + fixture: manifest.fixtureSha256, + manifest: memoryFeedbackTiebreakerFileSha256(manifestText) + } + }) + return { fixtureText, manifestText, checksumsText } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.ts b/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.ts new file mode 100644 index 000000000..0fa66c8fe --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-fixture-loader.ts @@ -0,0 +1,183 @@ +import { createHash } from 'node:crypto' +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { z } from 'zod' +import { type MemoryRecord } from '../contracts/memory.js' +import { memoryInScope, memoryLifecycleState } from './memory-ranking.js' +import { + MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION, + MemoryFeedbackTiebreakerFixtureFile, + MemoryFeedbackTiebreakerManifest, + type MemoryFeedbackTiebreakerFixture, + type MemoryFeedbackTiebreakerManifestValue +} from './memory-feedback-tiebreaker-contracts.js' + +const Hash = z.string().regex(/^[a-f0-9]{64}$/u) + +const ChecksumsFile = z.object({ + schemaVersion: z.literal(1), + evaluationVersion: z.literal(MEMORY_FEEDBACK_TIEBREAKER_EVALUATION_VERSION), + algorithm: z.literal('sha256'), + files: z.object({ fixture: Hash, manifest: Hash }).strict() +}).strict() + +export type MemoryFeedbackTiebreakerDataset = { + fixture: MemoryFeedbackTiebreakerFixture + manifest: MemoryFeedbackTiebreakerManifestValue + sourceHashes: z.infer['files'] +} + +export const DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS = Object.freeze({ + fixture: fileURLToPath(new URL('./fixtures/memory-feedback-tiebreaker-fixtures.v1.json', import.meta.url)), + manifest: fileURLToPath(new URL('./fixtures/memory-feedback-tiebreaker-manifest.v1.json', import.meta.url)), + checksums: fileURLToPath(new URL('./fixtures/memory-feedback-tiebreaker-checksums.v1.json', import.meta.url)) +}) + +export async function loadMemoryFeedbackTiebreakerDataset( + paths = DEFAULT_MEMORY_FEEDBACK_TIEBREAKER_PATHS +): Promise { + const [fixtureText, manifestText, checksumsText] = await Promise.all([ + readFile(paths.fixture, 'utf8'), + readFile(paths.manifest, 'utf8'), + readFile(paths.checksums, 'utf8') + ]) + return parseMemoryFeedbackTiebreakerDataset({ fixtureText, manifestText, checksumsText }) +} + +export function parseMemoryFeedbackTiebreakerDataset(input: { + fixtureText: string + manifestText: string + checksumsText: string +}): MemoryFeedbackTiebreakerDataset { + const fixture = MemoryFeedbackTiebreakerFixtureFile.parse(parseJson(input.fixtureText, 'fixture')) + const manifest = MemoryFeedbackTiebreakerManifest.parse(parseJson(input.manifestText, 'manifest')) + const checksums = ChecksumsFile.parse(parseJson(input.checksumsText, 'checksums')) + const sourceHashes = { + fixture: memoryFeedbackTiebreakerFileSha256(input.fixtureText), + manifest: memoryFeedbackTiebreakerFileSha256(input.manifestText) + } + const errors: string[] = [] + + requireEqual(sourceHashes.fixture, checksums.files.fixture, 'fixture checksum', errors) + requireEqual(sourceHashes.manifest, checksums.files.manifest, 'manifest checksum', errors) + requireEqual(sourceHashes.fixture, manifest.fixtureSha256, 'fixture manifest hash', errors) + requireEqual(fixture.evaluationNow, manifest.evaluationNow, 'evaluationNow', errors) + validateCounts(fixture, manifest, errors) + validateLabels(fixture, errors) + validateSplitIsolation(fixture, manifest, errors) + validateAnonymousContent(fixture, errors) + + if (errors.length > 0) { + throw new Error(`invalid memory feedback tiebreaker dataset: ${errors.join('; ')}`) + } + return { fixture, manifest, sourceHashes } +} + +export function memoryFeedbackTiebreakerFileSha256(text: string): string { + return createHash('sha256').update(text.replace(/\r\n?/gu, '\n')).digest('hex') +} + +function validateCounts( + fixture: MemoryFeedbackTiebreakerFixture, + manifest: MemoryFeedbackTiebreakerManifestValue, + errors: string[] +): void { + for (const partition of ['development', 'holdout'] as const) { + const actual = fixture.cases.filter((item) => item.partition === partition).length + requireEqual(actual, manifest.splitCounts[partition], `${partition} count`, errors) + } + for (const quota of manifest.stratumQuotas) { + for (const partition of ['development', 'holdout'] as const) { + const actual = fixture.cases.filter( + (item) => item.partition === partition && item.stratum === quota.stratum + ).length + requireEqual(actual, quota[partition], `${partition} ${quota.stratum} quota`, errors) + } + } +} + +function validateLabels(fixture: MemoryFeedbackTiebreakerFixture, errors: string[]): void { + const records = new Map(fixture.records.map((record) => [record.id, record])) + const nowMs = Date.parse(fixture.evaluationNow) + for (const item of fixture.cases) { + const noResult = item.stratum === 'no-result-control' + if (noResult !== (item.expectedIds.length === 0)) { + errors.push(`${item.id} no-result label mismatch`) + } + if (noResult && item.preferredPairs.length > 0) errors.push(`${item.id} no-result case has a preference`) + if (!noResult && item.preferredPairs.length === 0) errors.push(`${item.id} lacks a preferred pair`) + for (const id of item.expectedIds) { + const record = records.get(id) + if (record && !isAvailable(record, item.workspace, nowMs)) { + errors.push(`${item.id} expects unavailable memory ${id}`) + } + } + } +} + +function isAvailable(record: MemoryRecord, workspace: string, nowMs: number): boolean { + return memoryInScope(record, { workspace }) && memoryLifecycleState(record, nowMs) === 'active' +} + +function validateSplitIsolation( + fixture: MemoryFeedbackTiebreakerFixture, + manifest: MemoryFeedbackTiebreakerManifestValue, + errors: string[] +): void { + const development = fixture.cases.filter((item) => item.partition === 'development') + const holdout = fixture.cases.filter((item) => item.partition === 'holdout') + for (const left of development) { + for (const right of holdout) { + const leftTokens = normalizedTokens(left.query) + const rightTokens = normalizedTokens(right.query) + const similarity = tokenJaccard(leftTokens, rightTokens) + if (similarity >= manifest.normalization.nearDuplicateTokenJaccard) { + errors.push(`${left.id} and ${right.id} are cross-split near-duplicates (${similarity.toFixed(3)})`) + } + } + } +} + +function normalizedTokens(value: string): ReadonlySet { + const normalized = value.normalize('NFKC').toLocaleLowerCase('und') + return new Set(normalized.match(/[\p{L}\p{N}_]+/gu) ?? []) +} + +function tokenJaccard(left: ReadonlySet, right: ReadonlySet): number { + const intersection = [...left].filter((token) => right.has(token)).length + const union = new Set([...left, ...right]).size + return union === 0 ? 1 : intersection / union +} + +function validateAnonymousContent(fixture: MemoryFeedbackTiebreakerFixture, errors: string[]): void { + const values = [ + ...fixture.records.map((record) => record.content), + ...fixture.cases.flatMap((item) => [item.query, item.rationale, ...item.preferredPairs.map((pair) => pair.rationale)]) + ] + const forbiddenPatterns = [ + /(?:^|\s)[a-z]:[\\/]/iu, + /\\\\[^\\\s]+\\[^\s]+/u, + /file:\/\//iu, + /(?:^|\s)\/(?:users|home|var|tmp|etc)\//iu, + /\b(?:api[_-]?key|token|secret|password)\s*[:=]\s*\S+/iu, + /\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/iu + ] + for (const value of values) { + if (forbiddenPatterns.some((pattern) => pattern.test(value))) { + errors.push('fixture contains a machine path, credential, or personal identifier') + return + } + } +} + +function parseJson(text: string, label: string): unknown { + try { + return JSON.parse(text) as unknown + } catch { + throw new Error(`invalid memory feedback tiebreaker ${label} JSON`) + } +} + +function requireEqual(actual: unknown, expected: unknown, label: string, errors: string[]): void { + if (actual !== expected) errors.push(`${label} mismatch`) +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-fixtures.test.ts b/kun/src/memory/memory-feedback-tiebreaker-fixtures.test.ts new file mode 100644 index 000000000..bedb0d79d --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-fixtures.test.ts @@ -0,0 +1,41 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { MemoryFeedbackEvent } from '../contracts/memory-feedback.js' +import { MemoryRecord } from '../contracts/memory.js' +import { MemoryFeedbackTiebreakerFixtureFile } from './memory-feedback-tiebreaker-contracts.js' + +const fixturePath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-fixtures.v1.json', + import.meta.url +)) + +describe('memory feedback tiebreaker fixtures', () => { + it('uses only production-valid synthetic records and feedback events', async () => { + const fixture = MemoryFeedbackTiebreakerFixtureFile.parse( + JSON.parse(await readFile(fixturePath, 'utf8')) + ) + + expect(fixture.records).toHaveLength(19) + expect(fixture.events).toHaveLength(13) + expect(fixture.records.every((record) => MemoryRecord.safeParse(record).success)).toBe(true) + expect(fixture.events.every((event) => MemoryFeedbackEvent.safeParse(event).success)).toBe(true) + }) + + it('freezes balanced development and holdout labels for every stratum', async () => { + const fixture = MemoryFeedbackTiebreakerFixtureFile.parse( + JSON.parse(await readFile(fixturePath, 'utf8')) + ) + const counts = new Map() + + for (const item of fixture.cases) { + const key = `${item.partition}:${item.stratum}` + counts.set(key, (counts.get(key) ?? 0) + 1) + expect(item.rationale.length).toBeGreaterThan(0) + for (const pair of item.preferredPairs) expect(pair.rationale.length).toBeGreaterThan(0) + } + + expect(fixture.cases).toHaveLength(36) + expect([...counts.values()]).toEqual(Array.from({ length: 18 }, () => 2)) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-foundation.test.ts b/kun/src/memory/memory-feedback-tiebreaker-foundation.test.ts new file mode 100644 index 000000000..f0ef72744 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-foundation.test.ts @@ -0,0 +1,52 @@ +import { describe, expect, it } from 'vitest' +import { DEFAULT_KUN_CAPABILITIES_CONFIG } from '../contracts/capabilities.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { + rankMemoryFeedbackTiebreakerDevelopment, + rankMemoryFeedbackTiebreakerFoundationCase +} from './memory-feedback-tiebreaker-foundation.js' +import { retrieveMemoryRecords } from './memory-retrieval.js' + +describe('memory feedback tiebreaker lexical foundation', () => { + it('matches direct post-#1308 foundation retrieval for every development case', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const results = rankMemoryFeedbackTiebreakerDevelopment(fixture) + const records = new Map(fixture.records.map((record) => [record.id, record])) + + expect(results).toHaveLength(18) + for (const item of fixture.cases.filter((candidate) => candidate.partition === 'development')) { + const actual = results.find((result) => result.caseId === item.id)! + const direct = retrieveMemoryRecords({ + records: item.candidateIds.map((id) => records.get(id)!), + request: { + query: item.query, + workspace: item.workspace, + limit: item.limit, + promptCharacterBudget: item.promptCharacterBudget + }, + policy: { ...DEFAULT_KUN_CAPABILITIES_CONFIG.memory, enabled: true }, + mode: 'filesystem-fallback', + nowIso: fixture.evaluationNow, + relevanceMode: 'foundation-v1' + }) + + expect(actual.selectedIds).toEqual(direct.records.map((record) => record.id)) + expect(actual.rankings).toEqual(direct.trace.rankings.map((ranking) => ({ + memoryId: ranking.memoryId, + foundationScore: ranking.features.finalScore, + selected: ranking.selected + }))) + } + }) + + it('keeps no-result controls empty and emits non-negative adjacent gaps', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + + for (const item of fixture.cases.filter((candidate) => + candidate.partition === 'development' && candidate.stratum === 'no-result-control')) { + const result = rankMemoryFeedbackTiebreakerFoundationCase(fixture, item) + expect(result.selectedIds).toEqual([]) + expect(result.gaps.every((gap) => gap.gap >= 0)).toBe(true) + } + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-foundation.ts b/kun/src/memory/memory-feedback-tiebreaker-foundation.ts new file mode 100644 index 000000000..505496d00 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-foundation.ts @@ -0,0 +1,78 @@ +import { DEFAULT_KUN_CAPABILITIES_CONFIG } from '../contracts/capabilities.js' +import type { MemoryRecord, MemoryRetrievalTrace } from '../contracts/memory.js' +import type { MemoryFeedbackTiebreakerFixture } from './memory-feedback-tiebreaker-contracts.js' +import { retrieveMemoryRecords } from './memory-retrieval.js' + +type FixtureCase = MemoryFeedbackTiebreakerFixture['cases'][number] + +export type MemoryFeedbackTiebreakerFoundationRanking = { + memoryId: string + foundationScore: number + selected: boolean +} + +export type MemoryFeedbackTiebreakerFoundationGap = { + higherId: string + lowerId: string + gap: number +} + +export type MemoryFeedbackTiebreakerFoundationResult = { + caseId: string + rankings: MemoryFeedbackTiebreakerFoundationRanking[] + selectedIds: string[] + gaps: MemoryFeedbackTiebreakerFoundationGap[] + trace: MemoryRetrievalTrace +} + +export function rankMemoryFeedbackTiebreakerFoundationCase( + fixture: MemoryFeedbackTiebreakerFixture, + item: FixtureCase +): MemoryFeedbackTiebreakerFoundationResult { + const recordsById = new Map(fixture.records.map((record) => [record.id, record])) + const records = item.candidateIds + .map((id) => recordsById.get(id)) + .filter((record): record is MemoryRecord => record !== undefined) + const result = retrieveMemoryRecords({ + records, + request: { + query: item.query, + workspace: item.workspace, + limit: item.limit, + promptCharacterBudget: item.promptCharacterBudget + }, + policy: { ...DEFAULT_KUN_CAPABILITIES_CONFIG.memory, enabled: true }, + mode: 'filesystem-fallback', + nowIso: fixture.evaluationNow, + relevanceMode: 'foundation-v1' + }) + const rankings = result.trace.rankings.map((ranking) => ({ + memoryId: ranking.memoryId, + foundationScore: ranking.features.finalScore, + selected: ranking.selected + })) + + return { + caseId: item.id, + rankings, + selectedIds: result.records.map((record) => record.id), + gaps: rankings.slice(1).map((ranking, index) => ({ + higherId: rankings[index]!.memoryId, + lowerId: ranking.memoryId, + gap: round(rankings[index]!.foundationScore - ranking.foundationScore) + })), + trace: result.trace + } +} + +export function rankMemoryFeedbackTiebreakerDevelopment( + fixture: MemoryFeedbackTiebreakerFixture +): MemoryFeedbackTiebreakerFoundationResult[] { + return fixture.cases + .filter((item) => item.partition === 'development') + .map((item) => rankMemoryFeedbackTiebreakerFoundationCase(fixture, item)) +} + +function round(value: number): number { + return Math.round(value * 1_000_000) / 1_000_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-gates.ts b/kun/src/memory/memory-feedback-tiebreaker-gates.ts new file mode 100644 index 000000000..af72ddf98 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-gates.ts @@ -0,0 +1,24 @@ +import type { + MemoryFeedbackTiebreakerDevelopmentArtifactValue, + MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' + +type Configuration = MemoryFeedbackTiebreakerDevelopmentArtifactValue['configurations'][number] + +export function evaluateMemoryFeedbackTiebreakerGates( + foundation: MemoryFeedbackTiebreakerDevelopmentArtifactValue['foundation'], + candidate: Pick, + plan: MemoryFeedbackTiebreakerPlan +): Configuration['gates'] { + const { metrics, bootstrapLowerBounds: bounds } = candidate + const { localBenefit: local, globalNonRegression: global, safety: safe } = plan.gates + const localBenefit = metrics.pairAccuracy - foundation.pairAccuracy >= local.minimumPairAccuracyGain && + bounds.pairAccuracyGain >= local.minimumPairAccuracyGainLowerBound + const globalNonRegression = bounds.recallGain >= global.minimumRecallGainLowerBound && + bounds.mrrGain >= global.minimumMrrGainLowerBound && + foundation.precisionAtK - metrics.precisionAtK <= global.maximumPrecisionDecline && + foundation.abstentionAccuracy - metrics.abstentionAccuracy <= global.maximumAbstentionDecline + const safety = metrics.explicitForbiddenSelections <= safe.maximumExplicitForbiddenSelections && + metrics.authorizationOrLifecycleViolations <= safe.maximumAuthorizationOrLifecycleViolations + return { localBenefit, globalNonRegression, safety, passed: localBenefit && globalNonRegression && safety } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.test.ts b/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.test.ts new file mode 100644 index 000000000..93a4100dc --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.test.ts @@ -0,0 +1,21 @@ +import { mkdtemp, readdir, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { runMemoryFeedbackTiebreakerHoldout } from './memory-feedback-tiebreaker-holdout-runner.js' + +describe('memory feedback tiebreaker holdout runner', () => { + it('refuses scoring the retired version even in a fresh output directory', async () => { + const outputDirectory = await mkdtemp(join(tmpdir(), 'kun-tiebreaker-runner-')) + try { + await expect(runMemoryFeedbackTiebreakerHoldout({ + outputDirectory, + independentReviewConfirmed: true, + lockedAt: '2026-09-21T00:00:00.000Z' + })).rejects.toThrow(/v1 is retired/u) + expect(await readdir(outputDirectory)).toEqual([]) + } finally { + await rm(outputDirectory, { recursive: true, force: true }) + } + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.ts b/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.ts new file mode 100644 index 000000000..78916642e --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-holdout-runner.ts @@ -0,0 +1,11 @@ +import type { MemoryFeedbackTiebreakerEvidenceValue } from './memory-feedback-tiebreaker-contracts.js' + +// Repeated real-data tests consumed v1. Preserve its historical artifact; +// never score it again. See holdout-execution-audit.md for the limitations. +export async function runMemoryFeedbackTiebreakerHoldout(_input: { + outputDirectory: string + independentReviewConfirmed: boolean + lockedAt: string +}): Promise { + throw new Error('Memory feedback tiebreaker v1 is retired: see holdout-execution-audit.md') +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-holdout.test.ts b/kun/src/memory/memory-feedback-tiebreaker-holdout.test.ts new file mode 100644 index 000000000..e483ae662 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-holdout.test.ts @@ -0,0 +1,176 @@ +import { mkdtemp, readFile, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerCandidateLock, + MemoryFeedbackTiebreakerDecisionPlan, + MemoryFeedbackTiebreakerDevelopmentArtifact, + memoryFeedbackTiebreakerArtifactSha256 +} from './memory-feedback-tiebreaker-contracts.js' +import { + persistLockedMemoryFeedbackTiebreakerHoldout, + runLockedMemoryFeedbackTiebreakerHoldout +} from './memory-feedback-tiebreaker-holdout.js' + +const fixturePath = (name: string) => fileURLToPath(new URL(`./fixtures/${name}`, import.meta.url)) +const temporaryRoots: string[] = [] +afterEach(async () => { + await Promise.all(temporaryRoots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('memory feedback tiebreaker holdout guard', () => { + it('persists only one concurrent attempt and refuses subsequent runs', async () => { + const input = await persistenceInput() + const evidence = syntheticEvidence(input) + const evaluate = vi.fn(async () => evidence) + const results = await Promise.allSettled([ + persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate }), + persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate }) + ]) + expect(results.filter((result) => result.status === 'fulfilled')).toHaveLength(1) + expect(evaluate).toHaveBeenCalledOnce() + const path = join(input.outputDirectory, `${input.lock.decisionId}.json`) + const before = await readFile(path, 'utf8') + expect(JSON.parse(before)).toEqual(evidence) + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate })) + .rejects.toMatchObject({ code: 'EEXIST' }) + expect(await readFile(path, 'utf8')).toBe(before) + }) + + it('retains a failed attempt and cannot rerun after evaluation throws', async () => { + const input = await persistenceInput() + const evaluate = vi.fn(() => { throw new Error('interrupted') }) + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate })) + .rejects.toThrow('interrupted') + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate })) + .rejects.toMatchObject({ code: 'EEXIST' }) + expect(evaluate).toHaveBeenCalledOnce() + }) + + it('rejects a false go decision without publishing evidence', async () => { + const input = await persistenceInput() + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ + ...input, evaluate: () => ({ ...syntheticEvidence(input), decision: 'go' }) + })).rejects.toThrow(/decision disagrees/u) + expect(await readFile(join(input.outputDirectory, `${input.lock.decisionId}.json`), 'utf8')).toBe('') + }) + + it('rejects mismatched identities and rewritten development metrics', async () => { + for (const mutation of ['identity', 'development'] as const) { + const input = await persistenceInput() + const evidence = syntheticEvidence(input) + if (mutation === 'identity') evidence.selectedCandidateId = 'confirmation-gap_1' + else evidence.development = { + ...evidence.development, candidate: { ...evidence.development.candidate, recallAtK: 1 } + } + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ ...input, evaluate: () => evidence })) + .rejects.toThrow(mutation === 'identity' ? /does not match lock/u : /rewrites locked development/u) + } + }) + + it('does not reserve or score an unreviewed attempt', async () => { + const input = await persistenceInput() + const evaluate = vi.fn() + await expect(persistLockedMemoryFeedbackTiebreakerHoldout({ + ...input, independentReviewConfirmed: false, evaluate + })).rejects.toThrow(/requires independent lock review/u) + expect(evaluate).not.toHaveBeenCalled() + await expect(readFile(join(input.outputDirectory, `${input.lock.decisionId}.json`))) + .rejects.toMatchObject({ code: 'ENOENT' }) + }) + + it('rejects an invalid lock before the evaluation callback runs', async () => { + const input = await loadInput() + const evaluate = vi.fn() + const invalidLock = { + ...input.lock, + artifactHashes: { ...input.lock.artifactHashes, fixtureSha256: 'f'.repeat(64) } + } + + expect(() => runLockedMemoryFeedbackTiebreakerHoldout({ + ...input, + lock: invalidLock, + independentReviewConfirmed: true, + evidenceExists: false, + evaluate + })).toThrow(/does not match its dependencies/u) + expect(evaluate).not.toHaveBeenCalled() + }) + + it('requires independent review and refuses to overwrite evidence', async () => { + const input = await loadInput() + const evaluate = vi.fn() + const run = (independentReviewConfirmed: boolean, evidenceExists: boolean) => () => + runLockedMemoryFeedbackTiebreakerHoldout({ + ...input, + independentReviewConfirmed, + evidenceExists, + evaluate + }) + + expect(run(false, false)).toThrow(/requires independent lock review/u) + expect(run(true, true)).toThrow(/evidence already exists/u) + expect(evaluate).not.toHaveBeenCalled() + }) + + it('passes only the locked candidate id to an authorized evaluator', async () => { + const input = await loadInput() + const evaluate = vi.fn((candidateId: string) => ({ candidateId })) + + expect(runLockedMemoryFeedbackTiebreakerHoldout({ + ...input, + independentReviewConfirmed: true, + evidenceExists: false, + evaluate + })).toEqual({ candidateId: 'foundation-control' }) + expect(evaluate).toHaveBeenCalledOnce() + }) +}) + +async function persistenceInput() { + const outputDirectory = await mkdtemp(join(tmpdir(), 'kun-tiebreaker-holdout-test-')) + temporaryRoots.push(outputDirectory) + return { ...await loadInput(), outputDirectory, independentReviewConfirmed: true } +} + +// Synthetic writer payload only; no holdout cases are evaluated by these tests. +function syntheticEvidence(input: Awaited>) { + const candidate = input.development.configurations[0]! + const partition = { + foundation: input.development.foundation, + candidate: candidate.metrics, + bootstrapLowerBounds: candidate.bootstrapLowerBounds, + gates: { ...candidate.gates, privacy: true, determinism: true, resource: true } + } + return { + schemaVersion: 1, evaluationVersion: input.plan.evaluationVersion, + decisionId: input.lock.decisionId, status: 'final', + artifactHashes: { + ...input.plan.artifactHashes, + decisionPlanSha256: memoryFeedbackTiebreakerArtifactSha256(input.plan), + candidateLockSha256: memoryFeedbackTiebreakerArtifactSha256(input.lock) + }, + selectedCandidateId: input.lock.selectedCandidateId, holdoutRunCount: 1, + development: partition, holdout: partition, decision: 'no-go', + reasons: ['Synthetic persistence test; not scored holdout evidence.'] + } +} + +async function loadInput() { + const [calibrationText, planText, developmentText, lockText] = await Promise.all([ + readFile(fixturePath('memory-feedback-tiebreaker-calibration.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-plan.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-development.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-lock.v1.json'), 'utf8') + ]) + return { + calibration: MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(calibrationText)), + plan: MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(planText)), + development: MemoryFeedbackTiebreakerDevelopmentArtifact.parse(JSON.parse(developmentText)), + lock: MemoryFeedbackTiebreakerCandidateLock.parse(JSON.parse(lockText)), + lockedAt: '2026-09-16T00:00:00.000Z' + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-holdout.ts b/kun/src/memory/memory-feedback-tiebreaker-holdout.ts new file mode 100644 index 000000000..899e3e0a3 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-holdout.ts @@ -0,0 +1,99 @@ +import { mkdir, open } from 'node:fs/promises' +import { join } from 'node:path' +import { + MemoryFeedbackTiebreakerEvidence, + memoryFeedbackTiebreakerArtifactSha256, + type MemoryFeedbackTiebreakerCalibration, + type MemoryFeedbackTiebreakerDevelopmentArtifactValue, + type MemoryFeedbackTiebreakerLock, + type MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { evaluateMemoryFeedbackTiebreakerGates } from './memory-feedback-tiebreaker-gates.js' +import { assertMemoryFeedbackTiebreakerPrivateArtifact } from './memory-feedback-tiebreaker-privacy.js' +import { validateMemoryFeedbackTiebreakerCandidateLock } from './memory-feedback-tiebreaker-lock.js' + +export function runLockedMemoryFeedbackTiebreakerHoldout(input: { + lock: MemoryFeedbackTiebreakerLock + plan: MemoryFeedbackTiebreakerPlan + calibration: MemoryFeedbackTiebreakerCalibration + development: MemoryFeedbackTiebreakerDevelopmentArtifactValue + lockedAt: string + independentReviewConfirmed: boolean + evidenceExists: boolean + evaluate: (selectedCandidateId: string) => T +}): T { + const lock = validateMemoryFeedbackTiebreakerCandidateLock(input) + if (!input.independentReviewConfirmed) { + throw new Error('memory feedback tiebreaker holdout requires independent lock review') + } + if (input.evidenceExists) { + throw new Error('memory feedback tiebreaker holdout evidence already exists') + } + return input.evaluate(lock.selectedCandidateId) +} + +// Use one fixed output directory per decision registry. An empty/partial file is +// a consumed attempt after failure, not permission to rerun or delete evidence. +export async function persistLockedMemoryFeedbackTiebreakerHoldout(input: Omit< + Parameters[0], + 'evidenceExists' | 'evaluate' +> & { + outputDirectory: string + evaluate: (selectedCandidateId: string) => unknown | Promise +}) { + const selectedCandidateId = runLockedMemoryFeedbackTiebreakerHoldout({ + ...input, evidenceExists: false, evaluate: (id) => id + }) + await mkdir(input.outputDirectory, { recursive: true }) + const path = join(input.outputDirectory, `${input.lock.decisionId}.json`) + const handle = await open(path, 'wx', 0o600) + try { + await handle.sync() + const evidence = MemoryFeedbackTiebreakerEvidence.parse(await input.evaluate(selectedCandidateId)) + const expectedHashes = { + ...input.plan.artifactHashes, + decisionPlanSha256: memoryFeedbackTiebreakerArtifactSha256(input.plan), + candidateLockSha256: memoryFeedbackTiebreakerArtifactSha256(input.lock) + } + if (evidence.decisionId !== input.lock.decisionId || evidence.selectedCandidateId !== selectedCandidateId || + memoryFeedbackTiebreakerArtifactSha256(evidence.artifactHashes) !== + memoryFeedbackTiebreakerArtifactSha256(expectedHashes)) { + throw new Error('memory feedback tiebreaker evidence does not match lock') + } + const selected = input.development.configurations.find((candidate) => candidate.candidateId === selectedCandidateId)! + const expectedDevelopment = { + foundation: input.development.foundation, + candidate: selected.metrics, + bootstrapLowerBounds: selected.bootstrapLowerBounds + } + const { foundation, candidate, bootstrapLowerBounds } = evidence.development + if (memoryFeedbackTiebreakerArtifactSha256({ foundation, candidate, bootstrapLowerBounds }) !== + memoryFeedbackTiebreakerArtifactSha256(expectedDevelopment)) { + throw new Error('memory feedback tiebreaker evidence rewrites locked development') + } + for (const partition of [evidence.development, evidence.holdout]) { + const recomputed = evaluateMemoryFeedbackTiebreakerGates(partition.foundation, { + metrics: partition.candidate, bootstrapLowerBounds: partition.bootstrapLowerBounds + }, input.plan) + const { privacy, determinism, resource } = partition.gates + const expected = { ...recomputed, privacy, determinism, resource, + passed: recomputed.passed && privacy && determinism && resource } + if (memoryFeedbackTiebreakerArtifactSha256(expected) !== + memoryFeedbackTiebreakerArtifactSha256(partition.gates)) { + throw new Error('memory feedback tiebreaker evidence gates disagree with metrics') + } + } + const decision = evidence.development.gates.passed && evidence.holdout.gates.passed ? 'go' : 'no-go' + if (evidence.decision !== decision) throw new Error('memory feedback tiebreaker decision disagrees with gates') + assertMemoryFeedbackTiebreakerPrivateArtifact(evidence) + const serialized = `${JSON.stringify(evidence, null, 2)}\n` + if (Buffer.byteLength(serialized) > input.plan.gates.resource.maximumEvidenceBytes) { + throw new Error('memory feedback tiebreaker evidence exceeds byte limit') + } + await handle.writeFile(serialized, 'utf8') + await handle.sync() + return evidence + } finally { + await handle.close() + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-isolation.test.ts b/kun/src/memory/memory-feedback-tiebreaker-isolation.test.ts new file mode 100644 index 000000000..89afcb9fb --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-isolation.test.ts @@ -0,0 +1,61 @@ +import { readdir, readFile } from 'node:fs/promises' +import { dirname, extname, join, relative } from 'node:path' +import { fileURLToPath } from 'node:url' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { evaluateMemoryFeedbackTiebreakerCase } from './memory-feedback-tiebreaker-evaluator.js' + +const kunSourceRoot = dirname(dirname(fileURLToPath(import.meta.url))) +const rendererSourceRoot = join(kunSourceRoot, '..', '..', 'src') + +describe('memory feedback tiebreaker evaluator isolation', () => { + afterEach(() => vi.unstubAllGlobals()) + + it('is imported only by its offline modules and tests', async () => { + const sourceFiles = [ + ...await sourceFilesBelow(kunSourceRoot), + ...await sourceFilesBelow(rendererSourceRoot) + ] + const violations: string[] = [] + + for (const path of sourceFiles) { + const source = await readFile(path, 'utf8') + if (!source.includes('memory-feedback-tiebreaker-')) continue + if (path.split(/[\\/]/u).at(-1)?.startsWith('memory-feedback-tiebreaker-')) continue + violations.push(relative(join(kunSourceRoot, '..', '..'), path).replaceAll('\\', '/')) + } + + expect(violations).toEqual([]) + }) + + it('runs development evaluation without attempting network access', async () => { + let networkAttempts = 0 + vi.stubGlobal('fetch', async () => { + networkAttempts += 1 + throw new Error('network access is forbidden in the P3-B evaluator') + }) + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + + for (const item of fixture.cases.filter((candidate) => candidate.partition === 'development')) { + evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item, + signalRule: 'confirmation-correction', + maximumGap: 1 + }) + } + + expect(networkAttempts).toBe(0) + }) +}) + +async function sourceFilesBelow(root: string): Promise { + const entries = await readdir(root, { withFileTypes: true }) + const files: string[] = [] + for (const entry of entries) { + const path = join(root, entry.name) + if (entry.isDirectory()) files.push(...await sourceFilesBelow(path)) + else if (extname(entry.name) === '.ts' || extname(entry.name) === '.tsx') files.push(path) + } + return files +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-lock.test.ts b/kun/src/memory/memory-feedback-tiebreaker-lock.test.ts new file mode 100644 index 000000000..dbd88f597 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-lock.test.ts @@ -0,0 +1,89 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerCandidateLock, + MemoryFeedbackTiebreakerDecisionPlan, + MemoryFeedbackTiebreakerDevelopmentArtifact +} from './memory-feedback-tiebreaker-contracts.js' +import { + createMemoryFeedbackTiebreakerCandidateLock, + validateMemoryFeedbackTiebreakerCandidateLock +} from './memory-feedback-tiebreaker-lock.js' + +const fixturePath = (name: string) => fileURLToPath(new URL(`./fixtures/${name}`, import.meta.url)) + +describe('memory feedback tiebreaker candidate lock', () => { + it('rejects an internally inconsistent passing flag before creating a lock', async () => { + const input = await loadInput() + input.development.configurations[4]!.gates.passed = true + expect(() => createMemoryFeedbackTiebreakerCandidateLock(input)) + .toThrow(/gates disagree with metrics/u) + }) + + it('rejects omitted, duplicated, reordered or redefined configurations', async () => { + for (const mutation of ['omit', 'duplicate', 'reorder', 'gap', 'signal'] as const) { + const input = await loadInput() + const configs = input.development.configurations + if (mutation === 'omit') configs.pop() + if (mutation === 'duplicate') configs[1] = configs[0]! + if (mutation === 'reorder') configs.reverse() + if (mutation === 'gap') configs[1]!.maximumGap = 0.9 + if (mutation === 'signal') configs[1]!.signalRule = 'foundation-only' + expect(() => createMemoryFeedbackTiebreakerCandidateLock(input)) + .toThrow(/candidate grid mismatch/u) + } + }) + + it('uses the pre-registered fallback when no development candidate passes every gate', async () => { + const input = await loadInput() + const lock = createMemoryFeedbackTiebreakerCandidateLock(input) + + expect(lock.selectedCandidateId).toBe('foundation-control') + expect(lock.gates).toEqual(input.plan.gates) + expect(lock.holdoutRunLimit).toBe(1) + expect(validateMemoryFeedbackTiebreakerCandidateLock({ lock, ...input })).toEqual(lock) + }) + + it('rejects a lock after any development dependency is tampered with', async () => { + const input = await loadInput() + const lock = createMemoryFeedbackTiebreakerCandidateLock(input) + const tampered = { + ...input.development, + configurations: input.development.configurations.map((candidate, index) => index === 0 + ? { ...candidate, selectedIdsSha256: 'f'.repeat(64) } + : candidate) + } + + expect(() => validateMemoryFeedbackTiebreakerCandidateLock({ + lock, + ...input, + development: tampered + })).toThrow(/does not match its dependencies/u) + }) + + it('matches the frozen candidate lock', async () => { + const input = await loadInput() + const frozen = MemoryFeedbackTiebreakerCandidateLock.parse(JSON.parse(await readFile( + fixturePath('memory-feedback-tiebreaker-lock.v1.json'), + 'utf8' + ))) + + expect(frozen).toEqual(createMemoryFeedbackTiebreakerCandidateLock(input)) + }) +}) + +async function loadInput() { + const [calibrationText, planText, developmentText] = await Promise.all([ + readFile(fixturePath('memory-feedback-tiebreaker-calibration.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-plan.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-development.v1.json'), 'utf8') + ]) + return { + calibration: MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(calibrationText)), + plan: MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(planText)), + development: MemoryFeedbackTiebreakerDevelopmentArtifact.parse(JSON.parse(developmentText)), + lockedAt: '2026-09-16T00:00:00.000Z' + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-lock.ts b/kun/src/memory/memory-feedback-tiebreaker-lock.ts new file mode 100644 index 000000000..9fd2ef551 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-lock.ts @@ -0,0 +1,109 @@ +import { + MemoryFeedbackTiebreakerCandidateLock, + memoryFeedbackTiebreakerArtifactSha256, + type MemoryFeedbackTiebreakerCalibration, + type MemoryFeedbackTiebreakerDevelopmentArtifactValue, + type MemoryFeedbackTiebreakerLock, + type MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { MEMORY_FEEDBACK_TIEBREAKER_FOUNDATION_VERSION } from './memory-feedback-tiebreaker-calibration-report.js' +import { evaluateMemoryFeedbackTiebreakerGates } from './memory-feedback-tiebreaker-gates.js' + +export const MEMORY_FEEDBACK_TIEBREAKER_EVALUATOR_IDENTITY = + `memory-feedback-tiebreaker-v1|foundation=${MEMORY_FEEDBACK_TIEBREAKER_FOUNDATION_VERSION}|grouping=leader-relative` + +export function createMemoryFeedbackTiebreakerCandidateLock(input: { + plan: MemoryFeedbackTiebreakerPlan + calibration: MemoryFeedbackTiebreakerCalibration + development: MemoryFeedbackTiebreakerDevelopmentArtifactValue + lockedAt: string +}): MemoryFeedbackTiebreakerLock { + validateDependencies(input) + const selectedCandidateId = selectCandidate(input.development, input.plan) + return MemoryFeedbackTiebreakerCandidateLock.parse({ + schemaVersion: 1, + evaluationVersion: 'p3-feedback-tiebreaker-v1', + decisionId: input.plan.decisionId, + status: 'locked', + artifactHashes: { + ...input.plan.artifactHashes, + decisionPlanSha256: memoryFeedbackTiebreakerArtifactSha256(input.plan), + developmentReportSha256: memoryFeedbackTiebreakerArtifactSha256(input.development) + }, + selectedCandidateId, + gates: input.plan.gates, + evaluatorIdentity: MEMORY_FEEDBACK_TIEBREAKER_EVALUATOR_IDENTITY, + bootstrapSeed: input.plan.bootstrap.seed, + holdoutRunLimit: input.plan.bootstrap.holdoutRuns, + lockedAt: input.lockedAt + }) +} + +function validateDependencies(input: { + plan: MemoryFeedbackTiebreakerPlan + calibration: MemoryFeedbackTiebreakerCalibration + development: MemoryFeedbackTiebreakerDevelopmentArtifactValue +}): void { + if (input.plan.artifactHashes.calibrationSha256 !== + memoryFeedbackTiebreakerArtifactSha256(input.calibration)) { + throw new Error('memory feedback tiebreaker calibration hash mismatch') + } + const expectedDevelopmentHashes = { + ...input.plan.artifactHashes, + decisionPlanSha256: memoryFeedbackTiebreakerArtifactSha256(input.plan) + } + if (memoryFeedbackTiebreakerArtifactSha256(input.development.artifactHashes) !== + memoryFeedbackTiebreakerArtifactSha256(expectedDevelopmentHashes)) { + throw new Error('memory feedback tiebreaker development dependency hash mismatch') + } + const configurations = input.development.configurations + if (configurations.length !== input.plan.candidates.length) { + throw new Error('memory feedback tiebreaker candidate grid mismatch') + } + configurations.forEach((candidate, index) => { + const declared = input.plan.candidates[index]! + const gap = declared.boundaryId === null ? 0 : input.calibration.boundaryGrid + .find((boundary) => boundary.id === declared.boundaryId)?.maximumGap + if (candidate.candidateId !== declared.id || candidate.boundaryId !== declared.boundaryId || + candidate.signalRule !== declared.signalRule || candidate.maximumGap !== gap) { + throw new Error('memory feedback tiebreaker candidate grid mismatch') + } + const gates = evaluateMemoryFeedbackTiebreakerGates(input.development.foundation, candidate, input.plan) + if (memoryFeedbackTiebreakerArtifactSha256(gates) !== memoryFeedbackTiebreakerArtifactSha256(candidate.gates)) { + throw new Error('memory feedback tiebreaker gates disagree with metrics') + } + }) +} + +export function validateMemoryFeedbackTiebreakerCandidateLock(input: { + lock: unknown + plan: MemoryFeedbackTiebreakerPlan + calibration: MemoryFeedbackTiebreakerCalibration + development: MemoryFeedbackTiebreakerDevelopmentArtifactValue + lockedAt: string +}): MemoryFeedbackTiebreakerLock { + const lock = MemoryFeedbackTiebreakerCandidateLock.parse(input.lock) + const expected = createMemoryFeedbackTiebreakerCandidateLock(input) + if (memoryFeedbackTiebreakerArtifactSha256(lock) !== memoryFeedbackTiebreakerArtifactSha256(expected)) { + throw new Error('memory feedback tiebreaker candidate lock does not match its dependencies') + } + return lock +} + +function selectCandidate( + development: MemoryFeedbackTiebreakerDevelopmentArtifactValue, + plan: MemoryFeedbackTiebreakerPlan +): string { + const eligible = development.configurations.filter((candidate) => candidate.gates.passed) + eligible.sort((left, right) => + right.bootstrapLowerBounds.pairAccuracyGain - left.bootstrapLowerBounds.pairAccuracyGain || + (right.metrics.pairAccuracy - development.foundation.pairAccuracy) - + (left.metrics.pairAccuracy - development.foundation.pairAccuracy) || + left.maximumGap - right.maximumGap || + left.candidateId.localeCompare(right.candidateId)) + const selected = eligible[0]?.candidateId ?? plan.selectionRule.fallback + if (!plan.candidates.some((candidate) => candidate.id === selected)) { + throw new Error(`selected tiebreaker candidate is not pre-registered: ${selected}`) + } + return selected +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-metrics.test.ts b/kun/src/memory/memory-feedback-tiebreaker-metrics.test.ts new file mode 100644 index 000000000..9045dc41f --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-metrics.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it } from 'vitest' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { scoreMemoryFeedbackTiebreakerCases } from './memory-feedback-tiebreaker-metrics.js' + +describe('memory feedback tiebreaker metrics', () => { + it('reports local pair ordering separately from global relevance and abstention', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const pairCase = fixture.cases.find((item) => + item.partition === 'development' && item.stratum === 'confirmed-near-tie')! + const noResultCase = fixture.cases.find((item) => + item.partition === 'development' && item.stratum === 'no-result-control')! + const pair = pairCase.preferredPairs[0]! + const scored = scoreMemoryFeedbackTiebreakerCases({ + fixture, + results: [{ + caseId: pairCase.id, + rankings: [{ memoryId: pair.preferredId }, { memoryId: pair.otherId }], + selectedIds: [pair.preferredId] + }, { + caseId: noResultCase.id, + rankings: [], + selectedIds: [] + }] + }) + + expect(scored.metrics).toEqual({ + pairAccuracy: 1, + pairCount: 1, + recallAtK: 1, + precisionAtK: 1, + meanReciprocalRank: 1, + abstentionAccuracy: 1, + explicitForbiddenSelections: 0, + authorizationOrLifecycleViolations: 0, + rankedCaseCount: 1, + noResultCaseCount: 1, + caseCount: 2 + }) + }) + + it('counts a reversed pair and selected hard negative as separate failures', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === 'misleading-feedback-control')! + const pair = item.preferredPairs[0]! + const scored = scoreMemoryFeedbackTiebreakerCases({ + fixture, + results: [{ + caseId: item.id, + rankings: [{ memoryId: pair.otherId }, { memoryId: pair.preferredId }], + selectedIds: [pair.otherId] + }] + }) + + expect(scored.cases[0]).not.toHaveProperty('pairCorrect') + expect(scored.metrics.recallAtK).toBe(0) + expect(scored.metrics.explicitForbiddenSelections).toBe(1) + expect(scored.metrics.authorizationOrLifecycleViolations).toBe(0) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-metrics.ts b/kun/src/memory/memory-feedback-tiebreaker-metrics.ts new file mode 100644 index 000000000..b34b7c15b --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-metrics.ts @@ -0,0 +1,133 @@ +import type { MemoryRecord } from '../contracts/memory.js' +import { memoryInScope, memoryLifecycleState } from './memory-ranking.js' +import type { MemoryFeedbackTiebreakerFixture } from './memory-feedback-tiebreaker-contracts.js' + +type FixtureCase = MemoryFeedbackTiebreakerFixture['cases'][number] + +export type MemoryFeedbackTiebreakerScoredRanking = { + memoryId: string +} + +export type MemoryFeedbackTiebreakerScoredCase = { + caseId: string + rankings: MemoryFeedbackTiebreakerScoredRanking[] + selectedIds: string[] +} + +export type MemoryFeedbackTiebreakerCaseMetrics = { + caseId: string + pairCorrect?: number + recallAtK?: number + precisionAtK?: number + reciprocalRank?: number + abstentionCorrect?: number + explicitForbiddenSelections: number + authorizationOrLifecycleViolations: number +} + +export type MemoryFeedbackTiebreakerMetrics = { + pairAccuracy: number + pairCount: number + recallAtK: number + precisionAtK: number + meanReciprocalRank: number + abstentionAccuracy: number + explicitForbiddenSelections: number + authorizationOrLifecycleViolations: number + rankedCaseCount: number + noResultCaseCount: number + caseCount: number +} + +const LOCAL_BENEFIT_STRATA = new Set([ + 'confirmed-near-tie', + 'corrected-replacement-near-tie' +]) + +export function scoreMemoryFeedbackTiebreakerCases(input: { + fixture: MemoryFeedbackTiebreakerFixture + results: readonly MemoryFeedbackTiebreakerScoredCase[] +}): { cases: MemoryFeedbackTiebreakerCaseMetrics[]; metrics: MemoryFeedbackTiebreakerMetrics } { + const casesById = new Map(input.fixture.cases.map((item) => [item.id, item])) + const records = new Map(input.fixture.records.map((record) => [record.id, record])) + const nowMs = Date.parse(input.fixture.evaluationNow) + const cases = input.results.map((result) => { + const item = casesById.get(result.caseId) + if (!item) throw new Error(`unknown tiebreaker result case: ${result.caseId}`) + return scoreCase(item, result, records, nowMs) + }) + return { cases, metrics: summarize(cases) } +} + +function scoreCase( + item: FixtureCase, + result: MemoryFeedbackTiebreakerScoredCase, + records: ReadonlyMap, + nowMs: number +): MemoryFeedbackTiebreakerCaseMetrics { + const expected = new Set(item.expectedIds) + const relevantSelected = result.selectedIds.filter((id) => expected.has(id)).length + const firstRelevant = result.selectedIds.findIndex((id) => expected.has(id)) + const ranked = expected.size > 0 + const pairScores = LOCAL_BENEFIT_STRATA.has(item.stratum) + ? item.preferredPairs.map((pair) => preferredPairScore(result.rankings, pair)) + : [] + const violations = result.selectedIds.filter((id) => { + const record = records.get(id) + return !record || !memoryInScope(record, { workspace: item.workspace }) || + memoryLifecycleState(record, nowMs) !== 'active' + }).length + + return { + caseId: item.id, + ...(pairScores.length > 0 ? { pairCorrect: average(pairScores) } : {}), + ...(ranked ? { + recallAtK: relevantSelected / expected.size, + precisionAtK: result.selectedIds.length === 0 ? 0 : relevantSelected / result.selectedIds.length, + reciprocalRank: firstRelevant < 0 ? 0 : 1 / (firstRelevant + 1) + } : { abstentionCorrect: result.selectedIds.length === 0 ? 1 : 0 }), + explicitForbiddenSelections: result.selectedIds.filter((id) => item.forbiddenIds.includes(id)).length, + authorizationOrLifecycleViolations: violations + } +} + +function preferredPairScore( + rankings: readonly MemoryFeedbackTiebreakerScoredRanking[], + pair: FixtureCase['preferredPairs'][number] +): number { + const preferred = rankings.findIndex((ranking) => ranking.memoryId === pair.preferredId) + const other = rankings.findIndex((ranking) => ranking.memoryId === pair.otherId) + if (preferred < 0 || other < 0) return 0 + return preferred < other ? 1 : 0 +} + +function summarize(cases: readonly MemoryFeedbackTiebreakerCaseMetrics[]): MemoryFeedbackTiebreakerMetrics { + const pairs = cases.flatMap((item) => item.pairCorrect === undefined ? [] : [item.pairCorrect]) + const ranked = cases.filter((item) => item.recallAtK !== undefined) + const noResult = cases.filter((item) => item.abstentionCorrect !== undefined) + return { + pairAccuracy: round(average(pairs)), + pairCount: pairs.length, + recallAtK: round(average(ranked.map((item) => item.recallAtK!))), + precisionAtK: round(average(ranked.map((item) => item.precisionAtK!))), + meanReciprocalRank: round(average(ranked.map((item) => item.reciprocalRank!))), + abstentionAccuracy: round(average(noResult.map((item) => item.abstentionCorrect!))), + explicitForbiddenSelections: sum(cases.map((item) => item.explicitForbiddenSelections)), + authorizationOrLifecycleViolations: sum(cases.map((item) => item.authorizationOrLifecycleViolations)), + rankedCaseCount: ranked.length, + noResultCaseCount: noResult.length, + caseCount: cases.length + } +} + +function average(values: readonly number[]): number { + return values.length === 0 ? 0 : sum(values) / values.length +} + +function sum(values: readonly number[]): number { + return values.reduce((total, value) => total + value, 0) +} + +function round(value: number): number { + return Math.round(value * 1_000_000) / 1_000_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-plan.test.ts b/kun/src/memory/memory-feedback-tiebreaker-plan.test.ts new file mode 100644 index 000000000..6fc3f0264 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-plan.test.ts @@ -0,0 +1,85 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerDecisionPlan, + memoryFeedbackTiebreakerArtifactSha256 +} from './memory-feedback-tiebreaker-contracts.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { createMemoryFeedbackTiebreakerDecisionPlan } from './memory-feedback-tiebreaker-plan.js' + +const calibrationPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-calibration.v1.json', + import.meta.url +)) +const planPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-plan.v1.json', + import.meta.url +)) + +describe('memory feedback tiebreaker decision plan', () => { + it('pre-registers every candidate and gate before development candidate scoring', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const calibration = MemoryFeedbackTiebreakerCalibrationReport.parse( + JSON.parse(await readFile(calibrationPath, 'utf8')) + ) + const plan = createMemoryFeedbackTiebreakerDecisionPlan({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration + }) + + expect(plan.candidates).toHaveLength(7) + expect(plan.partitions.development).toHaveLength(18) + expect(plan.partitions.holdout).toHaveLength(18) + expect(plan.gates.localBenefit.minimumPairAccuracyGain).toBe(0.25) + expect(plan.gates.globalNonRegression.minimumRecallGainLowerBound).toBe(0) + expect(plan.bootstrap).toMatchObject({ seed: 20260916, resamples: 10_000, holdoutRuns: 1 }) + expect(plan.production).toEqual({ + rankingChanged: false, + dormantFeatureFlagAdded: false, + evaluatorOnly: true + }) + }) + + it('matches the frozen pre-registered plan artifact', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const calibration = MemoryFeedbackTiebreakerCalibrationReport.parse( + JSON.parse(await readFile(calibrationPath, 'utf8')) + ) + const generated = createMemoryFeedbackTiebreakerDecisionPlan({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration + }) + const frozen = MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(await readFile(planPath, 'utf8'))) + + expect(frozen).toEqual(generated) + }) + + it('changes identity when any gate or candidate definition changes', async () => { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const calibration = MemoryFeedbackTiebreakerCalibrationReport.parse( + JSON.parse(await readFile(calibrationPath, 'utf8')) + ) + const plan = createMemoryFeedbackTiebreakerDecisionPlan({ + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration + }) + const changed = { + ...plan, + gates: { + ...plan.gates, + localBenefit: { ...plan.gates.localBenefit, minimumPairAccuracyGain: 0.5 } + } + } + + expect(memoryFeedbackTiebreakerArtifactSha256(changed)) + .not.toBe(memoryFeedbackTiebreakerArtifactSha256(plan)) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-plan.ts b/kun/src/memory/memory-feedback-tiebreaker-plan.ts new file mode 100644 index 000000000..c54107aa5 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-plan.ts @@ -0,0 +1,97 @@ +import { + MemoryFeedbackTiebreakerDecisionPlan, + memoryFeedbackTiebreakerArtifactSha256, + type MemoryFeedbackTiebreakerCalibration, + type MemoryFeedbackTiebreakerFixture, + type MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { buildMemoryFeedbackTiebreakerCandidates } from './memory-feedback-tiebreaker-calibration.js' + +export const MEMORY_FEEDBACK_TIEBREAKER_DECISION_ID = 'kun-memory-feedback-tiebreaker-v1' + +export function createMemoryFeedbackTiebreakerDecisionPlan(input: { + fixture: MemoryFeedbackTiebreakerFixture + fixtureSha256: string + manifestSha256: string + calibration: MemoryFeedbackTiebreakerCalibration +}): MemoryFeedbackTiebreakerPlan { + return MemoryFeedbackTiebreakerDecisionPlan.parse({ + schemaVersion: 1, + evaluationVersion: 'p3-feedback-tiebreaker-v1', + decisionId: MEMORY_FEEDBACK_TIEBREAKER_DECISION_ID, + status: 'pre-registered', + artifactHashes: { + fixtureSha256: input.fixtureSha256, + manifestSha256: input.manifestSha256, + calibrationSha256: memoryFeedbackTiebreakerArtifactSha256(input.calibration) + }, + partitions: { + development: partitionIds(input.fixture, 'development'), + holdout: partitionIds(input.fixture, 'holdout') + }, + candidates: buildMemoryFeedbackTiebreakerCandidates(input.calibration.boundaryGrid), + selectionRule: { + orderBy: [ + 'all-required-gates', + 'pair-accuracy-lower-bound', + 'pair-accuracy-point-estimate', + 'smallest-boundary', + 'candidate-id' + ], + fallback: 'foundation-control' + }, + gates: { + localBenefit: { + minimumPairAccuracyGain: 0.25, + minimumPairAccuracyGainLowerBound: 0.001 + }, + globalNonRegression: { + minimumRecallGainLowerBound: 0, + minimumMrrGainLowerBound: 0, + maximumPrecisionDecline: 0, + maximumAbstentionDecline: 0 + }, + safety: { + maximumExplicitForbiddenSelections: 0, + maximumAuthorizationOrLifecycleViolations: 0, + maximumProductionRankingChanges: 0 + }, + privacy: { + queryTextInTrace: false, + memoryContentInTrace: false, + sourceExcerptInTrace: false, + machinePathInTrace: false, + credentialInTrace: false + }, + determinism: { repeatedRuns: 3, numericTolerance: 0.000001 }, + resource: { + maximumEvaluationMilliseconds: 5_000, + maximumTraceRankings: 64, + maximumEvidenceBytes: 1_000_000 + } + }, + bootstrap: { + method: 'paired-percentile-lower-bound', + seed: 20260916, + resamples: 10_000, + confidenceLevel: 0.95, + unit: 'case', + holdoutRuns: 1 + }, + production: { + rankingChanged: false, + dormantFeatureFlagAdded: false, + evaluatorOnly: true + } + }) +} + +function partitionIds( + fixture: MemoryFeedbackTiebreakerFixture, + partition: 'development' | 'holdout' +): string[] { + return fixture.cases + .filter((item) => item.partition === partition) + .map((item) => item.id) + .sort() +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-privacy.test.ts b/kun/src/memory/memory-feedback-tiebreaker-privacy.test.ts new file mode 100644 index 000000000..ff563ac70 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-privacy.test.ts @@ -0,0 +1,42 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { MemoryFeedbackTiebreakerDevelopmentArtifact } from './memory-feedback-tiebreaker-contracts.js' +import { assertMemoryFeedbackTiebreakerPrivateArtifact } from './memory-feedback-tiebreaker-privacy.js' + +const developmentPath = fileURLToPath(new URL( + './fixtures/memory-feedback-tiebreaker-development.v1.json', + import.meta.url +)) + +describe('memory feedback tiebreaker artifact privacy', () => { + it('accepts the bounded synthetic development artifact', async () => { + const artifact = MemoryFeedbackTiebreakerDevelopmentArtifact.parse( + JSON.parse(await readFile(developmentPath, 'utf8')) + ) + expect(() => assertMemoryFeedbackTiebreakerPrivateArtifact(artifact)).not.toThrow() + }) + + it.each([ + ['query text', { query: 'synthetic query' }], + ['Memory content', { memoryContent: 'synthetic content' }], + ['source excerpt', { sourceExcerpt: 'synthetic excerpt' }], + ['credential', { diagnostic: 'api_key=fixture-secret' }], + ['Windows path', { diagnostic: 'at C:\\Users\\Fixture\\memory.json' }], + ['UNC path', { diagnostic: '\\\\server\\share\\memory.json' }], + ['POSIX path', { diagnostic: 'at /home/fixture/memory.json' }], + ['file URL', { diagnostic: 'file:///tmp/memory.json' }], + ['email identity', { diagnostic: 'fixture@example.com' }], + ['oversized diagnostic', { diagnostic: 'x'.repeat(513) }] + ])('rejects %s without echoing its value', (_label, artifact) => { + let message = '' + try { + assertMemoryFeedbackTiebreakerPrivateArtifact(artifact) + } catch (error) { + message = error instanceof Error ? error.message : String(error) + } + + expect(message).toMatch(/memory feedback tiebreaker artifact/u) + expect(message).not.toContain(Object.values(artifact)[0]!) + }) +}) diff --git a/kun/src/memory/memory-feedback-tiebreaker-privacy.ts b/kun/src/memory/memory-feedback-tiebreaker-privacy.ts new file mode 100644 index 000000000..9fafea983 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-privacy.ts @@ -0,0 +1,37 @@ +const MAX_SAFE_TRACE_STRING_CHARS = 512 +const FORBIDDEN_KEYS = new Set(['query', 'content', 'memoryContent', 'source', 'sources', 'sourceExcerpt']) +const FORBIDDEN_VALUES = [ + /(?:^|\s)[a-z]:[\\/]/iu, + /\\\\[^\\\s]+\\[^\s]+/u, + /file:\/\//iu, + /(?:^|\s)\/(?:users|home|var|tmp|etc)\//iu, + /\b(?:api[_-]?key|token|secret|password)\s*[:=]\s*\S+/iu, + /\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/iu +] + +export function assertMemoryFeedbackTiebreakerPrivateArtifact(value: unknown): void { + visit(value) +} + +function visit(value: unknown): void { + if (typeof value === 'string') { + if (value.length > MAX_SAFE_TRACE_STRING_CHARS) { + throw new Error('memory feedback tiebreaker artifact contains an oversized string') + } + if (FORBIDDEN_VALUES.some((pattern) => pattern.test(value))) { + throw new Error('memory feedback tiebreaker artifact contains private text') + } + return + } + if (Array.isArray(value)) { + for (const item of value) visit(item) + return + } + if (!value || typeof value !== 'object') return + for (const [key, item] of Object.entries(value as Record)) { + if (FORBIDDEN_KEYS.has(key)) { + throw new Error('memory feedback tiebreaker artifact contains a forbidden text field') + } + visit(item) + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-resource.test.ts b/kun/src/memory/memory-feedback-tiebreaker-resource.test.ts new file mode 100644 index 000000000..f2e2882d4 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-resource.test.ts @@ -0,0 +1,85 @@ +import { readFile } from 'node:fs/promises' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + MemoryFeedbackTiebreakerCalibrationReport, + MemoryFeedbackTiebreakerDecisionPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { + evaluateAndMeasureMemoryFeedbackTiebreakerResources, + measureMemoryFeedbackTiebreakerResources +} from './memory-feedback-tiebreaker-resource.js' + +const fixturePath = (name: string) => fileURLToPath(new URL(`./fixtures/${name}`, import.meta.url)) + +describe('memory feedback tiebreaker resource gates', () => { + it('passes the frozen development evaluation under its resource ceilings', async () => { + const input = await loadInput() + const result = evaluateAndMeasureMemoryFeedbackTiebreakerResources(input) + + expect(result.measurement.passed).toBe(true) + expect(result.measurement.maximumTraceRankings).toBeLessThanOrEqual( + input.plan.gates.resource.maximumTraceRankings + ) + expect(result.measurement.evidenceBytes).toBeLessThanOrEqual( + input.plan.gates.resource.maximumEvidenceBytes + ) + }) + + it('fails an over-limit synthetic trace without truncating evidence', async () => { + const input = await loadInput() + const result = evaluateAndMeasureMemoryFeedbackTiebreakerResources(input) + const firstCase = result.report.candidates[0]!.cases[0]! + const overLimitReport = { + ...result.report, + candidates: result.report.candidates.map((candidate, index) => index === 0 + ? { + ...candidate, + cases: [{ + ...firstCase, + rankings: Array.from({ length: 65 }, () => firstCase.rankings[0]!) + }] + } + : candidate) + } + const constrainedPlan = { + ...input.plan, + gates: { + ...input.plan.gates, + resource: { + ...input.plan.gates.resource, + maximumTraceRankings: 1, + maximumEvidenceBytes: 1 + } + } + } + + const measurement = measureMemoryFeedbackTiebreakerResources({ + report: overLimitReport, + evidence: result.evidence, + elapsedMilliseconds: 0, + plan: constrainedPlan + }) + + expect(measurement.passed).toBe(false) + expect(measurement.failures).toEqual(['trace-rankings', 'evidence-bytes']) + expect(measurement.evidenceBytes).toBeGreaterThan(1) + expect(result.evidence.configurations).toHaveLength(input.plan.candidates.length) + }) +}) + +async function loadInput() { + const dataset = await loadMemoryFeedbackTiebreakerDataset() + const [calibrationText, planText] = await Promise.all([ + readFile(fixturePath('memory-feedback-tiebreaker-calibration.v1.json'), 'utf8'), + readFile(fixturePath('memory-feedback-tiebreaker-plan.v1.json'), 'utf8') + ]) + return { + fixture: dataset.fixture, + fixtureSha256: dataset.sourceHashes.fixture, + manifestSha256: dataset.sourceHashes.manifest, + calibration: MemoryFeedbackTiebreakerCalibrationReport.parse(JSON.parse(calibrationText)), + plan: MemoryFeedbackTiebreakerDecisionPlan.parse(JSON.parse(planText)) + } +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-resource.ts b/kun/src/memory/memory-feedback-tiebreaker-resource.ts new file mode 100644 index 000000000..03337cb4f --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-resource.ts @@ -0,0 +1,83 @@ +import type { + MemoryFeedbackTiebreakerDevelopmentArtifactValue, + MemoryFeedbackTiebreakerPlan +} from './memory-feedback-tiebreaker-contracts.js' +import { + compactMemoryFeedbackTiebreakerDevelopmentReport, + evaluateMemoryFeedbackTiebreakerDevelopment, + type MemoryFeedbackTiebreakerDevelopmentReport +} from './memory-feedback-tiebreaker-development.js' + +export type MemoryFeedbackTiebreakerResourceFailure = + | 'evaluation-time' + | 'trace-rankings' + | 'evidence-bytes' + +export type MemoryFeedbackTiebreakerResourceMeasurement = { + elapsedMilliseconds: number + maximumTraceRankings: number + evidenceBytes: number + failures: MemoryFeedbackTiebreakerResourceFailure[] + passed: boolean +} + +export function evaluateAndMeasureMemoryFeedbackTiebreakerResources(input: { + fixture: Parameters[0]['fixture'] + fixtureSha256: string + manifestSha256: string + calibration: Parameters[0]['calibration'] + plan: MemoryFeedbackTiebreakerPlan +}): { + report: MemoryFeedbackTiebreakerDevelopmentReport + evidence: MemoryFeedbackTiebreakerDevelopmentArtifactValue + measurement: MemoryFeedbackTiebreakerResourceMeasurement +} { + const startedAt = performance.now() + const report = evaluateMemoryFeedbackTiebreakerDevelopment(input) + const elapsedMilliseconds = performance.now() - startedAt + const evidence = compactMemoryFeedbackTiebreakerDevelopmentReport(report) + const measurement = measureMemoryFeedbackTiebreakerResources({ + report, + evidence, + elapsedMilliseconds, + plan: input.plan + }) + return { report, evidence, measurement } +} + +export function measureMemoryFeedbackTiebreakerResources(input: { + report: MemoryFeedbackTiebreakerDevelopmentReport + evidence: MemoryFeedbackTiebreakerDevelopmentArtifactValue + elapsedMilliseconds: number + plan: MemoryFeedbackTiebreakerPlan +}): MemoryFeedbackTiebreakerResourceMeasurement { + const traceCounts = [ + ...input.report.foundation.cases.map((item) => item.rankings.length), + ...input.report.candidates + .flatMap((candidate) => candidate.cases) + .map((item) => item.rankings.length) + ] + const maximumTraceRankings = Math.max(...traceCounts) + const evidenceBytes = Buffer.byteLength(JSON.stringify(input.evidence), 'utf8') + const failures: MemoryFeedbackTiebreakerResourceFailure[] = [] + if (input.elapsedMilliseconds > input.plan.gates.resource.maximumEvaluationMilliseconds) { + failures.push('evaluation-time') + } + if (maximumTraceRankings > input.plan.gates.resource.maximumTraceRankings) { + failures.push('trace-rankings') + } + if (evidenceBytes > input.plan.gates.resource.maximumEvidenceBytes) { + failures.push('evidence-bytes') + } + return { + elapsedMilliseconds: roundMilliseconds(input.elapsedMilliseconds), + maximumTraceRankings, + evidenceBytes, + failures, + passed: failures.length === 0 + } +} + +function roundMilliseconds(value: number): number { + return Math.round(value * 1_000) / 1_000 +} diff --git a/kun/src/memory/memory-feedback-tiebreaker-safety.test.ts b/kun/src/memory/memory-feedback-tiebreaker-safety.test.ts new file mode 100644 index 000000000..cb11556f2 --- /dev/null +++ b/kun/src/memory/memory-feedback-tiebreaker-safety.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it } from 'vitest' +import { MemoryRecord } from '../contracts/memory.js' +import { loadMemoryFeedbackTiebreakerDataset } from './memory-feedback-tiebreaker-fixture-loader.js' +import { evaluateMemoryFeedbackTiebreakerCase } from './memory-feedback-tiebreaker-evaluator.js' + +describe('memory feedback tiebreaker safety boundaries', () => { + it('never admits out-of-scope, inactive, or superseded hard negatives', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + for (const stratum of ['scope-safety', 'lifecycle-safety'] as const) { + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === stratum)! + const result = evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item, + signalRule: 'confirmation-correction', + maximumGap: 1 + }) + + expect(result.rankings.every((ranking) => !item.forbiddenIds.includes(ranking.memoryId))).toBe(true) + expect(result.forbiddenSelectedIds).toEqual([]) + expect(result.rankings.every((ranking) => + fixture.records.find((record) => record.id === ranking.memoryId)?.authority === 'reference')).toBe(true) + } + + const corrected = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === 'corrected-replacement-near-tie')! + const correctedResult = evaluateMemoryFeedbackTiebreakerCase({ + fixture, + item: corrected, + signalRule: 'confirmation-correction', + maximumGap: 1 + }) + expect(correctedResult.rankings.map((ranking) => ranking.memoryId)).not.toContain('mem_corr_old') + }) + + it.each([ + ['disabled', { disabledAt: '2026-09-01T00:00:00.000Z' }], + ['deleted', { deletedAt: '2026-09-01T00:00:00.000Z' }], + ['expired', { expiresAt: '2026-09-01T00:00:00.000Z' }], + ['future-valid', { validFrom: '2026-10-01T00:00:00.000Z' }], + ['superseded', { supersededAt: '2026-09-01T00:00:00.000Z' }] + ])('excludes a %s record before feedback ordering', async (_label, lifecycle) => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const item = fixture.cases.find((candidate) => + candidate.partition === 'development' && candidate.stratum === 'lifecycle-safety')! + const records = fixture.records.map((record) => record.id === 'mem_disabled_timeline' + ? withLifecycle(record, lifecycle) + : record) + const result = evaluateMemoryFeedbackTiebreakerCase({ + fixture: { ...fixture, records }, + item, + signalRule: 'confirmation-correction', + maximumGap: 1 + }) + + expect(result.rankings.map((ranking) => ranking.memoryId)).not.toContain('mem_disabled_timeline') + expect(result.selectedIds).toEqual(['mem_active_timeline']) + }) + + it('preserves positive-relevance abstention despite explicit feedback', async () => { + const { fixture } = await loadMemoryFeedbackTiebreakerDataset() + const cases = fixture.cases.filter((candidate) => + candidate.partition === 'development' && candidate.stratum === 'no-result-control') + + for (const item of cases) { + for (const signalRule of ['confirmation', 'confirmation-correction'] as const) { + const result = evaluateMemoryFeedbackTiebreakerCase({ fixture, item, signalRule, maximumGap: 1 }) + expect(result.rankings).toEqual([]) + expect(result.selectedIds).toEqual([]) + } + } + }) +}) + +function withLifecycle( + record: MemoryRecord, + lifecycle: Record +): MemoryRecord { + const copy = { ...record } + delete copy.disabledAt + delete copy.deletedAt + delete copy.expiresAt + delete copy.validFrom + delete copy.supersededAt + return MemoryRecord.parse({ ...copy, ...lifecycle }) +} diff --git a/openspec/changes/add-kun-memory-feedback-evolution/.openspec.yaml b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/.openspec.yaml similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/.openspec.yaml rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/.openspec.yaml diff --git a/openspec/changes/add-kun-memory-feedback-evolution/design.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/design.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/design.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/design.md diff --git a/openspec/changes/add-kun-memory-feedback-evolution/proposal.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/proposal.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/proposal.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/proposal.md diff --git a/openspec/changes/add-kun-memory-feedback-evolution/specs/memory-feedback-evolution/spec.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-feedback-evolution/spec.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/specs/memory-feedback-evolution/spec.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-feedback-evolution/spec.md diff --git a/openspec/changes/add-kun-memory-feedback-evolution/specs/memory-record-foundation/spec.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-record-foundation/spec.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/specs/memory-record-foundation/spec.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-record-foundation/spec.md diff --git a/openspec/changes/add-kun-memory-feedback-evolution/specs/memory-retrieval-foundation/spec.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-retrieval-foundation/spec.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/specs/memory-retrieval-foundation/spec.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/specs/memory-retrieval-foundation/spec.md diff --git a/openspec/changes/add-kun-memory-feedback-evolution/tasks.md b/openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/tasks.md similarity index 100% rename from openspec/changes/add-kun-memory-feedback-evolution/tasks.md rename to openspec/changes/archive/2026-09-17-add-kun-memory-feedback-evolution/tasks.md diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/.openspec.yaml b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/.openspec.yaml new file mode 100644 index 000000000..f08077847 --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/.openspec.yaml @@ -0,0 +1,2 @@ +schema: spec-driven +created: 2026-09-16 diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/delivery-checklist.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/delivery-checklist.md new file mode 100644 index 000000000..b7fb13e17 --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/delivery-checklist.md @@ -0,0 +1,126 @@ +# P3-B delivery checklist + +## Dependency checkpoint: 2026-09-17 + +- P3-A PR #1324: CI passed, still open and review required at head `dee7b413`. +- Importer PR #1326: merged as `d784a47d`; unrelated to this capability. +- Fetched upstream develop: `1d456093`. +- Preparation branch: `codex/prepare-memory-feedback-tiebreaker` on SunwardL/Kun. +- Shared P3-A ancestor: `f9b49d0e`; nine newer P3-A commits are not yet included. + +These are a dated checkpoint, not a claim that the branch is ready to merge. + +## Current synchronization checkpoint: 2026-09-20 + +- `upstream/develop` is synchronized at `4778f9eb`. +- P3-A implementation #1324 is merged at `8974700b`. +- P3-A closeout #1331 is closed without merge; its substantive documentation + changes are included in this branch as `c56c8e7c` and `374eedf7`. +- The final delivery will use one P3-B PR; no standalone P3-A closeout PR will + be reopened. +- Post-sync verification: the focused P3-B suite passed (18 files / 70 tests), + `npm run build:kun` passed, and strict OpenSpec/diff/file-line checks passed. + Full `npm run typecheck` remains blocked by the pre-existing missing + `phonemizer` declaration in `src/main/services/local-kokoro-worker-entry.ts`; + this is not introduced by P3-B. + +## Work that can proceed before the dependency merges + +- [x] Reproduce development and candidate lock tests without scoring holdout. +- [x] Record development rejection and distinguish explicit forbidden selections + from authorization/lifecycle violations in `development-review.md`. +- [x] Keep data, evaluator, tests and documentation as meaningful separate commits. +- [ ] Obtain independent review of labels, lock and the control-only holdout + question described in `development-review.md`. +- [x] Review correction evidence against the P3-A event/aggregate contract and + test real correction, replay, compaction and restart. Replacement evidence is + deliberately distinct from the old record's aggregate correction count. + Repeat integration checks after the final P3-A baseline synchronization. + +## After the P3-A implementation merge + +1. Verify #1324 is MERGED and record its actual merge SHA, not just green checks. +2. Keep the P3-A canonical-spec synchronization and completed-change archive in + the same final P3-B PR. Preserve frozen P3-A evidence; do not reopen a + standalone closeout PR. +3. Fetch upstream develop and origin, confirm the worktree is clean, then + rebase this preparation branch onto the latest baseline. Preserve meaningful + P3-B commits; do not replay obsolete P3-A implementation commits as new + P3-B changes. Use force-with-lease only if a rebase requires it. +4. Review the final three-dot diff against upstream develop. It should contain + only this evaluation capability, anonymous artifacts and tests; no importer, + production ranking, runtime composition or UI changes. +5. Verify P3-A review fixes survived: runtime feedback getter wiring, isolated + diagnostics, disabled correction semantics, confirmation gating, malformed + ledger recovery documentation, IPC allowlist and runtime-config line budget. +6. Reproduce frozen development evidence and hashes on the merged baseline. + If semantics changed, report the mismatch; do not overwrite frozen inputs or + silently regenerate a more favorable decision under the same version. +7. Complete the independently reviewed decision workflow. Do not treat a + foundation fallback lock as go or report an unexecuted holdout as passed. +8. Run focused P3-B and existing Memory tests, build:kun, typecheck, build, lint, + file-lines, strict OpenSpec validation and diff check. Record failures honestly. +9. Finish tasks 6.2/6.3 and the stage report, including immutable evidence hashes, + holdout run count and remaining P4-A work. Update the local roadmap. +10. Push the verified branch and create one PR containing the P3-B evaluation + and the P3-A spec/archive closeout to `KunAgent/Kun:develop`. + Immediately verify the base, head and URL. CI success on #1324 does not + certify this new PR; it needs its own checks. + +## PR body outline (complete evidence before submitting) + +### Summary + +Evaluate bounded explicit-feedback tie-breakers offline. Production Memory +ranking remains unchanged. P3-A ledger/confirm/correct infrastructure is retained. + +### Changes + +Anonymous stratified data, baseline-derived near-tie grid, deterministic +comparison, versioned gates and lock, privacy/resource checks, development +report and independently reviewed decision outcome. + +### Tests + +Preparation checkpoint: 17 P3-B test files / 62 tests passed on 2026-09-17; +strict OpenSpec validation and changed-test ESLint passed. Replace this checkpoint +with final post-rebase verification results before submitting. + +### Decision and limitations + +Development currently has no eligible feedback candidate. Include final reviewed +outcome, artifact hashes, actual holdout run count and sample-size limitations. +Do not claim production relevance gains or enable ranking from this PR. + +## Final closeout checkpoint: 2026-09-23 + +This checkpoint supersedes the earlier preparation steps above. The P3-B +development grid and candidate lock are frozen and reproducible on the latest +`upstream/develop@65f6a55c4`. No feedback candidate passes all development gates; +the v1 decision is closed at development no-go. The historical holdout output is +not decision-grade because its runner allowed repeated temporary-directory runs +and did not measure every gate. The runner is retired, the output and all frozen +inputs are unchanged, and no holdout metrics are used in the decision. + +The final PR includes the offline P3-B evaluation and P3-A canonical-spec/archive +closeout together. It does not include a UI toggle, feedback-based production +ranking, or a production integration decision. `feedback.enabled` remains off by +default, and any future production-ranking proposal requires a separately +versioned decision that passes its gates. + +Post-rebase checks recorded for this closeout: P3-B focused suite passed +(19 files / 71 tests); all Kun Memory tests passed (57 files / 296 tests); +`npm run build:kun`, Memory-directory ESLint, strict OpenSpec validation, and +`git diff --check` passed. Root `npm run typecheck` reports two `AgentModelSettings` +type errors in an unchanged upstream file and cannot resolve Excalidraw because +that declared dependency is missing from this local `node_modules`. Root +`npm run build` stops on the same missing local package. Root `npm run lint` +stops before ESLint because unchanged upstream files exceed the file-line gate: +`register-app-file-ipc-handlers.ts` (707 lines) and `remote-bridge.js` (750). +The evaluator/Memory ESLint run passes. These failures are reported separately; +they are not described as passing gates. The PR must state plainly that holdout +is not decision-grade and that this is a development-only no-go. + +PR #1339 is open at , with base +`develop` and head `SunwardL:codex/prepare-memory-feedback-tiebreaker`. The initial +GitHub Quality gates check is pending; creation does not imply review or merge. diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/design.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/design.md new file mode 100644 index 000000000..6e71132b0 --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/design.md @@ -0,0 +1,145 @@ +## Context + +See `proposal.md` for motivation. P3-A provides an opt-in append-only feedback ledger, rebuildable aggregates, explicit confirmation/correction, versioned supersession, and an evaluator-only v1 additive candidate. That candidate added as much as `0.25` to the foundation score and correctly produced a holdout no-go because it did not improve the global relevance lower bounds. Production retrieval still uses the post-#1308 lexical/FTS5 foundation and does not read feedback aggregates. + +P3-B must test a narrower hypothesis without rewriting the frozen v1 artifacts: explicit user evidence may be useful only when the lexical foundation is already uncertain between authorized active candidates. The other agent's importer work is isolated in a separate worktree and does not share this change's files. + +## Goals / Non-Goals + +**Goals:** + +- Define and evaluate a bounded, deterministic near-tie reranking family. +- Derive candidate boundaries from development foundation-score distributions rather than adopting an unsupported percentage. +- Measure local pair-ordering benefit and global non-regression as distinct claims. +- Preserve holdout blindness, privacy, authorization, abstention, lifecycle, determinism, and resource limits. +- Produce a reproducible go/no-go artifact that can stand even when the answer is no-go. + +**Non-Goals:** + +- Changing production retrieval, prompt injection, settings, Manager/API/UI behavior, or packaged dependencies. +- Reopening embedding retrieval or adding a vector model. +- Treating retrieval frequency, freshness, confidence, or importance as user confirmation. +- Replacing the correction/supersession lifecycle path with a second relation mechanism. +- Designing P4 relation suggestions or Working Context in this change. +- Re-evaluating the deterministic `decideMemoryCandidate()` dispatch logic. + +## Decisions + +### 1. Add a new evaluation capability instead of modifying P3-A + +Create `memory-feedback-ranking-evaluation` as a separate capability. P3-A's contract already requires a separately versioned passed decision before production ranking can use feedback, so P3-B supplies that decision process without changing feedback collection or canonical Memory behavior. + +The P3-B evaluator remains a separate capability from `memory-feedback-evolution`. Because the standalone P3-A closeout PR was closed without merging, the final P3-B delivery may include the already-reviewed canonical spec synchronization and archive move as documentation-only closeout commits. This does not add production behavior or couple the evaluator to a ranking rollout. + +### 2. Use a lexical admission boundary followed by a near-tie window + +Run the existing foundation authorization, lifecycle, positive-relevance, result-count, and prompt-budget logic first. The evaluator then groups only admitted candidates whose foundation-score gaps satisfy one declared near-tie rule. Feedback can reorder within that group but cannot add a record, fill an abstention, cross a budget, or move a wide-gap candidate. + +An unconstrained additive score was rejected because v1 could add `0.25` regardless of foundation uncertainty. A percentage such as 5% or 10% is not adopted as a requirement; it may appear only as a labeled sensitivity-analysis point if derived and frozen before holdout. + +### 3. Derive a small boundary grid from baseline-only development evidence + +Before running feedback candidates, compute adjacent foundation-score gaps and labeled foundation errors on the development split. A versioned calibration report applies a fixed derivation procedure to produce a small finite grid, for example selected empirical gap quantiles plus an exact-tie control and a no-rerank control. The report records all source gaps, the procedure, resulting values, inclusivity, and maximum window size. + +The grid is allowed to use development labels because development is the declared tuning split. Directly choosing a boundary after inspecting candidate success, searching arbitrary values, or using holdout gaps is rejected as post-hoc tuning. + +### 4. Compare evidence rules without numeric popularity weights + +The primary candidate family uses deterministic evidence classes inside a near-tie group: + +1. confirmation-aware: explicit confirmation outranks neutral evidence; +2. confirmation-and-correction-aware: explicit confirmation and the active replacement identified by explicit correction are compared through a pre-declared evidence precedence; +3. foundation-only controls at every boundary. + +Within the same evidence class, preserve foundation score and then stable Memory id order. Retrieval frequency and last-retrieved time are emitted only as shadow features and never participate in the comparator or grid-selection rule. + +Numeric additive weights were rejected for this decision because they obscure the actual swap boundary and invite the same broad influence tested by v1. The development evidence may eliminate an evidence rule, but it cannot invent a new rule outside the pre-registration. + +### 5. Build a new anonymous fixture version with explicit strata + +Do not extend or overwrite the four-case P3-A v1 fixture. Add a new schema with synthetic Memory records, valid feedback events, queries, preferred/unpreferred pairs, expected sets, explicit hard negatives, and rationale fields. The manifest fixes exact counts before candidate scoring. + +Required strata are: + +- a confirmed relevant record should beat a close unconfirmed distractor; +- an active corrected replacement should beat a close stale or neutral distractor; +- high retrieval frequency without explicit evidence must not improve rank; +- no-feedback cases must preserve foundation order; +- wide-gap cases must preserve foundation order even when the lower record is confirmed; +- no-result queries must remain empty; +- out-of-scope and inactive records remain invisible despite stronger feedback; +- cases where feedback points at the wrong relevant record expose false-positive trade-offs. + +Normalized duplicate and near-duplicate cases cannot cross development and holdout. A fixture validator rejects invalid feedback identity, scope, lifecycle, labels, targets, quotas, or hashes before any score is calculated. + +### 6. Separate local-benefit gates from global non-regression gates + +The primary benefit metric is labeled pair-ordering accuracy inside eligible near-tie cases. The candidate must improve this metric under a pre-registered point estimate and paired-bootstrap lower-bound gate. Global Recall@K, Precision@K, MRR, abstention, explicit forbidden selections, and authorization violations are separate non-regression or zero-tolerance gates. + +This avoids demanding that a tie-breaker drive a large global Recall/MRR gain while still preventing an easy pass based only on hand-picked local examples. Exact deltas, confidence bounds, resamples, seed, and tolerances belong to the versioned decision plan and are frozen before holdout. + +### 7. Use a two-stage lock and one holdout decision + +Development execution has three ordered steps: + +1. create the baseline-only calibration report and finite boundary grid; +2. evaluate every declared evidence-rule and boundary combination; +3. apply a deterministic selection rule and write a lock artifact. + +The lock includes fixture and manifest hashes, selected candidate, gates, runtime/evaluator identity, seed, and resource ceilings. Holdout execution refuses to run without the lock and writes one immutable evidence file. Any later parameter or label change requires a new decision version. + +Preparation hardening (2026-09-17): lock validation recomputes relevance/safety +gates from reported metrics and verifies the complete ordered candidate grid. +This does not replace reproducing those metrics from frozen development inputs. +The durable writer reserves `.json` with exclusive creation before +calling the evaluator, validates output identity, development metrics and decision +consistency, then flushes the evidence. A failed or interrupted attempt leaves an +empty/partial file and remains consumed. Do not delete it to retry. All invocations +must use the same reviewed output directory; this is not protection against a +human copying the experiment elsewhere or deliberately bypassing the writer. +The synchronous helper is a preflight guard, not a durable execution registry. +Independent review remains external; a caller-supplied boolean is not proof of it. +The final scoring runner must supply measured privacy/determinism/resource gates; +the writer validates private strings and output size but does not invent those +measurements. No formal holdout has run during this hardening work. + +Correction evidence retains the frozen v1 trace name `correctionCount`, but +counts events whose `replacementMemoryId` is the admitted record. Production +aggregates instead count corrections of `event.memoryId` (the old record). +An integration test exercises real correction, replay, compaction and restart; +only the active replacement is admitted and receives the evaluator signal. + +### 8. Keep evaluator dependencies one-way + +Offline modules may import production contracts and pure foundation-ranking helpers. Production Memory, server, Manager, renderer, and runtime modules cannot import the evaluator. Tests enforce this boundary through import checks and production-parity assertions. + +No network, model file, native dependency, or user data directory is required. Traces use case and Memory ids plus numeric features; human-readable content remains in the checked-in fixture rather than duplicated into evidence. + +## Risks / Trade-offs + +- **Near-tie boundaries can overfit a small development set** → freeze the derivation procedure, include wide-gap and no-rerank controls, report all configurations, and require paired holdout confidence gates. +- **Human pair labels can encode author preference** → require per-case rationales, independent review of ambiguous cases, explicit disagreement recording, and separate safety labels. +- **Confirmation can reinforce an old but still active fact** → lifecycle filtering remains first, corrected old versions are excluded, and feedback cannot override wide foundation gaps. +- **Correction evidence can be double-counted with lifecycle** → trace lifecycle admission separately and compare confirmation-only with confirmation-plus-correction candidates rather than assuming correction must help. +- **A narrow local metric can hide broad regressions** → require global non-regression, abstention, hard-negative, and zero-leak gates in addition to local benefit. +- **P3-A closeout is not a separate merged PR** → keep the evaluator and production behavior independent, rebase onto the latest `develop`, and include only the necessary canonical spec synchronization and archive move in the single final P3-B PR. + +## Migration Plan + +1. Add and validate the new anonymous fixture schema, manifest, and baseline-only calibration report. +2. Add the evaluator candidate grid, deterministic lock artifact, and development-only reports. +3. Review and freeze the selected candidate and every gate before enabling the holdout command. +4. Run holdout once and publish the immutable go/no-go evidence. +5. Merge evaluation artifacts without production imports or runtime changes. If the result is go, create a separate production-integration OpenSpec change; if no-go, stop without rollback because production was never changed. + +### V1 execution disposition (2026-09-23) + +The planned one-time holdout step was not completed as specified. The runner was +found to permit repeated scoring in temporary output directories and to report +privacy, determinism, and resource gates without measuring all of them. Its v1 +entry point is retired; the historical output is preserved but is not +decision-grade evidence. Because the locked selection is the unchanged +foundation control and every feedback candidate failed the development safety +gate, this version is closed at development no-go without another holdout run. +This is a documented execution deviation, not a holdout no-go. Any renewed +holdout evaluation requires a new decision version and fresh review. diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/development-review.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/development-review.md new file mode 100644 index 000000000..2a4af2bfa --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/development-review.md @@ -0,0 +1,119 @@ +# P3-B development review + +## Outcome and scope + +The frozen `p3-feedback-tiebreaker-v1` development grid has no eligible feedback +candidate. The contributor closes this version at development no-go; there is no +decision-grade holdout result, and no holdout metric is part of this decision. +The lock selects `foundation-control` because the pre-registered selection rule +uses it as the fallback. A valid lock is not a passed decision gate. + +Production ranking, canonical Memory, and frozen P3-A evidence are unchanged. + +## Holdout review and correction + +An independent Opus 4.8 review confirmed the frozen inputs and lock without reading +holdout labels or results. Its development-label assessment was summary-report based, +not a per-case label audit. A historical foundation-control run exists, but it was +later found to have been executed repeatedly in temporary output directories and to +have incomplete gate measurement. See `holdout-execution-audit.md`; it is not +decision-grade holdout evidence. No production ranking path was changed. Even a +future go would require a separate production-integration change. + +## Method and evidence + +Artifacts are under `kun/src/memory/fixtures/`, with the common prefix +`memory-feedback-tiebreaker-` and suffix `.v1.json`: + +- `calibration`: development-only adjacent foundation gaps, floor-index + quantiles 0.25/0.5/0.75, deduplicated inclusive boundaries 0/0.0075/0.01125. +- `plan`: seven configurations; foundation control plus confirmation-only and + confirmation-plus-correction at each boundary. Leader-relative groups have + a maximum reorder window of three. Frequency is shadow-only. +- `development`: 18 synthetic cases, including four labeled preference pairs, + 16 ranked cases and two no-result cases. These are not real-traffic estimates. +- `lock`: fixed candidate, gates, dependency hashes, evaluator identity and seed. + +Bootstrap uses 10,000 paired case resamples, seed 20260916, confidence 0.95. +Calibration boundaries and all gates remain frozen; no labels or parameters +were changed after inspecting this outcome. + +| Development metric | Foundation | Confirmation + correction, gap_1 or gap_2 | +| --- | --- | --- | +| Pair accuracy | 0 | 0.75 | +| Recall / Precision / MRR | 0.4375 | 0.625 | +| Pair gain lower bound | 0 | 0.25 | +| Recall / MRR gain lower bound | 0 | 0.0625 | +| Abstention accuracy | 1 | 1 | +| Explicit forbidden selections | 6 | 3 | +| Authorization / lifecycle violations | 0 | 0 | +| All required gates passed | false | false | + +Both improved configurations pass local-benefit and global non-regression +checks, but fail the zero-explicit-forbidden-selection gate. Reduced errors +are not zero errors. These same-scope relevance failures must not be described +as authorization leaks. The foundation also fails this absolute gate; it is +retained as an unchanged control, not certified as error-free. + +## Privacy, resources and reproducibility + +Traces are restricted to synthetic ids and numeric features; query text, +Memory content, source excerpts, credentials and machine paths are excluded. +Resource ceilings are 5,000 ms, 64 rankings per trace and 1,000,000 serialized +evidence bytes. These are evaluator limits, not measured production overhead +or a memory-RSS budget. Resource tests measure development execution and reject +over-limit output without truncation; no holdout resource result is claimed. + +From `kun/`, run the installed Vitest executable with: + +```text +vitest run src/memory/memory-feedback-tiebreaker +``` + +This verifies fixture integrity, development reproduction, lock validation, +safety, privacy, resource bounds and holdout guards. Guard tests do not publish +a scored holdout decision. Full repository gates must run again after rebasing. + +## Final decision and holdout limitation (2026-09-23) + +An independent Opus 4.8 review confirmed the candidate lock, frozen input hashes, +development summary, and pre-registered fallback without reading holdout labels +or results. That review authorized a run but did not audit the later runner and +does not certify its execution. The subsequent execution audit found repeated +temporary-directory scoring and incomplete gate measurement. The historical +output is therefore retained only as a record of the flawed execution, not as a +holdout decision. + +The contributor closes v1 at the development no-go rather than rerunning the +retired version. This is an explicit deviation from the original one-holdout +plan; it does not convert the development result into a holdout result. The +selected fallback is an unchanged control, and every feedback candidate failed +the zero-explicit-forbidden-selection gate on development. No gate was relaxed, +no frozen data was edited, and production ranking remains unchanged. A future +attempt must use a new decision version and independently reviewed execution +controls. + +During an earlier source review, some holdout label text appeared in search +output; no holdout metrics were computed and no parameters were tuned. That +review is not represented as holdout-label-blind. This limitation remains part +of the audit trail. + +## Label review limitations (2026-09-17) + +The frequent-unconfirmed development control labels package lint forbidden while +package validation is expected, although both active records answer a broad +package-check query and neither has explicit confirmation/correction evidence. +The misleading-feedback control labels three review passes current while the +two-pass record is active, confirmed and has higher importance. Such cases can +expose the limits of feedback, but require acknowledging that a bounded +tie-breaker lacks evidence to infer the hidden label and must also preserve +neutral ordering. Absolute zero-forbidden selection is still the frozen rule; +this review does not relax it or relabel examples to pass. + +These limitations constrain interpretation of the development rejection; they +do not invalidate or rewrite its recorded numbers. A later independently +reviewed dataset would need a new version, not edits to these frozen inputs. +During source review, some holdout label text appeared in search output. No +holdout metrics were computed and no parameters were tuned. Nevertheless, this +review is not holdout-label-blind or an independent approval of the author's +implementation. Obtain a separate review before resolving task 4.6. diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/holdout-execution-audit.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/holdout-execution-audit.md new file mode 100644 index 000000000..e153d8fb5 --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/holdout-execution-audit.md @@ -0,0 +1,43 @@ +# Holdout execution correction + +This audit supersedes earlier claims of a compliant single holdout run and fully +measured holdout gates. Production ranking and the frozen development artifacts +are unchanged. The development no-go remains the supported decision. + +## Execution deviation + +The independent Opus 4.8 review authorized one foundation-control holdout. +The runner test was incorrectly described as synthetic: it loaded the real frozen +holdout, scored it in temporary directories, and deleted temporary outputs during +cleanup. The recorded session shows at least five successful test invocations plus +the repository write. The artifact's `holdoutRunCount: 1` does not account for those +attempts or internal determinism replays; it is not proof of one-shot compliance. + +The same-directory exclusive writer worked, but changing directories bypassed +experiment-level discipline. The real-data runner is now retired and its +environment-triggered execution test removed. The regression test checks rejection +without loading/scoring data. Generic writer tests retain synthetic payloads. + +## Measurement limitations + +- Development privacy/determinism/resource flags were assigned true, not measured + by this runner. +- Holdout timing excluded bootstrap and deterministic replay; the determinism + check compared selected IDs, not full numeric scores and metrics. +- Privacy validation checked the final payload, not all intermediate traces. +- Bootstrap lower bounds were originally literal zero for the identity control. + Later computation cannot retroactively certify the first execution. +- The loader checked fixture/manifest checksums, but the runner did not explicitly + bind those hashes to the plan. Non-null boundaries also silently used zero. + +## Final disposition (2026-09-23) + +Preserve `kun/src/memory/fixtures/kun-memory-feedback-tiebreaker-v1.json` unchanged +as historical output, NOT independent decision-grade holdout evidence. Do not edit +its flags, delete it, or rerun v1 to make the history appear compliant. The +pre-execution review remains valid within its scope; it does not certify the later +implementation. The contributor closes v1 at the development no-go and accepts +the execution deviation explicitly: no valid holdout result exists, and no holdout +metrics are used in the decision. Task 4.6 records this closure rather than a +completed holdout run. Any renewed holdout work requires a new decision version, +fresh execution controls, and independent review. Production ranking is unchanged. diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/proposal.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/proposal.md new file mode 100644 index 000000000..83bdbb66a --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/proposal.md @@ -0,0 +1,29 @@ +## Why + +P3-A established a private feedback ledger and explicit confirmation/correction flows, but its first additive ranking candidate produced no holdout gain and correctly remained out of production. A separately versioned decision is needed to test the narrower intended use of feedback: resolving genuinely close lexical candidates without weakening global relevance, abstention, authorization, lifecycle, privacy, determinism, or desktop resource bounds. + +## What Changes + +- Add an anonymous, deterministic evaluation dataset centered on human-labeled near-tie lexical cases, with separate development and holdout splits, fixed hashes, explicit relevance rationales, and scope/lifecycle hard negatives. +- Pre-register a finite near-tie definition grid and bounded feedback candidates before examining holdout results. The grid is derived from foundation-score distributions and labeled error costs; no percentage or weight is treated as proven in advance. +- Compare the unchanged post-#1308 lexical foundation with evaluator-only tie-breakers that can use explicit confirmation and correction evidence only inside the admitted near-tie window. +- Keep retrieval frequency shadow-only so repeated exposure cannot create a self-reinforcing production advantage. +- Report near-tie pair-ordering accuracy separately from global Recall@K, Precision@K, MRR, abstention, false-positive selection, safety, determinism, privacy, and resource metrics. Use deterministic paired bootstrap intervals for global non-regression and any claimed local benefit. +- Preserve dev-to-holdout discipline: select only from the pre-registered grid on development data, lock the candidate and gates, then run holdout once. A no-go is a complete and valid outcome. +- Keep all candidates offline. A go recommendation can authorize a later production-integration proposal but cannot change retrieval behavior, add a dormant ranking flag, or modify canonical Memory in this change. + +## Capabilities + +### New Capabilities + +- `memory-feedback-ranking-evaluation`: Defines the anonymous data contract, near-tie candidate boundary, shadow-only signals, uncertainty and safety gates, holdout discipline, and evaluator-only production isolation for a versioned P3-B feedback-ranking decision. + +### Modified Capabilities + +None. + +## Impact + +- Adds versioned anonymous fixtures, manifests, decision plans, evaluator code, traces, evidence, and focused tests under the Kun Memory evaluation surface. +- Reads existing lexical foundation scores and derives explicit replacement evidence from P3-A feedback events through offline adapters only; replacement evidence is not the old record's production aggregate correction count. +- Does not modify production Memory retrieval, prompt injection, canonical records, Manager/HTTP/UI behavior, settings, packaged dependencies, or the frozen P3-A v1 plan and evidence. diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/specs/memory-feedback-ranking-evaluation/spec.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/specs/memory-feedback-ranking-evaluation/spec.md new file mode 100644 index 000000000..62b83d6a6 --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/specs/memory-feedback-ranking-evaluation/spec.md @@ -0,0 +1,167 @@ +## Purpose + +Define a reproducible and privacy-preserving decision process for testing whether explicit Memory feedback can safely resolve close lexical rankings without changing production behavior. + +## ADDED Requirements + +### Requirement: Decision data is anonymous, stratified, and frozen + +The evaluator SHALL use a strictly validated synthetic Memory, feedback, and query dataset with separate development and holdout splits. The versioned manifest SHALL record exact split counts, stratum quotas, fixed evaluation time, normalization rules, human-readable relevance and pair-preference rationales, and SHA-256 hashes for every decision artifact. It SHALL include near-tie positive cases, wide-gap controls, no-feedback controls, high-frequency-unconfirmed controls, no-result queries, and scope/lifecycle hard negatives. + +#### Scenario: Run the default evaluation + +- **WHEN** a developer runs the feedback tie-breaker evaluation without an explicit fixture override +- **THEN** it reads only checked-in synthetic data and does not inspect canonical user Memory or feedback files + +#### Scenario: Validate a malformed dataset + +- **WHEN** a fixture contains an unknown field, duplicate id, invalid Memory or feedback event, unresolved target, conflicting label, missing rationale, invalid split quota, or unsupported version +- **THEN** the evaluator rejects the complete dataset before scoring a candidate + +#### Scenario: Modify a frozen artifact + +- **WHEN** a record, event, query, label, rationale, split, candidate rule, gate, or normalization rule changes +- **THEN** the decision version and affected hashes change before new evidence can be accepted + +### Requirement: Near-tie boundaries are derived and pre-registered + +The decision plan SHALL define near-tie from lexical foundation scores through a finite, deterministic candidate grid selected on development data only. The plan SHALL record the gap formula, boundary inclusivity, grouping behavior, maximum reorder window, stable tie order, and derivation evidence before holdout evaluation. No boundary or effect limit SHALL be treated as proven merely because it was suggested in planning. + +#### Scenario: Compare a near-tie pair + +- **WHEN** two authorized active candidates fall inside a declared near-tie boundary +- **THEN** an evaluator candidate may reorder only those admitted candidates according to its pre-registered explicit-feedback rule + +#### Scenario: Compare a wide-gap pair + +- **WHEN** the foundation-score gap exceeds the locked near-tie boundary +- **THEN** feedback cannot reverse their foundation order + +#### Scenario: Encounter an exact boundary + +- **WHEN** a score gap equals a declared boundary +- **THEN** every repeated run applies the recorded inclusive or exclusive rule and produces the same order + +### Requirement: Explicit evidence is bounded and retrieval frequency is shadow-only + +Candidate scoring SHALL distinguish explicit confirmation, explicit correction of the active replacement, retrieval frequency, and absence of feedback. Retrieval count, last-retrieved time, or repeated exposure SHALL be reported for analysis but SHALL NOT affect candidate admission, ordering, selection, or decision gates. Confirmation and correction evidence SHALL affect only candidates already admitted to the same near-tie window after scope and lifecycle filtering. + +#### Scenario: Compare a frequently retrieved unconfirmed record + +- **WHEN** a near-tie record has many retrieval impressions but no explicit confirmation or correction +- **THEN** its frequency is visible in the trace but cannot improve its rank + +#### Scenario: Prefer an explicitly confirmed record inside a near tie + +- **WHEN** the labeled preferred record is explicitly confirmed and falls inside the locked near-tie window +- **THEN** a confirmation-aware candidate may rank it ahead according to the locked deterministic rule + +#### Scenario: Exclude a superseded record with historical feedback + +- **WHEN** a superseded record has stronger historical feedback than its active replacement +- **THEN** lifecycle filtering excludes the old record before feedback evaluation and feedback cannot restore it + +### Requirement: Authorization and abstention precede feedback evaluation + +The evaluator SHALL apply the production scope, authority, lifecycle, positive-relevance, result-count, and prompt-budget boundaries before feedback-specific processing. A feedback candidate SHALL NOT introduce a record rejected by the lexical foundation, fill an empty result, cross scope, change authority, or reactivate a disabled, deleted, expired, future-valid, or superseded record. + +#### Scenario: Foundation retrieval abstains + +- **WHEN** the lexical foundation admits no positively relevant record +- **THEN** every feedback candidate returns no result regardless of feedback history + +#### Scenario: Out-of-scope record has stronger feedback + +- **WHEN** an unauthorized record has more confirmation or correction evidence than all authorized records +- **THEN** it is absent from scores, selected ids, traces, and context projections + +#### Scenario: Feedback would exceed a prompt budget + +- **WHEN** a reordered candidate set would exceed the shared result or character budget +- **THEN** evaluation applies the unchanged production budget and reports only the records that fit + +### Requirement: Local benefit and global non-regression are evaluated separately + +The report SHALL measure human-labeled near-tie pair-ordering accuracy separately from Recall@K, Precision@K, mean reciprocal rank, abstention accuracy, explicit hard-negative selections, and authorization-safety violations. It SHALL publish per-case paired deltas against the unchanged post-#1308 lexical foundation and deterministic bootstrap confidence intervals using the pre-registered seed, resample count, confidence level, and gate thresholds. + +#### Scenario: Improve a labeled near-tie pair + +- **WHEN** a candidate correctly reverses a foundation misordering inside the declared near-tie window +- **THEN** the report counts the local ordering improvement and independently reports its effect on global metrics + +#### Scenario: Improve local order but regress global quality + +- **WHEN** near-tie pair accuracy improves while a required global non-regression, abstention, or safety gate fails +- **THEN** the final decision is no-go + +#### Scenario: Report uncertainty + +- **WHEN** development or locked holdout evaluation completes +- **THEN** the report includes point estimates, paired confidence intervals, lower-bound gate results, and the exact evaluated case count for every decision metric + +### Requirement: Development tuning is bounded and holdout-blind + +The evaluator SHALL run only the pre-declared finite candidate grid on development data, record every result, select at most one candidate using the locked selection rule, and persist a lock artifact containing the chosen candidate, decision gates, artifact hashes, and evaluator identity. Holdout evaluation SHALL fail until that lock is valid and SHALL produce a final decision at most once for the decision version. + +#### Scenario: Tune the candidate grid + +- **WHEN** a developer runs development evaluation +- **THEN** every declared boundary and explicit-signal combination is evaluated in deterministic order without exposing holdout metrics + +#### Scenario: Attempt holdout before locking + +- **WHEN** the candidate, gates, hashes, or evaluator identity are not locked +- **THEN** holdout evaluation fails without emitting holdout labels, per-case results, or aggregate metrics + +#### Scenario: Attempt to retune after holdout + +- **WHEN** a holdout decision exists and any candidate parameter or gate changes +- **THEN** the evaluator requires a new decision version rather than overwriting the frozen evidence + +#### Scenario: Report flags contradict metrics + +- **WHEN** a development report's gate flags or candidate grid disagree with its metrics and pre-registered plan +- **THEN** candidate lock creation fails rather than trusting a reported passing flag + +#### Scenario: Concurrent or interrupted holdout write + +- **WHEN** two executions use the same decision output directory or a previous execution was interrupted after reserving its evidence file +- **THEN** exclusive creation permits at most one scoring attempt and preserves incomplete evidence for review instead of automatically retrying + +#### Scenario: Final output changes locked development evidence + +- **WHEN** a holdout callback returns another candidate identity, changed development metrics, or a decision inconsistent with its gates +- **THEN** the writer rejects the payload and retains the consumed attempt without publishing it as valid evidence + +### Requirement: Evaluation is deterministic, private, and resource-bounded + +Candidate identity SHALL include the evaluator version, foundation version, near-tie rule, signal rule, normalization, fixed evaluation time, and artifact hashes. Repeated runs SHALL produce identical selected ids and scores within the declared tolerance, SHALL perform no network request, SHALL omit query text, Memory content, source excerpts, credentials, and machine paths from bounded traces, and SHALL enforce pre-registered runtime and trace-size ceilings. + +#### Scenario: Repeat a locked evaluation + +- **WHEN** the same locked artifacts and runtime inputs are evaluated again +- **THEN** candidate order, selected ids, metrics, and decision results are reproducible within the declared tolerance + +#### Scenario: Instrument network access + +- **WHEN** network access is denied or instrumented to fail during evaluation +- **THEN** the evaluator completes from local synthetic artifacts without attempting a network request + +#### Scenario: Exceed a resource ceiling + +- **WHEN** evaluation time, trace count, or serialized evidence exceeds a pre-registered bound +- **THEN** the affected candidate fails the resource gate and cannot receive a go recommendation + +### Requirement: A decision cannot change production ranking + +This change SHALL remain evaluator-only. A go result SHALL authorize only a separate production-integration proposal, while a no-go result SHALL preserve the P3-A feedback infrastructure and unchanged lexical production ranking. Neither outcome SHALL add a dormant ranking flag, import evaluator code from a production module, or mutate canonical Memory and feedback data. + +#### Scenario: Evaluation concludes go + +- **WHEN** one locked candidate passes every local-benefit, global non-regression, uncertainty, safety, determinism, privacy, and resource gate +- **THEN** the report records a go recommendation and production behavior remains unchanged pending a separate OpenSpec change + +#### Scenario: Evaluation concludes no-go + +- **WHEN** no candidate passes every pre-registered gate +- **THEN** the report records no-go, retains P3-A confirm/correct/audit behavior, and leaves lexical ranking unchanged diff --git a/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/tasks.md b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/tasks.md new file mode 100644 index 000000000..13ec1f7ca --- /dev/null +++ b/openspec/changes/evaluate-kun-memory-feedback-tiebreaker/tasks.md @@ -0,0 +1,67 @@ +## 1. Versioned Contracts And Pre-registration + +- [x] 1.1 Add strict schemas for the P3-B fixture, manifest, calibration report, decision plan, candidate lock, and evidence artifacts; verify unknown fields, unsupported versions, duplicate ids, and invalid hashes are rejected by focused contract tests. +- [x] 1.2 Define the baseline-only adjacent-gap calibration procedure, including exact-tie/no-rerank controls, finite grid-size ceiling, boundary inclusivity, group construction, and maximum reorder window; verify a deterministic fixture produces the expected ordered grid. +- [x] 1.3 Define the pre-registered candidate family and selection rule for foundation-only, confirmation-aware, and confirmation-plus-correction-aware comparators; verify retrieval frequency is present only as a shadow trace field and cannot affect comparator output. +- [x] 1.4 Define local-benefit, global non-regression, bootstrap, safety, privacy, determinism, and resource gates in the versioned plan before candidate holdout scoring; verify the plan hash changes when any gate or candidate rule changes. + +## 2. Anonymous Decision Data + +- [x] 2.1 Create a synthetic Memory and feedback corpus with near-tie confirmation, active corrected replacement, high-frequency-unconfirmed, no-feedback, wide-gap, no-result, scope, and lifecycle strata; verify every record and event passes the production contracts without using real user data. +- [x] 2.2 Add development and holdout queries with expected sets, preferred/unpreferred pairs, explicit hard negatives, stratum labels, and human-readable rationales; verify exact quotas and referenced ids through the fixture validator. +- [x] 2.3 Reject normalized duplicate and near-duplicate cases across splits and reject ambiguous labels or invalid preference targets; verify focused negative fixtures fail before retrieval executes. +- [x] 2.4 Publish the versioned manifest and SHA-256 checksums for all frozen inputs; verify a one-byte fixture change invalidates the manifest. + +## 3. Offline Candidate Evaluation + +- [x] 3.1 Reproduce the unchanged post-#1308 lexical foundation ranking from authorized active fixture records and emit adjacent foundation-score gaps; verify selected ids and scores match direct foundation retrieval for every case. +- [x] 3.2 Implement deterministic near-tie grouping for every declared boundary with foundation-score and stable-id fallback order; verify exact-boundary, multi-record group, and wide-gap behavior. +- [x] 3.3 Implement confirmation-aware and confirmation-plus-correction-aware evaluator comparators that reorder only admitted near-tie candidates; verify they cannot add candidates, fill abstentions, cross budgets, or restore inactive records. +- [x] 3.4 Emit retrieval count and last-retrieved time as bounded shadow features; verify changing only retrieval frequency leaves every candidate order and selected id byte-for-byte unchanged. +- [x] 3.5 Enforce evaluator-only imports and offline execution; verify production Memory, Manager, server, runtime, and renderer modules do not import P3-B evaluator modules and an instrumented network attempt count remains zero. + +## 4. Metrics, Locking, And Holdout Discipline + +- [x] 4.1 Report near-tie pair-ordering accuracy separately from Recall@K, Precision@K, MRR, abstention, explicit hard-negative selection, and authorization violations; verify metric denominators and per-case deltas on hand-calculated fixtures. +- [x] 4.2 Add deterministic paired-bootstrap intervals using the locked seed, resample count, confidence level, and unit; verify repeated runs produce identical intervals and lower-bound results. +- [x] 4.3 Run the complete finite grid on development data and write a calibration report containing every evaluated configuration; verify configuration ordering and selected ids are reproducible. +- [x] 4.4 Apply the pre-registered selection rule and write one candidate lock containing artifact hashes, evaluator identity, selected candidate, gates, seed, and resource ceilings; verify tampering with any dependency invalidates the lock. +- [x] 4.5 Prevent holdout scoring without a valid lock and prevent overwriting completed holdout evidence for the same decision version; verify rejected attempts emit no holdout metrics or per-case results. +- [x] 4.6 Close v1 at development no-go after auditing holdout integrity: preserve the historical output as non-decision-grade, retire the v1 runner, and prohibit rerunning this version. This is an explicit deviation from the originally planned holdout run, not a claim that holdout was run or passed; verify the final decision and limitations are documented without relying on holdout metrics. + +## 5. Safety, Privacy, And Resource Coverage + +- [x] 5.1 Add scope, authority, disabled, deleted, expired, future-valid, and superseded hard-negative tests; verify every authorization or lifecycle violation forces no-go with zero tolerance. +- [x] 5.2 Add no-result and positive-relevance tests; verify feedback candidates preserve lexical abstention and never fill the result budget with unrelated records. +- [x] 5.3 Add privacy tests for query text, Memory content, source excerpts, credentials, Windows/UNC/POSIX/file URLs, and oversized diagnostics; verify traces retain only bounded synthetic ids, numeric features, and safe evaluator metadata. +- [x] 5.4 Add deterministic replay and serialization tests across supported runtime environments; verify selected ids, numeric scores within tolerance, metrics, hashes, and decisions remain stable. +- [x] 5.5 Measure evaluation duration, trace count, and serialized artifact bytes against pre-registered ceilings; verify a synthetic over-limit candidate fails the resource gate without truncating decision-critical evidence. + +## 6. Baseline Synchronization And Documentation + +Closeout checkpoint (2026-09-23): the branch is rebased on +`upstream/develop@65f6a55c4`. The frozen development grid has no eligible +feedback candidate. The holdout audit found repeated temporary-directory scoring +and unmeasured gates; the historical output is preserved but is not decision-grade, +and the v1 runner now rejects every attempt. The contributor closes this version +at development no-go without claiming a holdout result. Production ranking and +all frozen v1 inputs remain unchanged. + +Review hardening: tasks 4.4/4.5 now additionally cover recomputed gate flags, +complete grid validation, exclusive evidence-file reservation, interruption and +concurrent execution rejection, and binding final output to locked development. +Correction integration now covers real service events through compaction/restart. +The frozen fixtures and lock are unchanged; label concerns and holdout-label +exposure are recorded in `development-review.md`, not silently resolved. + +- [x] 6.1 Fetch the latest `upstream/develop`, rebase this branch, and verify the final diff contains the P3-B evaluation capability plus the required P3-A canonical spec synchronization/archive closeout, with no importer or production-ranking changes. +- [x] 6.2 Document the calibration method, candidate identities, local/global metrics, privacy model, resource results, and the development no-go; state that no decision-grade holdout result exists, production integration requires a separate passed decision, and P3-A infrastructure is retained. +- [x] 6.3 Update `D:\learning\Review_md\kun-memory-roadmap.md` and create a stage note under `D:\learning\Review_md\codex`; record the commit series, checks, immutable development-input hashes, historical holdout execution limitation, no-go disposition, and remaining P4-A work. + +## 7. Validation And Delivery + +- [x] 7.1 Run the focused P3-B contract, fixture, evaluator, safety, privacy, determinism, and workflow tests; verify all new cases pass. +- [x] 7.2 Run the existing Kun Memory focused suites, `npm run build:kun`, `npm run typecheck`, `npm run build`, `npm run lint`, and `npm run check:file-lines`; separate any pre-existing baseline failures from change-introduced failures. +- [x] 7.3 Run strict OpenSpec validation and `git diff --check`; verify every authored text file remains within 700 physical lines. +- [x] 7.4 Commit each meaningful artifact group with Angular-style messages and push the preparation branch to SunwardL; keep the formal PR closed until the decision and documentation tasks are complete, and do not open a separate P3-A closeout PR. +- [x] 7.5 Create one PR targeting `KunAgent/Kun:develop` with the P3-B evaluation and P3-A spec/archive closeout, including Summary, Changes, Tests, decision outcome, evidence hashes, and the unchanged-production-ranking statement; verify PR #1339 uses base `develop` and head `SunwardL:codex/prepare-memory-feedback-tiebreaker`. diff --git a/openspec/specs/memory-feedback-evolution/spec.md b/openspec/specs/memory-feedback-evolution/spec.md new file mode 100644 index 000000000..79ad5be14 --- /dev/null +++ b/openspec/specs/memory-feedback-evolution/spec.md @@ -0,0 +1,144 @@ +# memory-feedback-evolution Specification + +## Purpose +Define a privacy-bounded and auditable feedback history that distinguishes Memory injection from explicit user confirmation and correction, supports versioned evolution, and permits offline ranking evaluation without granting feedback new authority. + +## Requirements + +### Requirement: Feedback events are strict, private, and idempotent + +Kun SHALL validate versioned feedback events with a unique event id, event kind, target Memory id, occurrence time, and bounded turn identity when applicable. Feedback events SHALL NOT persist query text, Memory content, model output, credentials, source excerpts, or local filesystem paths. Replaying an event id with the same payload SHALL be a no-op, while replaying it with a different payload SHALL fail closed. + +#### Scenario: Record an injected memory + +- **WHEN** an authorized active Memory is selected and assembled into a model turn context +- **THEN** Kun may record one `retrieved` event containing bounded identity metadata but no query or Memory body + +#### Scenario: Replay an event after reconnect + +- **WHEN** Manager or runtime replay submits an already persisted event id with the same payload +- **THEN** the ledger retains one event and aggregate counts do not increase + +#### Scenario: Reuse an event id with different data + +- **WHEN** a caller submits an existing event id with a different kind, Memory id, time, or version link +- **THEN** validation rejects the conflicting event without modifying durable feedback state + +### Requirement: Retrieval impressions and user confirmation remain distinct + +Kun SHALL treat a retrieval impression only as evidence that a Memory was included in generated context. It SHALL NOT treat retrieval, model continuation, repeated ranking, or UI search as user confirmation. A confirmation or correction SHALL require an explicit user action against an authorized visible Memory. + +#### Scenario: Search the Memory settings list + +- **WHEN** a user searches, filters, previews, or opens a Memory without confirming it +- **THEN** no retrieval, confirmation, or correction feedback event is recorded + +#### Scenario: A turn receives Memory context + +- **WHEN** a Memory reference block is assembled for a turn and the model request proceeds +- **THEN** the event contributes only to retrieval usage and does not increase confirmation or correctness signals + +#### Scenario: User explicitly confirms a Memory + +- **WHEN** the user invokes confirmation for an active Memory visible in the current scope +- **THEN** Kun records one explicit confirmation event attributable to that user action + +### Requirement: Feedback persistence is recoverable and independently diagnosable + +Kun SHALL persist feedback through one Manager-owned append-only ledger per data root, SHALL serialize local and remote mutations through the owning process, and SHALL expose bounded diagnostics for malformed tails, duplicate events, projection state, and write failures. Feedback corruption or unavailability SHALL NOT corrupt canonical Memory records. + +#### Scenario: Process exits during an append + +- **WHEN** the feedback ledger ends with a partial or malformed event after restart +- **THEN** Kun preserves valid prior events, excludes the malformed tail from aggregation, and reports a bounded degraded reason + +#### Scenario: Multiple runtimes record the same event + +- **WHEN** GUI, TUI, or reconnected runtime clients submit the same stable event through Manager +- **THEN** the Manager-owned ledger commits it at most once and all clients observe the same aggregate state + +#### Scenario: Feedback storage is unavailable + +- **WHEN** the ledger cannot open, append, validate, or rebuild its projection +- **THEN** diagnostics report feedback degradation without disabling canonical Memory CRUD or retrieval + +### Requirement: Usage aggregates are rebuildable and do not mutate Memory freshness + +Kun SHALL derive bounded per-Memory retrieval, confirmation, and correction counts and last-event timestamps from validated feedback events. Aggregate state SHALL be a rebuildable projection and SHALL NOT update canonical Memory content, `updatedAt`, `observedAt`, confidence, importance, or freshness. + +#### Scenario: A memory is injected repeatedly + +- **WHEN** multiple distinct retrieval events are recorded for one Memory +- **THEN** its retrieval count and last-retrieved time advance while its canonical timestamps and ranking features remain unchanged + +#### Scenario: Rebuild feedback aggregates + +- **WHEN** an aggregate projection is missing, stale, or corrupt while valid feedback events remain +- **THEN** Kun rebuilds equivalent counts and timestamps without rewriting Memory records + +#### Scenario: An inactive memory has historical feedback + +- **WHEN** a Memory becomes disabled, deleted, expired, or superseded +- **THEN** its feedback remains auditable but cannot restore lifecycle eligibility or make the record injectable + +### Requirement: Corrections create auditable Memory versions + +An explicit correction SHALL create a new canonical Memory in the same scope, link it to the prior record with `supersedes`, mark the prior record superseded through the canonical mutation path, and record the old and new ids in correction feedback. The operation SHALL be retry-safe and SHALL NOT overwrite the prior content or erase its evidence. + +#### Scenario: Correct an active memory + +- **WHEN** a user submits corrected content for an authorized active Memory +- **THEN** Kun creates one new reference-authority record, supersedes the old record, records one correction event, and returns the new version + +#### Scenario: Retry after an uncertain response + +- **WHEN** a correction request is replayed after its canonical mutation or feedback append may already have succeeded +- **THEN** Kun reconciles the stable operation identity and does not create another Memory version or count another correction + +#### Scenario: Correct across scopes + +- **WHEN** a correction attempts to create the new version in a different user, workspace, or project scope +- **THEN** Kun rejects the operation without mutating either record or the feedback ledger + +#### Scenario: Interrupted correction can never complete + +- **WHEN** a persisted correction receipt is reconciled after its prior Memory was purged or otherwise can never be corrected again +- **THEN** Kun marks the receipt abandoned instead of retrying on every startup, keeps only a bounded tail of terminal receipts, and answers a later replay of the same operation id with a bounded error rather than duplicating work + +### Requirement: Feedback collection is opt-in and failure-isolated + +Persistent feedback collection SHALL be disabled by default. When disabled or degraded, Kun SHALL preserve the same Memory selected ids, ordering, context text, retrieval trace, turn outcome, and canonical mutations except for an explicit correction request whose own validation or mutation fails. + +#### Scenario: Compare feedback enabled and disabled retrieval + +- **WHEN** the same records, query, policy, time, and budget are evaluated with feedback collection enabled and disabled +- **THEN** the production retrieval result, order, features, and formatted Memory context are identical + +#### Scenario: Retrieval feedback append fails + +- **WHEN** an otherwise valid turn cannot persist its best-effort retrieval event +- **THEN** the turn continues with the already selected Memory context and diagnostics record the bounded feedback failure + +#### Scenario: User disables feedback + +- **WHEN** persistent feedback collection is disabled +- **THEN** no new retrieval or confirmation events are written and core Memory retrieval remains available + +### Requirement: Feedback-aware ranking requires a separate passed gate + +This change SHALL provide anonymous deterministic fixtures and an offline candidate that reports retrieval frequency, explicit confirmation, correction, freshness, importance, confidence, and final candidate score as separate bounded features. Retrieval frequency SHALL use a pre-registered logarithmic transform. No feedback feature SHALL affect production ranking unless a separately versioned decision passes relevance, uncertainty, safety, determinism, privacy, and resource gates. + +#### Scenario: Evaluate a frequently retrieved but unconfirmed memory + +- **WHEN** an anonymous fixture contains many retrieval impressions without explicit confirmation +- **THEN** the evaluator reports the bounded frequency feature separately and does not label the Memory as confirmed or authoritative + +#### Scenario: Feedback candidate violates scope or lifecycle + +- **WHEN** a candidate selects an out-of-scope, disabled, deleted, expired, or superseded record because of feedback +- **THEN** the decision is no-go regardless of ranked-quality improvement + +#### Scenario: Offline candidate fails its benefit gate + +- **WHEN** paired evaluation does not meet every pre-registered lower-bound and safety threshold +- **THEN** production keeps the existing lexical ranking and no dormant feedback weight is introduced diff --git a/openspec/specs/memory-record-foundation/spec.md b/openspec/specs/memory-record-foundation/spec.md index 4fe8c0864..7ecaa37c9 100644 --- a/openspec/specs/memory-record-foundation/spec.md +++ b/openspec/specs/memory-record-foundation/spec.md @@ -107,7 +107,7 @@ Production serve mode SHALL route canonical memory and index operations through ### Requirement: Lifecycle mutations keep projections consistent -Create, update, disable, restore, supersede, delete, and purge operations SHALL update canonical state first and SHALL make every derived memory index converge to that state. +Create, update, disable, restore, supersede, correct, delete, and purge operations SHALL update canonical state first and SHALL make every derived memory index converge to that state. An explicit correction SHALL create a new same-scope record and supersede the prior record rather than overwriting the prior content. #### Scenario: Delete an indexed memory @@ -124,6 +124,11 @@ Create, update, disable, restore, supersede, delete, and purge operations SHALL - **WHEN** canonical mutation succeeds but index projection fails - **THEN** the operation reports canonical success with degraded diagnostics and reconciliation repairs the projection later +#### Scenario: Correct a canonical memory + +- **WHEN** an authorized user explicitly corrects the content of an active Memory +- **THEN** the new record supersedes the old record in the same scope, the old record remains auditable, and derived indexes converge without exposing both versions as active + ### Requirement: Memory index state is diagnosable without exposing secrets Memory diagnostics SHALL report canonical count, index status and schema, indexed/stale counts, backfill status, and bounded failure reasons while preserving current diagnostic fields. diff --git a/openspec/specs/memory-retrieval-foundation/spec.md b/openspec/specs/memory-retrieval-foundation/spec.md index a78ad5904..1642c477d 100644 --- a/openspec/specs/memory-retrieval-foundation/spec.md +++ b/openspec/specs/memory-retrieval-foundation/spec.md @@ -186,3 +186,27 @@ The repository SHALL include anonymous deterministic retrieval fixtures and a sc - **WHEN** developers run the evaluation on a machine with real Kun data - **THEN** the harness reads only checked-in anonymous fixtures unless the user explicitly invokes a separate local diagnostic mode + +### Requirement: Feedback capture reflects actual injection without changing retrieval + +Kun SHALL record retrieval feedback only after an authorized active Memory has survived relevance and prompt budgets and has been assembled into a turn's Memory context. Search, list, diagnostics, ranking candidates that are not selected, and evaluator runs SHALL NOT record production retrieval feedback. Feedback collection and persistence SHALL remain outside the production ranking calculation until a separate decision gate authorizes it. + +#### Scenario: Candidate is excluded by the prompt budget + +- **WHEN** a relevant Memory is ranked but omitted or truncated away by the final prompt budget +- **THEN** Kun does not record that omitted record as retrieved for the turn + +#### Scenario: Memory is included through filesystem fallback + +- **WHEN** SQLite is degraded and an authorized relevant Memory is assembled through filesystem fallback +- **THEN** Kun may record the same retrieval feedback shape without changing fallback ordering or eligibility + +#### Scenario: Feedback recording is unavailable + +- **WHEN** selected Memory context has already been assembled but feedback recording is disabled or fails +- **THEN** selected ids, ranking features, context text, and turn completion remain identical to the no-feedback baseline + +#### Scenario: Inspect a retrieval trace before ranking feedback is approved + +- **WHEN** diagnostics expose the latest production retrieval +- **THEN** its final score contains only the approved lexical-foundation features and no feedback contribution diff --git a/src/renderer/src/components/rooms/AgentModelSettings.test.ts b/src/renderer/src/components/rooms/AgentModelSettings.test.ts index 1458e7486..0dd993154 100644 --- a/src/renderer/src/components/rooms/AgentModelSettings.test.ts +++ b/src/renderer/src/components/rooms/AgentModelSettings.test.ts @@ -45,6 +45,7 @@ const api = vi.hoisted(() => ({ fastSource: 'agent' } as AgentModels })) +const initialSnapshot = api.snapshot vi.mock('./agent-client', async (original) => ({ ...(await original()), @@ -66,6 +67,7 @@ describe('AgentModelSettings immediate apply', () => { let renderer: ReactTestRenderer beforeEach(async () => { await i18n.changeLanguage('en') + api.snapshot = initialSnapshot api.refresh.mockReset() api.saveAgentModels.mockReset().mockResolvedValue({ ...api.snapshot, @@ -92,6 +94,20 @@ describe('AgentModelSettings immediate apply', () => { const select = (label: string) => renderer.root.findByProps({ 'aria-label': label }) + it('treats a model without a provider id as inheritance', async () => { + const unavailable = { ...api.snapshot.options[0], model: 'legacy-model', providerId: undefined } + api.snapshot = { ...api.snapshot, options: [...api.snapshot.options, unavailable] } + await render() + await act(async () => { + select('Main model').props.onChange({ target: { value: modelBindingKey(unavailable) } }) + }) + expect(api.saveAgentModels).toHaveBeenCalledWith('agent-1', { + expectedRevision: 2, + modelRef: null, + fastModelRef: api.snapshot.agent.fastModelRef + }) + }) + it('saves the main model as soon as it changes and does not keep a Save button', async () => { const onSaved = await render() const next = { providerId: 'deepseek', accountId: 'acct-1', model: 'deepseek-chat' }