From a542ddf31c644e3916bf95d8eb814f119c7707e2 Mon Sep 17 00:00:00 2001 From: drewOrc <36374426+drewOrc@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:59:18 +0800 Subject: [PATCH 1/2] results: learning-rate pilot selects 5e-5 for both encoders Validation only, k=100, seed 42. BERT 5e-5 reuses the equivalent AC2 run. Both optima sit at the top of the frozen grid; the protocol is not widened after seeing the results, and the README will state the limit. --- DEVLOG.md | 30 ++++ configs/curve.yaml | 6 +- results/pilots/lr.json | 313 +++++++++++++++++++++++++++++++++++++++++ 3 files changed, 347 insertions(+), 2 deletions(-) create mode 100644 results/pilots/lr.json diff --git a/DEVLOG.md b/DEVLOG.md index 9e5efdf..a33cfa4 100644 --- a/DEVLOG.md +++ b/DEVLOG.md @@ -80,6 +80,36 @@ --- +## 2026-09-23 (晚):learning rate pilot(validation only) + +### 本次工作 / 執行摘要 +- `make pilot-lr`:k=100、seed 42,兩個 encoder 各掃 {1e-5, 2e-5, 5e-5},只看 validation。BERT 5e-5 重用 AC2 seed 42(判定為等價 run),其餘 5 次新訓練。 +- Drew 確認兩者都用 5e-5,寫入 `configs/curve.yaml`。 + +### 核心發現 / 數據 +| 模型 | 1e-5 | 2e-5 | 5e-5 | +|---|---:|---:|---:| +| BERT(val in-scope / OOS recall) | 90.17% / 45% | 95.73% / 66% | **96.73% / 68%** | +| ModernBERT | 94.77% / 67% | 95.93% / 72% | **97.13% / 75%** | + +- **兩個模型的最佳值都在掃描範圍上限。** 只能說「在 {1e-5, 2e-5, 5e-5} 中 5e-5 最好」,真正的最佳 LR 可能更高。Drew 決定照凍結的協定接受 5e-5,不在看到結果後擴充範圍;兩者停在同一個邊界,比較仍對稱。README 需寫明此限制。 +- LR 太小會學不完:BERT 1e-5 在第 5 個 epoch 的 loss 仍有 0.40(5e-5 為 0.03),val in-scope 低 6.5pp。 +- 單一 seed、val 只有 100 筆 OOS,ModernBERT 與 BERT 的 OOS recall 差 7 筆,不下結論。 +- 每次 run 時間(M4):BERT 約 15 分鐘,ModernBERT 約 21 到 23 分鐘。 + +### Blockers / 遇到的問題 +- (無) + +### Next +- [ ] `make pilot-steps`(k=5、seed 42,S_min ∈ {100, 200, 400})→ Drew 確認 → 寫入 `configs/curve.yaml` +- [ ] `make curve MODEL=bert`、`make curve MODEL=modernbert`、`make oos-ablation` + +### Files / Budget +- `results/pilots/lr.json`、`configs/curve.yaml` +- API 花費:US$0 + +--- + ## 2026-09-23:AC2 通過(BERT 流程正確性檢查) ### 本次工作 / 執行摘要 diff --git a/configs/curve.yaml b/configs/curve.yaml index e9a7aaf..78a2b5e 100644 --- a/configs/curve.yaml +++ b/configs/curve.yaml @@ -9,7 +9,9 @@ min_train_steps: null models: bert: config: configs/bert-base.yaml - learning_rate: null + # results/pilots/lr.json: val in-scope 2705 / 2872 / 2902 of 3000 at 1e-5 / 2e-5 / 5e-5 + learning_rate: 5.0e-5 modernbert: config: configs/modernbert-base.yaml - learning_rate: null + # results/pilots/lr.json: val in-scope 2843 / 2878 / 2914 of 3000 at 1e-5 / 2e-5 / 5e-5 + learning_rate: 5.0e-5 diff --git a/results/pilots/lr.json b/results/pilots/lr.json new file mode 100644 index 0000000..64edf76 --- /dev/null +++ b/results/pilots/lr.json @@ -0,0 +1,313 @@ +{ + "pilot": "learning_rate", + "split": "validation", + "k": 100, + "seed": 42, + "grid": [ + 1e-05, + 2e-05, + 5e-05 + ], + "rule": "per model: max validation in_scope_correct, then oos_correct, then smaller lr", + "points": [ + { + "model": "bert", + "value": 1e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 1e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 883.27 + }, + "validation": { + "in_scope_correct": 2705, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9016666666666666, + "oos_correct": 45, + "oos_n": 100, + "oos_recall_151": 0.45, + "n": 3100 + } + }, + { + "model": "bert", + "value": 2e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 2e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 890.813 + }, + "validation": { + "in_scope_correct": 2872, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9573333333333334, + "oos_correct": 66, + "oos_n": 100, + "oos_recall_151": 0.66, + "n": 3100 + } + }, + { + "model": "bert", + "value": 5e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 5e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "reused bert-base-uncased-full-seed42 (equivalent run; validation logits from its archive)", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": null, + "train_sample_sha256": null, + "step_plan": null, + "global_step": 2385, + "train_wall_seconds": 886.184 + }, + "validation": { + "in_scope_correct": 2902, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9673333333333334, + "oos_correct": 68, + "oos_n": 100, + "oos_recall_151": 0.68, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 1e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 1e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1405.978 + }, + "validation": { + "in_scope_correct": 2843, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9476666666666667, + "oos_correct": 67, + "oos_n": 100, + "oos_recall_151": 0.67, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 2e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 2e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1251.981 + }, + "validation": { + "in_scope_correct": 2878, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9593333333333334, + "oos_correct": 72, + "oos_n": 100, + "oos_recall_151": 0.72, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 5e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 5e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1255.688 + }, + "validation": { + "in_scope_correct": 2914, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9713333333333334, + "oos_correct": 75, + "oos_n": 100, + "oos_recall_151": 0.75, + "n": 3100 + } + } + ], + "selected": { + "bert": 5e-05, + "modernbert": 5e-05 + }, + "note": "not applied automatically; copy into configs/curve.yaml by hand" +} From 057fb40d1d0ae6887351697bb56393cb884506ea Mon Sep 17 00:00:00 2001 From: drewOrc <36374426+drewOrc@users.noreply.github.com> Date: Wed, 23 Sep 2026 17:03:27 +0800 Subject: [PATCH 2/2] Check hand-copied protocol values against the committed pilot outputs The old test pinned the protocol to its unfilled state and broke as soon as the first pilot value was copied in. The new test compares every value in configs/curve.yaml with the selection recorded in results/pilots/*.json, and requires null while a pilot output is absent, so a mistyped value fails CI. --- tests/test_protocol.py | 27 ++++++++++++++++++++++----- 1 file changed, 22 insertions(+), 5 deletions(-) diff --git a/tests/test_protocol.py b/tests/test_protocol.py index c89f1f0..37212f7 100644 --- a/tests/test_protocol.py +++ b/tests/test_protocol.py @@ -33,12 +33,29 @@ def decided(tmp_path) -> CurveProtocol: return load_protocol(protocol_file(tmp_path, s_min=400, bert="5.0e-5", modern="2.0e-5")) -def test_the_committed_protocol_loads_with_both_pilots_still_open(): +def test_the_committed_protocol_matches_the_committed_pilot_outputs(): + """Values copied by hand into configs/curve.yaml must equal what the pilots selected. + + A pilot whose output is not committed yet must still be null in the protocol. + """ + import json + protocol = load_protocol(ROOT / "configs" / "curve.yaml") - assert protocol.min_train_steps is None - assert protocol.learning_rates == {"bert": None, "modernbert": None} - with pytest.raises(ProtocolError, match="make pilot-lr"): - protocol.curve_configs("bert") + lr_file = ROOT / "results" / "pilots" / "lr.json" + if lr_file.exists(): + selected = json.loads(lr_file.read_text())["selected"] + assert json.loads(lr_file.read_text())["split"] == "validation" + assert protocol.learning_rates == selected + else: + assert protocol.learning_rates == {"bert": None, "modernbert": None} + with pytest.raises(ProtocolError, match="make pilot-lr"): + protocol.curve_configs("bert") + steps_file = ROOT / "results" / "pilots" / "steps.json" + if steps_file.exists(): + assert json.loads(steps_file.read_text())["split"] == "validation" + assert protocol.min_train_steps == json.loads(steps_file.read_text())["selected"] + else: + assert protocol.min_train_steps is None def test_curve_refuses_to_start_until_s_min_is_chosen(tmp_path):