diff --git a/DEVLOG.md b/DEVLOG.md index 9e5efdf..a33cfa4 100644 --- a/DEVLOG.md +++ b/DEVLOG.md @@ -80,6 +80,36 @@ --- +## 2026-09-23 (晚):learning rate pilot(validation only) + +### 本次工作 / 執行摘要 +- `make pilot-lr`:k=100、seed 42,兩個 encoder 各掃 {1e-5, 2e-5, 5e-5},只看 validation。BERT 5e-5 重用 AC2 seed 42(判定為等價 run),其餘 5 次新訓練。 +- Drew 確認兩者都用 5e-5,寫入 `configs/curve.yaml`。 + +### 核心發現 / 數據 +| 模型 | 1e-5 | 2e-5 | 5e-5 | +|---|---:|---:|---:| +| BERT(val in-scope / OOS recall) | 90.17% / 45% | 95.73% / 66% | **96.73% / 68%** | +| ModernBERT | 94.77% / 67% | 95.93% / 72% | **97.13% / 75%** | + +- **兩個模型的最佳值都在掃描範圍上限。** 只能說「在 {1e-5, 2e-5, 5e-5} 中 5e-5 最好」,真正的最佳 LR 可能更高。Drew 決定照凍結的協定接受 5e-5,不在看到結果後擴充範圍;兩者停在同一個邊界,比較仍對稱。README 需寫明此限制。 +- LR 太小會學不完:BERT 1e-5 在第 5 個 epoch 的 loss 仍有 0.40(5e-5 為 0.03),val in-scope 低 6.5pp。 +- 單一 seed、val 只有 100 筆 OOS,ModernBERT 與 BERT 的 OOS recall 差 7 筆,不下結論。 +- 每次 run 時間(M4):BERT 約 15 分鐘,ModernBERT 約 21 到 23 分鐘。 + +### Blockers / 遇到的問題 +- (無) + +### Next +- [ ] `make pilot-steps`(k=5、seed 42,S_min ∈ {100, 200, 400})→ Drew 確認 → 寫入 `configs/curve.yaml` +- [ ] `make curve MODEL=bert`、`make curve MODEL=modernbert`、`make oos-ablation` + +### Files / Budget +- `results/pilots/lr.json`、`configs/curve.yaml` +- API 花費:US$0 + +--- + ## 2026-09-23:AC2 通過(BERT 流程正確性檢查) ### 本次工作 / 執行摘要 diff --git a/configs/curve.yaml b/configs/curve.yaml index e9a7aaf..78a2b5e 100644 --- a/configs/curve.yaml +++ b/configs/curve.yaml @@ -9,7 +9,9 @@ min_train_steps: null models: bert: config: configs/bert-base.yaml - learning_rate: null + # results/pilots/lr.json: val in-scope 2705 / 2872 / 2902 of 3000 at 1e-5 / 2e-5 / 5e-5 + learning_rate: 5.0e-5 modernbert: config: configs/modernbert-base.yaml - learning_rate: null + # results/pilots/lr.json: val in-scope 2843 / 2878 / 2914 of 3000 at 1e-5 / 2e-5 / 5e-5 + learning_rate: 5.0e-5 diff --git a/results/pilots/lr.json b/results/pilots/lr.json new file mode 100644 index 0000000..64edf76 --- /dev/null +++ b/results/pilots/lr.json @@ -0,0 +1,313 @@ +{ + "pilot": "learning_rate", + "split": "validation", + "k": 100, + "seed": 42, + "grid": [ + 1e-05, + 2e-05, + 5e-05 + ], + "rule": "per model: max validation in_scope_correct, then oos_correct, then smaller lr", + "points": [ + { + "model": "bert", + "value": 1e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 1e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 883.27 + }, + "validation": { + "in_scope_correct": 2705, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9016666666666666, + "oos_correct": 45, + "oos_n": 100, + "oos_recall_151": 0.45, + "n": 3100 + } + }, + { + "model": "bert", + "value": 2e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 2e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 890.813 + }, + "validation": { + "in_scope_correct": 2872, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9573333333333334, + "oos_correct": 66, + "oos_n": 100, + "oos_recall_151": 0.66, + "n": 3100 + } + }, + { + "model": "bert", + "value": 5e-05, + "run_name": "bert-base-uncased-k100-seed42", + "config": { + "model_name": "google-bert/bert-base-uncased", + "model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 5e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "reused bert-base-uncased-full-seed42 (equivalent run; validation logits from its archive)", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": null, + "train_sample_sha256": null, + "step_plan": null, + "global_step": 2385, + "train_wall_seconds": 886.184 + }, + "validation": { + "in_scope_correct": 2902, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9673333333333334, + "oos_correct": 68, + "oos_n": 100, + "oos_recall_151": 0.68, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 1e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 1e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1405.978 + }, + "validation": { + "in_scope_correct": 2843, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9476666666666667, + "oos_correct": 67, + "oos_n": 100, + "oos_recall_151": 0.67, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 2e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 2e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1251.981 + }, + "validation": { + "in_scope_correct": 2878, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9593333333333334, + "oos_correct": 72, + "oos_n": 100, + "oos_recall_151": 0.72, + "n": 3100 + } + }, + { + "model": "modernbert", + "value": 5e-05, + "run_name": "ModernBERT-base-k100-seed42", + "config": { + "model_name": "answerdotai/ModernBERT-base", + "model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", + "seed": 42, + "per_intent": null, + "k_shot": 100, + "oos_train": null, + "max_length": 64, + "learning_rate": 5e-05, + "weight_decay": 0.01, + "warmup_ratio": 0.1, + "num_train_epochs": 5, + "max_steps": -1, + "min_train_steps": null, + "train_batch_size": 32, + "eval_batch_size": 128, + "device": "auto", + "replace_classifier_head": false, + "eval_per_intent": null + }, + "source": "trained", + "training": { + "train_rows": 15250, + "oos_train_rows": 250, + "k_shot": 100, + "train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553", + "step_plan": { + "epoch_steps": 2385, + "min_train_steps": null, + "planned_steps": 2385, + "decided_by": "epochs", + "max_steps_arg": -1 + }, + "global_step": 2385, + "train_wall_seconds": 1255.688 + }, + "validation": { + "in_scope_correct": 2914, + "in_scope_n": 3000, + "in_scope_accuracy_150": 0.9713333333333334, + "oos_correct": 75, + "oos_n": 100, + "oos_recall_151": 0.75, + "n": 3100 + } + } + ], + "selected": { + "bert": 5e-05, + "modernbert": 5e-05 + }, + "note": "not applied automatically; copy into configs/curve.yaml by hand" +} diff --git a/tests/test_protocol.py b/tests/test_protocol.py index c89f1f0..37212f7 100644 --- a/tests/test_protocol.py +++ b/tests/test_protocol.py @@ -33,12 +33,29 @@ def decided(tmp_path) -> CurveProtocol: return load_protocol(protocol_file(tmp_path, s_min=400, bert="5.0e-5", modern="2.0e-5")) -def test_the_committed_protocol_loads_with_both_pilots_still_open(): +def test_the_committed_protocol_matches_the_committed_pilot_outputs(): + """Values copied by hand into configs/curve.yaml must equal what the pilots selected. + + A pilot whose output is not committed yet must still be null in the protocol. + """ + import json + protocol = load_protocol(ROOT / "configs" / "curve.yaml") - assert protocol.min_train_steps is None - assert protocol.learning_rates == {"bert": None, "modernbert": None} - with pytest.raises(ProtocolError, match="make pilot-lr"): - protocol.curve_configs("bert") + lr_file = ROOT / "results" / "pilots" / "lr.json" + if lr_file.exists(): + selected = json.loads(lr_file.read_text())["selected"] + assert json.loads(lr_file.read_text())["split"] == "validation" + assert protocol.learning_rates == selected + else: + assert protocol.learning_rates == {"bert": None, "modernbert": None} + with pytest.raises(ProtocolError, match="make pilot-lr"): + protocol.curve_configs("bert") + steps_file = ROOT / "results" / "pilots" / "steps.json" + if steps_file.exists(): + assert json.loads(steps_file.read_text())["split"] == "validation" + assert protocol.min_train_steps == json.loads(steps_file.read_text())["selected"] + else: + assert protocol.min_train_steps is None def test_curve_refuses_to_start_until_s_min_is_chosen(tmp_path):