Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 30 additions & 0 deletions DEVLOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,36 @@

---

## 2026-09-23 (晚):learning rate pilot(validation only)

### 本次工作 / 執行摘要
- `make pilot-lr`:k=100、seed 42,兩個 encoder 各掃 {1e-5, 2e-5, 5e-5},只看 validation。BERT 5e-5 重用 AC2 seed 42(判定為等價 run),其餘 5 次新訓練。
- Drew 確認兩者都用 5e-5,寫入 `configs/curve.yaml`。

### 核心發現 / 數據
| 模型 | 1e-5 | 2e-5 | 5e-5 |
|---|---:|---:|---:|
| BERT(val in-scope / OOS recall) | 90.17% / 45% | 95.73% / 66% | **96.73% / 68%** |
| ModernBERT | 94.77% / 67% | 95.93% / 72% | **97.13% / 75%** |

- **兩個模型的最佳值都在掃描範圍上限。** 只能說「在 {1e-5, 2e-5, 5e-5} 中 5e-5 最好」,真正的最佳 LR 可能更高。Drew 決定照凍結的協定接受 5e-5,不在看到結果後擴充範圍;兩者停在同一個邊界,比較仍對稱。README 需寫明此限制。
- LR 太小會學不完:BERT 1e-5 在第 5 個 epoch 的 loss 仍有 0.40(5e-5 為 0.03),val in-scope 低 6.5pp。
- 單一 seed、val 只有 100 筆 OOS,ModernBERT 與 BERT 的 OOS recall 差 7 筆,不下結論。
- 每次 run 時間(M4):BERT 約 15 分鐘,ModernBERT 約 21 到 23 分鐘。

### Blockers / 遇到的問題
- (無)

### Next
- [ ] `make pilot-steps`(k=5、seed 42,S_min ∈ {100, 200, 400})→ Drew 確認 → 寫入 `configs/curve.yaml`
- [ ] `make curve MODEL=bert`、`make curve MODEL=modernbert`、`make oos-ablation`

### Files / Budget
- `results/pilots/lr.json`、`configs/curve.yaml`
- API 花費:US$0

---

## 2026-09-23:AC2 通過(BERT 流程正確性檢查)

### 本次工作 / 執行摘要
Expand Down
6 changes: 4 additions & 2 deletions configs/curve.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,9 @@ min_train_steps: null
models:
bert:
config: configs/bert-base.yaml
learning_rate: null
# results/pilots/lr.json: val in-scope 2705 / 2872 / 2902 of 3000 at 1e-5 / 2e-5 / 5e-5
learning_rate: 5.0e-5
modernbert:
config: configs/modernbert-base.yaml
learning_rate: null
# results/pilots/lr.json: val in-scope 2843 / 2878 / 2914 of 3000 at 1e-5 / 2e-5 / 5e-5
learning_rate: 5.0e-5
313 changes: 313 additions & 0 deletions results/pilots/lr.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,313 @@
{
"pilot": "learning_rate",
"split": "validation",
"k": 100,
"seed": 42,
"grid": [
1e-05,
2e-05,
5e-05
],
"rule": "per model: max validation in_scope_correct, then oos_correct, then smaller lr",
"points": [
{
"model": "bert",
"value": 1e-05,
"run_name": "bert-base-uncased-k100-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 1e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": 100,
"train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553",
"step_plan": {
"epoch_steps": 2385,
"min_train_steps": null,
"planned_steps": 2385,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 2385,
"train_wall_seconds": 883.27
},
"validation": {
"in_scope_correct": 2705,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9016666666666666,
"oos_correct": 45,
"oos_n": 100,
"oos_recall_151": 0.45,
"n": 3100
}
},
{
"model": "bert",
"value": 2e-05,
"run_name": "bert-base-uncased-k100-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 2e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": 100,
"train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553",
"step_plan": {
"epoch_steps": 2385,
"min_train_steps": null,
"planned_steps": 2385,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 2385,
"train_wall_seconds": 890.813
},
"validation": {
"in_scope_correct": 2872,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9573333333333334,
"oos_correct": 66,
"oos_n": 100,
"oos_recall_151": 0.66,
"n": 3100
}
},
{
"model": "bert",
"value": 5e-05,
"run_name": "bert-base-uncased-k100-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "reused bert-base-uncased-full-seed42 (equivalent run; validation logits from its archive)",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": null,
"train_sample_sha256": null,
"step_plan": null,
"global_step": 2385,
"train_wall_seconds": 886.184
},
"validation": {
"in_scope_correct": 2902,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9673333333333334,
"oos_correct": 68,
"oos_n": 100,
"oos_recall_151": 0.68,
"n": 3100
}
},
{
"model": "modernbert",
"value": 1e-05,
"run_name": "ModernBERT-base-k100-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 1e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": 100,
"train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553",
"step_plan": {
"epoch_steps": 2385,
"min_train_steps": null,
"planned_steps": 2385,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 2385,
"train_wall_seconds": 1405.978
},
"validation": {
"in_scope_correct": 2843,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9476666666666667,
"oos_correct": 67,
"oos_n": 100,
"oos_recall_151": 0.67,
"n": 3100
}
},
{
"model": "modernbert",
"value": 2e-05,
"run_name": "ModernBERT-base-k100-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 2e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": 100,
"train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553",
"step_plan": {
"epoch_steps": 2385,
"min_train_steps": null,
"planned_steps": 2385,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 2385,
"train_wall_seconds": 1251.981
},
"validation": {
"in_scope_correct": 2878,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9593333333333334,
"oos_correct": 72,
"oos_n": 100,
"oos_recall_151": 0.72,
"n": 3100
}
},
{
"model": "modernbert",
"value": 5e-05,
"run_name": "ModernBERT-base-k100-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 100,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": null,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 15250,
"oos_train_rows": 250,
"k_shot": 100,
"train_sample_sha256": "7a3ebd2cd2f985a26e815c229cc54add6763f783bee572c5e7948636c66c7553",
"step_plan": {
"epoch_steps": 2385,
"min_train_steps": null,
"planned_steps": 2385,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 2385,
"train_wall_seconds": 1255.688
},
"validation": {
"in_scope_correct": 2914,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.9713333333333334,
"oos_correct": 75,
"oos_n": 100,
"oos_recall_151": 0.75,
"n": 3100
}
}
],
"selected": {
"bert": 5e-05,
"modernbert": 5e-05
},
"note": "not applied automatically; copy into configs/curve.yaml by hand"
}
27 changes: 22 additions & 5 deletions tests/test_protocol.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,12 +33,29 @@ def decided(tmp_path) -> CurveProtocol:
return load_protocol(protocol_file(tmp_path, s_min=400, bert="5.0e-5", modern="2.0e-5"))


def test_the_committed_protocol_loads_with_both_pilots_still_open():
def test_the_committed_protocol_matches_the_committed_pilot_outputs():
"""Values copied by hand into configs/curve.yaml must equal what the pilots selected.

A pilot whose output is not committed yet must still be null in the protocol.
"""
import json

protocol = load_protocol(ROOT / "configs" / "curve.yaml")
assert protocol.min_train_steps is None
assert protocol.learning_rates == {"bert": None, "modernbert": None}
with pytest.raises(ProtocolError, match="make pilot-lr"):
protocol.curve_configs("bert")
lr_file = ROOT / "results" / "pilots" / "lr.json"
if lr_file.exists():
selected = json.loads(lr_file.read_text())["selected"]
assert json.loads(lr_file.read_text())["split"] == "validation"
assert protocol.learning_rates == selected
else:
assert protocol.learning_rates == {"bert": None, "modernbert": None}
with pytest.raises(ProtocolError, match="make pilot-lr"):
protocol.curve_configs("bert")
steps_file = ROOT / "results" / "pilots" / "steps.json"
if steps_file.exists():
assert json.loads(steps_file.read_text())["split"] == "validation"
assert protocol.min_train_steps == json.loads(steps_file.read_text())["selected"]
else:
assert protocol.min_train_steps is None


def test_curve_refuses_to_start_until_s_min_is_chosen(tmp_path):
Expand Down
Loading