Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 30 additions & 0 deletions DEVLOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,36 @@

---

## 2026-09-23 (晚 2):S_min pilot(validation only),協定凍結

### 本次工作 / 執行摘要
- `make pilot-steps`:k=5(每個 intent 5 筆、OOS 13 筆)、seed 42、lr 5e-5,兩個 encoder 各試 S_min ∈ {100, 200, 400},只看 validation。
- 規則:兩個模型 val in-scope 答對數加總最高者,平手取較小 S_min。選出 400,Drew 確認,寫入 `configs/curve.yaml`。
- **超參數協定至此全部凍結**:lr 5e-5(兩個模型)、S_min 400、5 epochs、batch 32、max_length 64。正式曲線開跑後不再依中途結果修改任何一項。

### 核心發現 / 數據
| S_min(實際步數) | BERT val in-scope / OOS | ModernBERT val in-scope / OOS |
|---|---:|---:|
| 100(120) | 9.73% / 10% | 56.20% / 16% |
| 200(200) | 36.73% / 45% | 63.43% / 21% |
| 400(400) | **75.87% / 47%** | **67.20% / 27%** |

- **又落在範圍上限,而且小 k 尚未訓練飽和**:BERT 從 200 到 400 步仍大幅上升(37% → 76%)。所以 k=1、5、10 的曲線點量的是「400 步預算下」的表現,不是模型極限;README 必須寫明。k ≥ 25 時 5 個 epoch 已超過 400 步,S_min 不生效。
- **學習速度 vs 最終表現**:步數很少時 ModernBERT 學得快很多(120 步 56% vs 10%),到 400 步 BERT 反超。「誰比較有效率」取決於訓練預算。單一 seed,不下結論,等正式曲線。
- `S_min=100` 在 k=5 實際是 120 步,因為 5 個 epoch = 120 步 > 100,符合 max(S_min, epoch 步數) 的定義。

### Blockers / 遇到的問題
- (無)

### Next
- [ ] `make curve MODEL=bert`、`make curve MODEL=modernbert`、`make oos-ablation`(約 5 小時)

### Files / Budget
- `results/pilots/steps.json`、`configs/curve.yaml`
- API 花費:US$0

---

## 2026-09-23 (晚):learning rate pilot(validation only)

### 本次工作 / 執行摘要
Expand Down
3 changes: 2 additions & 1 deletion configs/curve.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,8 @@
# null means that pilot has not been run; `make curve` refuses to start.
# Allowed values: learning_rate in {1.0e-5, 2.0e-5, 5.0e-5} (write the
# ".0": YAML reads 5e-5 as a string), min_train_steps in {100, 200, 400}.
min_train_steps: null
# results/pilots/steps.json: sum of val in-scope correct (both models, k=5) 1978 / 3005 / 4292 at 100 / 200 / 400
min_train_steps: 400
models:
bert:
config: configs/bert-base.yaml
Expand Down
316 changes: 316 additions & 0 deletions results/pilots/steps.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,316 @@
{
"pilot": "min_train_steps",
"split": "validation",
"k": 5,
"seed": 42,
"grid": [
100,
200,
400
],
"rule": "sum of both models' validation in_scope_correct, then smaller S_min",
"points": [
{
"model": "bert",
"value": 100,
"run_name": "bert-base-uncased-k5-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 100,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 100,
"planned_steps": 120,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 120,
"train_wall_seconds": 45.059
},
"validation": {
"in_scope_correct": 292,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.09733333333333333,
"oos_correct": 10,
"oos_n": 100,
"oos_recall_151": 0.1,
"n": 3100
}
},
{
"model": "modernbert",
"value": 100,
"run_name": "ModernBERT-base-k5-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 100,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 100,
"planned_steps": 120,
"decided_by": "epochs",
"max_steps_arg": -1
},
"global_step": 120,
"train_wall_seconds": 66.097
},
"validation": {
"in_scope_correct": 1686,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.562,
"oos_correct": 16,
"oos_n": 100,
"oos_recall_151": 0.16,
"n": 3100
}
},
{
"model": "bert",
"value": 200,
"run_name": "bert-base-uncased-k5-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 200,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 200,
"planned_steps": 200,
"decided_by": "min_train_steps",
"max_steps_arg": 200
},
"global_step": 200,
"train_wall_seconds": 71.255
},
"validation": {
"in_scope_correct": 1102,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.36733333333333335,
"oos_correct": 45,
"oos_n": 100,
"oos_recall_151": 0.45,
"n": 3100
}
},
{
"model": "modernbert",
"value": 200,
"run_name": "ModernBERT-base-k5-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 200,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 200,
"planned_steps": 200,
"decided_by": "min_train_steps",
"max_steps_arg": 200
},
"global_step": 200,
"train_wall_seconds": 104.118
},
"validation": {
"in_scope_correct": 1903,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.6343333333333333,
"oos_correct": 21,
"oos_n": 100,
"oos_recall_151": 0.21,
"n": 3100
}
},
{
"model": "bert",
"value": 400,
"run_name": "bert-base-uncased-k5-seed42",
"config": {
"model_name": "google-bert/bert-base-uncased",
"model_revision": "86b5e0934494bd15c9632b12f734a8a67f723594",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 400,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 400,
"planned_steps": 400,
"decided_by": "min_train_steps",
"max_steps_arg": 400
},
"global_step": 400,
"train_wall_seconds": 146.499
},
"validation": {
"in_scope_correct": 2276,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.7586666666666667,
"oos_correct": 47,
"oos_n": 100,
"oos_recall_151": 0.47,
"n": 3100
}
},
{
"model": "modernbert",
"value": 400,
"run_name": "ModernBERT-base-k5-seed42",
"config": {
"model_name": "answerdotai/ModernBERT-base",
"model_revision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"seed": 42,
"per_intent": null,
"k_shot": 5,
"oos_train": null,
"max_length": 64,
"learning_rate": 5e-05,
"weight_decay": 0.01,
"warmup_ratio": 0.1,
"num_train_epochs": 5,
"max_steps": -1,
"min_train_steps": 400,
"train_batch_size": 32,
"eval_batch_size": 128,
"device": "auto",
"replace_classifier_head": false,
"eval_per_intent": null
},
"source": "trained",
"training": {
"train_rows": 763,
"oos_train_rows": 13,
"k_shot": 5,
"train_sample_sha256": "bfabf08971d3a16a502e25b2d5a7f1ce204504a526d5b3cb34fec9346dd8caa7",
"step_plan": {
"epoch_steps": 120,
"min_train_steps": 400,
"planned_steps": 400,
"decided_by": "min_train_steps",
"max_steps_arg": 400
},
"global_step": 400,
"train_wall_seconds": 215.494
},
"validation": {
"in_scope_correct": 2016,
"in_scope_n": 3000,
"in_scope_accuracy_150": 0.672,
"oos_correct": 27,
"oos_n": 100,
"oos_recall_151": 0.27,
"n": 3100
}
}
],
"selected": 400,
"note": "not applied automatically; copy into configs/curve.yaml by hand"
}
Loading