typed-decisions/rl_agent_config.json
847 B · 36 lines · json Raw
1 {
2 "encoder": "answerdotai/ModernBERT-large",
3 "head_layers": 2,
4 "max_len": 1024,
5 "head_max_len": 256,
6 "max_prefixes": 6,
7 "act_costs": {
8 "escalate": 0.5
9 },
10 "cost_wrong_act": 3.0,
11 "amp_dtype": "bf16",
12 "model_name": "laya-typed-decisions",
13 "temperature": [
14 1.0148024559020996,
15 1.0374259948730469,
16 1.0575125217437744
17 ],
18 "temperature_by_options": {
19 "choice:3-5": 1.7601518630981445,
20 "choice:6-10": 1.0000158548355103,
21 "score:3-5": 1.2514300346374512,
22 "noul:2": 1.983399510383606,
23 "choice:11+": 0.10058280825614929,
24 "choice:2": 1.9063563346862793
25 },
26 "training": {
27 "updates": 7313,
28 "epochs_completed": 1,
29 "hours": 1.96,
30 "world_size": 1,
31 "fine_tuned_from_checkpoint": true
32 },
33 "gradient_checkpointing": true,
34 "max_tokens_per_batch": 4096,
35 "fine_tuned": true
36 }