Views
No views yet
latest1064(0.6B)/(40M-20M)/(reason-predict,r=0.03125,k=8)/(lr=0.00015,epochs=2,replay=0.5)/(seed=0)/(val-r=1.0)1trainee:
2 short_name: 0.6B
3 model_id: Qwen/Qwen3-0.6B
4 dtype: bfloat16
5train:
6 data:
7 documents:
8 short_name: 40M-20M
9 dataset_id: JackHsieh/statML-arxiv-40M-20M
10 split: train
11 n_samples: null
12 exact_document_length: 4096
13 chunking:
14 rule: r=0.03125,k=8
15 chunk_size: 8
16 path: outputs/prestar/chunking/r=0.03125,k=8/40M-20M/train.parquet
17 thoughts:
18 short_name: reason-predict
19 dataset_id: JackHsieh/32B-reason-predict.rule-r-0.03125-k-8.L-512.statml-arxiv
20 split: train
21 n_samples: null
22 replay:
23 short_name: full-replay
24 dataset_id: JackHsieh/dclm-replay.seq-4096.tokens-32B
25 split: train
26 n_samples: null
27 exact_document_length: 4096
28 serving:
29 use_thoughts: true
30 replay_proportion: 0.5
31 thoughts_per_step: 1
32 fixed_main_order: true
33 fixed_replay_order: true
34 enable: true
35 docs_per_batch: 32
36 total_epochs: 2
37 total_steps: null
38 grad_clip_norm: 1.0
39 seed: 0
40 early_stopping:
41 enable: false
42 mode: null
43 metric: null
44 patience_evals: null
45 min_delta: null
46 loss: {}
47 execution:
48 thoughtless_pass_doc_microbatch: 4
49 thoughtful_pass_thought_microbatch: 2
50 save_incremental_per_epoch: false
51 save_per_token_log_probs: false
52 save_chunk_token_snippets: false
53 optimizer:
54 style: adamw
55 lr: 0.00015
56 weight_decay: 0.01
57 beta1: 0.9
58 beta2: 0.95
59 eps: 1.0e-08
60 lr_scheduler:
61 style: cosine
62 total_steps: null
63 min_lr_ratio: 0.0
64 warmup_steps: null
65 warmup_ratio: 0.05
66 stable_steps: null
67 decay_steps: null
68eval:
69 data:
70 documents:
71 short_name: 40M-20M
72 dataset_id: JackHsieh/statML-arxiv-40M-20M
73 split: test
74 n_samples: null
75 exact_document_length: 4096
76 chunking:
77 rule: r=1.0,k=8
78 chunk_size: 8
79 path: outputs/prestar/chunking/r=1.0,k=8/40M-20M/test.parquet
80 thoughts:
81 short_name: 32B-predict-k=8,L=16
82 dataset_id: JackHsieh/32B-predict.rule-r-1.0-k-8.L-16.statml-arxiv
83 split: test
84 n_samples: null
85 serving:
86 use_thoughts: true
87 thoughts_per_pass: 1
88 enable: true
89 eval_every_n_steps: null
90 eval_every_n_epochs: -1
91 eval_on_first_step: false
92 execution:
93 docs_per_batch: null
94 thoughtless_pass_doc_microbatch: 32
95 thoughtful_pass_thought_microbatch: 8
96 incremental_save_every_n_batches: 0
97 save_per_token_log_probs: true
98 save_chunk_token_snippets: false
99checkpointing:
100 resume:
101 from_latest: true
102 from_path: null
103 same_wandb_run: true
104 best:
105 enable: false
106 metric: val-dynamics/val_loss
107 direction: min
108 max_keep: 1
109 includes_resume_state: false
110 hub:
111 enable: false
112 repo_id: null
113 private: false
114 model_card_template_path: null
115 latest:
116 enable: true
117 max_keep: 1
118 includes_resume_state: true
119 cadence:
120 unit: epochs
121 every: 0.25
122 hub:
123 enable: true
124 repo_id: null
125 private: false
126 model_card_template_path: null
127logging:
128 flush_every_n_steps: 1
129 train_metrics_reduction: aggregate
130 file:
131 per_batch_metrics_jsonl: true
132 wandb:
133 enable: true
134 project: prestar
135 entity: latent-thoughts
136 group: thoughtful-CPT-32B-predict-fp32
137 name: null
138 tags: []
139 n_examples_logged: 10
140 notes_template_path: configs/prestar/wandb_templates/default.j2
141 debug:
142 enabled: false
143 cross_rank_checks: false
144hardware:
145 num_gpus: 8
146 cpus_per_gpu: 8
147 gpu_type: null
148 peak_bf16_tflops: null
149system:
150 vllm: null
151 trainee:
152 attention_implementation: flash_attention_2
153 enable_gradient_checkpointing: false
154 enable_activation_offload: false
155 fsdp:
156 param_offload: false
157 optimizer_offload: false
158 fsdp_size: -1
159 strategy: no_shard
160 reshard_after_forward: false
161 master_weights_fp32: true
162run_name: (0.6B)/(40M-20M)/(reason-predict,r=0.03125,k=8)/(lr=0.00015,epochs=2,replay=0.5)/(seed=0)/(val-r=1.0)
163outputs_root: outputs/prestar