Views
No views yet
0.4.11adapter: lora
2base_model: TinyLlama/TinyLlama_v1.1
3bf16: auto
4chat_template: llama3
5cosine_min_lr_ratio: 0.1
6data_processes: 16
7dataset_prepared_path: null
8datasets:
9- data_files:
10 - 9c0aa893642f5c99_train_data.json
11 ds_type: json
12 field: content
13 path: /workspace/input_data/9c0aa893642f5c99_train_data.json
14 type: completion
15debug: null
16deepspeed: null
17device_map: auto
18do_eval: true
19early_stopping_patience: 1
20eval_batch_size: 1
21eval_sample_packing: false
22eval_steps: 25
23evaluation_strategy: steps
24flash_attention: false
25fp16: null
26fsdp: null
27fsdp_config: null
28gradient_accumulation_steps: 32
29gradient_checkpointing: true
30group_by_length: true
31hub_model_id: dada22231/744640a0-3eb5-44b9-ba5b-3af8cb509915
32hub_strategy: checkpoint
33hub_token: null
34hub_username: dada22231
35learning_rate: 0.0001
36load_in_4bit: false
37load_in_8bit: false
38local_rank: null
39logging_steps: 1
40lora_alpha: 64
41lora_dropout: 0.05
42lora_fan_in_fan_out: null
43lora_model_dir: null
44lora_r: 32
45lora_target_linear: true
46lora_target_modules:
47- q_proj
48- v_proj
49lr_scheduler: cosine
50max_grad_norm: 1.0
51max_memory:
52 0: 70GiB
53 1: 70GiB
54 2: 70GiB
55 3: 70GiB
56max_steps: 95
57micro_batch_size: 1
58mlflow_experiment_name: /tmp/9c0aa893642f5c99_train_data.json
59model_type: AutoModelForCausalLM
60num_epochs: 3
61optim_args:
62 adam_beta1: 0.9
63 adam_beta2: 0.95
64 adam_epsilon: 1e-5
65optimizer: adamw_torch
66output_dir: miner_id_24
67pad_to_sequence_len: true
68repository_id: dada22231/744640a0-3eb5-44b9-ba5b-3af8cb509915
69resume_from_checkpoint: null
70s2_attention: null
71sample_packing: false
72save_steps: 25
73save_strategy: steps
74sequence_len: 2048
75strict: false
76tf32: false
77tokenizer_type: AutoTokenizer
78torch_compile: false
79train_on_inputs: false
80trust_remote_code: true
81val_set_size: 50
82wandb_entity: null
83wandb_mode: online
84wandb_name: 744640a0-3eb5-44b9-ba5b-3af8cb509915
85wandb_project: Public_TuningSN
86wandb_runid: 744640a0-3eb5-44b9-ba5b-3af8cb509915
87warmup_ratio: 0.04
88weight_decay: 0.01
89xformers_attention: null
90| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 1.0835 | 0.0050 | 1 | 1.1378 |
| 0.9751 | 0.1244 | 25 | 0.8718 |
| 0.8583 | 0.2488 | 50 | 0.8091 |
| 0.8392 | 0.3732 | 75 | 0.7902 |