Views
No views yet
0.5.21adapter: lora
2base_model: echarlaix/tiny-random-mistral
3bf16: auto
4chat_template: llama3
5datasets:
6- data_files:
7 - 42fa4b965cededb3_train_data.json
8 ds_type: json
9 format: custom
10 path: /runs/taopanda-4_d0616b9d-99ec-4e9e-86fa-82059ce33170/42fa4b965cededb3_train_data.json
11 preprocessing:
12 - shuffle: true
13 type:
14 field: null
15 field_input: doc
16 field_instruction: original_text
17 field_output: edited_summary
18 field_system: null
19 format: null
20 no_input_format: null
21 system_format: '{system}'
22 system_prompt: ''
23debug: null
24deepspeed: null
25device_map: auto
26early_stopping_patience: 4
27eval_batch_size: 4
28eval_max_new_tokens: 128
29eval_steps: 25
30eval_strategy: steps
31fp16: null
32gradient_accumulation_steps: 4
33gradient_checkpointing: true
34group_by_length: true
35hub_model_id: taopanda-4/957a944b-e075-4b04-8c43-d11fbfdd15aa
36hub_strategy: every_save
37learning_rate: 0.00010312140429884754
38load_best_model_at_end: true
39load_in_4bit: false
40load_in_8bit: false
41local_rank: null
42logging_steps: 1
43lora_alpha: 64
44lora_dropout: 0.05
45lora_fan_in_fan_out: true
46lora_model_dir: null
47lora_r: 32
48lora_target_linear: true
49lr_scheduler: cosine
50max_grad_norm: 1.0
51max_steps: 750
52micro_batch_size: 16
53model_type: AutoModelForCausalLM
54num_epochs: 32
55optimizer: paged_adamw_32bit
56output_dir: ./outputs/lora-out/taopanda-4_d0616b9d-99ec-4e9e-86fa-82059ce33170
57pad_to_sequence_len: true
58resume_from_checkpoint: null
59s2_attention: null
60save_steps: 25
61save_total_limit: 5
62seed: 40883
63sequence_len: 512
64special_tokens:
65 pad_token: </s>
66strict: false
67tf32: true
68tokenizer_type: AutoTokenizer
69train_on_inputs: false
70trust_remote_code: true
71val_set_size: 0.05
72wandb_entity: fatcat87-taopanda
73wandb_mode: online
74wandb_name: taopanda-4_d0616b9d-99ec-4e9e-86fa-82059ce33170
75wandb_project: subnet56
76wandb_runid: taopanda-4_d0616b9d-99ec-4e9e-86fa-82059ce33170
77warmup_ratio: 0.1
78weight_decay: 0.1
79xformers_attention: null
80| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 10.3814 | 0.0667 | 1 | 10.3775 |
| 10.3723 | 1.6667 | 25 | 10.3749 |
| 10.3609 | 3.3333 | 50 | 10.3615 |
| 10.3307 | 5.0 | 75 | 10.3299 |
| 10.3229 | 6.6667 | 100 | 10.3222 |
| 10.3171 | 8.3333 | 125 | 10.3189 |
| 10.3157 | 10.0 | 150 | 10.3158 |
| 10.3147 | 11.6667 | 175 | 10.3147 |
| 10.3109 | 13.3333 | 200 | 10.3143 |
| 10.3121 | 15.0 | 225 | 10.3137 |
| 10.3133 | 16.6667 | 250 | 10.3133 |
| 10.3113 | 18.3333 | 275 | 10.3129 |
| 10.3142 | 20.0 | 300 | 10.3124 |
| 10.3117 | 21.6667 | 325 | 10.3119 |
| 10.3107 | 23.3333 | 350 | 10.3118 |
| 10.309 | 25.0 | 375 | 10.3113 |
| 10.3134 | 26.6667 | 400 | 10.3110 |
| 10.3083 | 28.3333 | 425 | 10.3108 |
| 10.3103 | 30.0 | 450 | 10.3105 |
| 10.312 | 31.6667 | 475 | 10.3104 |