Views
No views yet
0.4.11adapter: lora
2base_model: Xenova/tiny-random-Phi3ForCausalLM
3bf16: auto
4chat_template: llama3
5cosine_min_lr_ratio: 0.1
6data_processes: 16
7dataset_prepared_path: null
8datasets:
9- data_files:
10 - c96b4f7be2a48b05_train_data.json
11 ds_type: json
12 field: prompt_body
13 path: /workspace/input_data/c96b4f7be2a48b05_train_data.json
14 type: completion
15debug: null
16deepspeed: null
17device_map: auto
18do_eval: true
19eval_batch_size: 1
20eval_sample_packing: false
21eval_steps: 25
22evaluation_strategy: steps
23flash_attention: false
24fp16: null
25fsdp: null
26fsdp_config: null
27gradient_accumulation_steps: 32
28gradient_checkpointing: true
29group_by_length: true
30hub_model_id: dada22231/b8d95989-54fa-4b5c-a7d7-31724cf3ca3a
31hub_strategy: checkpoint
32hub_token: null
33hub_username: dada22231
34learning_rate: 0.0001
35load_in_4bit: false
36load_in_8bit: false
37local_rank: null
38logging_steps: 1
39lora_alpha: 64
40lora_dropout: 0.05
41lora_fan_in_fan_out: null
42lora_model_dir: null
43lora_r: 32
44lora_target_linear: true
45lora_target_modules:
46- q_proj
47- v_proj
48lr_scheduler: cosine
49max_grad_norm: 1.0
50max_memory:
51 0: 70GiB
52 1: 70GiB
53 2: 70GiB
54 3: 70GiB
55max_steps: 50
56micro_batch_size: 1
57mlflow_experiment_name: /tmp/c96b4f7be2a48b05_train_data.json
58model_type: AutoModelForCausalLM
59num_epochs: 3
60optim_args:
61 adam_beta1: 0.9
62 adam_beta2: 0.95
63 adam_epsilon: 1e-5
64optimizer: adamw_torch
65output_dir: miner_id_24
66pad_to_sequence_len: true
67repository_id: dada22231/b8d95989-54fa-4b5c-a7d7-31724cf3ca3a
68resume_from_checkpoint: null
69s2_attention: null
70sample_packing: false
71save_steps: 25
72save_strategy: steps
73sequence_len: 2048
74strict: false
75tf32: false
76tokenizer_type: AutoTokenizer
77torch_compile: false
78train_on_inputs: false
79trust_remote_code: true
80val_set_size: 50
81wandb_entity: null
82wandb_mode: online
83wandb_name: b8d95989-54fa-4b5c-a7d7-31724cf3ca3a
84wandb_project: Public_TuningSN
85wandb_runid: b8d95989-54fa-4b5c-a7d7-31724cf3ca3a
86warmup_ratio: 0.04
87weight_decay: 0.01
88xformers_attention: null
89| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.0 | 0.0047 | 1 | nan |
| 0.0 | 0.1187 | 25 | nan |
| 0.0 | 0.2374 | 50 | nan |