Views
No views yet
0.4.11adapter: lora
2base_model: katuni4ka/tiny-random-qwen1.5-moe
3bf16: true
4chat_template: llama3
5cosine_min_lr_ratio: 0.1
6data_processes: 16
7dataset_prepared_path: null
8datasets:
9- data_files:
10 - 3eb43ecc97e35e49_train_data.json
11 ds_type: json
12 format: custom
13 path: /workspace/input_data/3eb43ecc97e35e49_train_data.json
14 type:
15 field_input: selftext
16 field_instruction: title
17 field_output: document
18 format: '{instruction} {input}'
19 no_input_format: '{instruction}'
20 system_format: '{system}'
21 system_prompt: ''
22ddp_bucket_cap_mb: 25
23ddp_find_unused_parameters: false
24debug: null
25deepspeed: null
26device_map: auto
27do_eval: true
28eval_batch_size: 1
29eval_sample_packing: false
30eval_steps: 25
31evaluation_strategy: steps
32flash_attention: true
33fp16: null
34fsdp: null
35fsdp_config: null
36gradient_accumulation_steps: 8
37gradient_checkpointing: true
38gradient_clipping: 1.0
39group_by_length: true
40hub_model_id: dada22231/7af86c7a-d8b4-4095-a5be-2cdbf598f16c
41hub_strategy: checkpoint
42hub_token: null
43hub_username: dada22231
44learning_rate: 0.0001
45local_rank: null
46logging_steps: 1
47lora_alpha: 64
48lora_dropout: 0.05
49lora_fan_in_fan_out: null
50lora_model_dir: null
51lora_r: 32
52lora_target_linear: true
53lora_target_modules:
54- q_proj
55- v_proj
56lr_scheduler: cosine
57max_grad_norm: 1.0
58max_memory:
59 '0': 75GiB
60 '1': 75GiB
61 '2': 75GiB
62 '3': 75GiB
63max_steps: 50
64micro_batch_size: 4
65mlflow_experiment_name: null
66model_type: AutoModelForCausalLM
67num_epochs: 3
68optim_args:
69 adam_beta1: 0.9
70 adam_beta2: 0.95
71 adam_epsilon: 1e-5
72optimizer: adamw_torch
73output_dir: miner_id_24
74pad_to_sequence_len: true
75repository_id: dada22231/7af86c7a-d8b4-4095-a5be-2cdbf598f16c
76resume_from_checkpoint: null
77s2_attention: null
78sample_packing: false
79save_steps: 25
80save_strategy: steps
81sequence_len: 2048
82strict: false
83tf32: true
84tokenizer_type: AutoTokenizer
85torch_compile: false
86train_on_inputs: false
87trust_remote_code: true
88val_set_size: 50
89wandb_entity: null
90wandb_mode: online
91wandb_name: 7af86c7a-d8b4-4095-a5be-2cdbf598f16c
92wandb_project: Public_TuningSN
93wandb_runid: 7af86c7a-d8b4-4095-a5be-2cdbf598f16c
94warmup_ratio: 0.03
95weight_decay: 0.01
96xformers_attention: null
97| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 0.0 | 0.0006 | 1 | nan |
| 0.0 | 0.0140 | 25 | nan |
| 0.0 | 0.0280 | 50 | nan |