feat: QLoRA fine-tune v1 on M5 + fused model (M2)
This commit is contained in:
16
configs/lora.yaml
Normal file
16
configs/lora.yaml
Normal file
@@ -0,0 +1,16 @@
|
||||
model: mlx-community/Qwen2.5-7B-Instruct-4bit
|
||||
train: true
|
||||
data: data/train
|
||||
adapter_path: adapters/buddhagpt-v1
|
||||
batch_size: 1
|
||||
grad_accumulation_steps: 8
|
||||
iters: 1200
|
||||
learning_rate: 1e-5
|
||||
num_layers: 16
|
||||
lora_parameters:
|
||||
rank: 16
|
||||
scale: 20.0
|
||||
dropout: 0.05
|
||||
max_seq_length: 1024
|
||||
steps_per_eval: 200
|
||||
save_every: 200
|
||||
Reference in New Issue
Block a user