adapter_config.json 945 B

12345678910111213141516171819202122232425262728293031323334353637383940
  1. {
  2. "adapter_path": "adapters",
  3. "batch_size": 1,
  4. "config": null,
  5. "data": "data",
  6. "fine_tune_type": "lora",
  7. "grad_accumulation_steps": 1,
  8. "grad_checkpoint": true,
  9. "iters": 2026,
  10. "learning_rate": 1e-05,
  11. "lora_parameters": {
  12. "rank": 8,
  13. "dropout": 0.0,
  14. "scale": 20.0
  15. },
  16. "lr_schedule": null,
  17. "mask_prompt": false,
  18. "max_seq_length": 4096,
  19. "model": "mlx-community/Meta-Llama-3.1-8B-Instruct-4bit",
  20. "num_layers": 16,
  21. "optimizer": "adam",
  22. "optimizer_config": {
  23. "adam": {},
  24. "adamw": {},
  25. "muon": {},
  26. "sgd": {},
  27. "adafactor": {}
  28. },
  29. "project_name": null,
  30. "report_to": null,
  31. "resume_adapter_file": "adapters/adapters.safetensors",
  32. "save_every": 100,
  33. "seed": 0,
  34. "steps_per_eval": 100,
  35. "steps_per_report": 10,
  36. "test": false,
  37. "test_batches": 500,
  38. "train": true,
  39. "val_batches": 25
  40. }