# DPO policy-update plugin on the same SAMPG loop
# Docs: https://seben.robotsmali.org/docs/hyperparams/
project_name: sebeni-bam-dpo
algorithm: dpo
# working_dir: ./runs/bam-dpo-001

model:
  model_name: HuggingFaceTB/SmolLM2-135M
  use_peft: true
  lora_r: 16
  lora_alpha: 32
  lora_dropout: 0.1

data:
  default_lang: bam
  languages: [bam]
  scheme: preference

dpo:
  learning_rate: 5.0e-6
  per_device_train_batch_size: 2
  gradient_accumulation_steps: 8
  max_steps: 10
  num_train_epochs: 1.0
  beta: 0.1
  loss_type: sigmoid
  warmup_ratio: 0.0
  weight_decay: 0.0
  lr_scheduler_type: cosine
  seed: 42

distillation:
  enabled: true
  backend: algorithmic
  tau: 0.5

safety:
  enabled: true
