library_name: peft
license: gemma
base_model: mlabonne/gemma-3-27b-it-abliterated
tags:
- generated_from_trainer
datasets: - data/bdsm_train.parquet
model-index: - name: outputs/libry_bdsm
results: []
See axolotl config
axolotl version: 0.9.2
base_model: mlabonne/gemma-3-27b-it-abliterated
base_model_config: mlabonne/gemma-3-27b-it-abliterated
model_type: Gemma3ForConditionalGeneration
tokenizer_type: AutoTokenizer
load_in_4bit: true
bnb_4bit_compute_dtype: bfloat16
bnb_4bit_quant_type: nf4
bnb_4bit_use_double_quant: true
strict: false
datasets:
- path: data/bdsm_train.parquet
type: chat_template
field_messages: messages
split: train
val_set_size: 0.0
test_datasets:
- path: data/bdsm_test.parquet # <-- ADD THIS: Path to your validation dataset
type: chat_template # <-- Type should match your training dataset
field_messages: messages
split: train
chat_template: tokenizer_default
#val_set_size: 0.05 # This will now be ignored if you have a separate 'validation' split defined above
save_safetensors: true
output_dir: ./outputs/libry_bdsm # <-- OPTIONAL: Consider renaming for clarity with the new model
sequence_len: 8192 # Keep this for now, but be mindful of memory for 27B without Flash Attention
wandb_project: nemo-finetune
wandb_watch: all
wandb_name: LibryBDSM
# Batch & Training Config
micro_batch_size: 2
gradient_accumulation_steps: 5
num_epochs: 5
# Optimizer
optimizer: paged_adamw_32bit # DeepSpeed will manage the optimizer with its settings
adam_beta2: 0.95
adam_epsilon: 0.00001
lr_scheduler: cosine
learning_rate: 1e-4
weight_decay: 0.1
warmup_ratio: 0.03
# Memory & Precision
train_on_inputs: false
train_on_eos: turn
bf16: true
fp16: false
tf32: false
sample_packing: true
pad_to_sequence_len: true # Recommended with sample_packing
group_by_length: false
roles_to_train: ["assistant"]
gradient_checkpointing: true
#gradient_checkpointing_kwargs:
# use_reentrant: false
#resume_from_checkpoint: ./outputs/libry_run2/checkpoint-379 # <-- CHANGE THIS: Set to 'true' to continue from the last checkpoint
#auto_resume_from_checkpoint: false
logging_steps: 1
flash_attention: false # DISABLED Flash Attention
flash_attention_v2: false # DISABLED Flash Attention V2
attn_implementation: eager # Set attention implementation back to eager
# LoRA Settings
adapter: qlora
lora_r: 32
lora_alpha: 16
lora_dropout: 0.05
lora_target_modules: ["q_proj", "k_proj", "v_proj", "o_proj", "gate_proj", "up_proj", "down_proj"] # Check these for Gemma
lora_bias: none
lora_task_type: CAUSAL_LM
# DeepSpeed Configuration (Optimized)
deepspeed: deepspeed_configs/zero2_cpu_offload.json # Point to your DeepSpeed config file
# Hugging Face Upload (Disabled)
push_to_hub: false
evals_per_epoch: 2
eval_batch_size: 2 # Consistent with micro_batch_size
eval_sample_packing: true # <-- OPTIONAL: Changed back to true for efficiency and consistency
eval_table_size: 0
special_tokens:
eos_token: "</s>"
bos_token: "<s>"
unk_token: "<unk>"
pad_token: "</s>"
outputs/libry_bdsm
This model is a fine-tuned version of mlabonne/gemma-3-27b-it-abliterated on the data/bdsm_train.parquet dataset.
It achieves the following results on the evaluation set:
- Loss: 0.8456
Model description
More information needed
Intended uses & limitations
More information needed
Training and evaluation data
More information needed
Training procedure
Training hyperparameters
The following hyperparameters were used during training:
- learning_rate: 0.0001
- train_batch_size: 2
- eval_batch_size: 2
- seed: 42
- distributed_type: multi-GPU
- gradient_accumulation_steps: 5
- total_train_batch_size: 10
- optimizer: Use paged_adamw_32bit with betas=(0.9,0.95) and epsilon=1e-05 and optimizer_args=No additional optimizer arguments
- lr_scheduler_type: cosine
- lr_scheduler_warmup_steps: 27
- num_epochs: 5.0
Training results
| Training Loss | Epoch | Step | Validation Loss |
|---|---|---|---|
| 4.8698 | 0.0054 | 1 | 4.7277 |
| 1.273 | 0.4989 | 92 | 1.2436 |
| 0.9912 | 0.9978 | 184 | 1.0105 |
| 0.9568 | 1.4935 | 276 | 0.9461 |
| 0.9143 | 1.9924 | 368 | 0.9112 |
| 0.8872 | 2.4881 | 460 | 0.8921 |
| 0.8644 | 2.9870 | 552 | 0.8765 |
| 0.8649 | 3.4826 | 644 | 0.8669 |
| 0.8555 | 3.9816 | 736 | 0.8579 |
| 0.8393 | 4.4772 | 828 | 0.8533 |
| 0.8189 | 4.9761 | 920 | 0.8456 |
Framework versions
- PEFT 0.15.2
- Transformers 4.51.3
- Pytorch 2.6.0+cu124
- Datasets 3.5.1
- Tokenizers 0.21.1