Mistral-Nemo-Gutenberg-Encore-12B
7
11
license:apache-2.0
by
nbeerbower
Language Model
OTHER
12B params
New
7 downloads
Early-stage
Edge AI:
Mobile
Laptop
Server
27GB+ RAM
Mobile
Laptop
Server
Quick Summary
AI model with specialized capabilities.
Device Compatibility
Mobile
4-6GB RAM
Laptop
16GB RAM
Server
GPU
Minimum Recommended
12GB+ RAM
Code Examples
ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)ORPO configtext
orpo_args = ORPOConfig(
learning_rate=8e-6,
lr_scheduler_type="cosine",
warmup_ratio=0.05,
max_length=4096,
max_prompt_length=1024,
max_completion_length=4096,
beta=0.1,
per_device_train_batch_size=1,
per_device_eval_batch_size=1,
gradient_accumulation_steps=64,
optim="paged_adamw_8bit",
num_train_epochs=3,
max_grad_norm=0.5,
bf16=True,
)Deploy This Model
Production-ready deployment in minutes
Together.ai
Instant API access to this model
Production-ready inference API. Start free, scale to millions.
Try Free APIReplicate
One-click model deployment
Run models in the cloud with simple API. No DevOps required.
Deploy NowDisclosure: We may earn a commission from these partners. This helps keep LLMYourWay free.