From 6ee1d53ce743672879af2dedd46ce23e0c1530d6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Quentin=20Gallou=C3=A9dec?= Date: Fri, 2 Oct 2026 18:04:25 +0000 Subject: [PATCH 1/4] Declare gradient accumulation loss scaling with loss_is_scaled_for_ga --- .../async_distillation/async_distillation_trainer.py | 6 +----- trl/experimental/async_grpo/async_grpo_trainer.py | 6 +----- trl/experimental/cpo/cpo_trainer.py | 6 +----- trl/experimental/orpo/orpo_trainer.py | 6 +----- trl/experimental/sdft/sdft_trainer.py | 3 +-- trl/experimental/sdpo/sdpo_trainer.py | 3 +-- .../server_distillation_trainer.py | 6 +++--- trl/experimental/ssd/ssd_trainer.py | 4 +--- trl/trainer/base_trainer.py | 9 +++++++++ trl/trainer/distillation_trainer.py | 12 +----------- trl/trainer/dpo_trainer.py | 6 +----- trl/trainer/grpo_trainer.py | 11 +---------- trl/trainer/kto_trainer.py | 6 +----- trl/trainer/reward_trainer.py | 6 +----- trl/trainer/rloo_trainer.py | 5 +---- 15 files changed, 25 insertions(+), 70 deletions(-) diff --git a/trl/experimental/async_distillation/async_distillation_trainer.py b/trl/experimental/async_distillation/async_distillation_trainer.py index 3a87c862e73..eb60c837885 100644 --- a/trl/experimental/async_distillation/async_distillation_trainer.py +++ b/trl/experimental/async_distillation/async_distillation_trainer.py @@ -933,6 +933,7 @@ class AsyncDistillationTrainer(_BaseTrainer): _tag_names = ["trl", "async-distillation"] _name = "AsyncDistillation" + loss_is_scaled_for_ga = True _paper = { "title": "On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes", "id": "2306.13649", @@ -1004,12 +1005,7 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - compute_loss_func="non-None value to disable scaling", ) - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether - # the model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False precision = self.accelerator.mixed_precision dtype = { diff --git a/trl/experimental/async_grpo/async_grpo_trainer.py b/trl/experimental/async_grpo/async_grpo_trainer.py index 166d37b05ed..ca6b9ad7bf5 100644 --- a/trl/experimental/async_grpo/async_grpo_trainer.py +++ b/trl/experimental/async_grpo/async_grpo_trainer.py @@ -1026,6 +1026,7 @@ class AsyncGRPOTrainer(_BaseTrainer): _tag_names = ["trl", "async-grpo"] _name = "AsyncGRPO" + loss_is_scaled_for_ga = True _paper = { "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models", "id": "2402.03300", @@ -1200,12 +1201,7 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - compute_loss_func="non-None value to disable scaling", ) - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False precision = self.accelerator.mixed_precision dtype = { diff --git a/trl/experimental/cpo/cpo_trainer.py b/trl/experimental/cpo/cpo_trainer.py index f1300f5de03..02821186c5e 100644 --- a/trl/experimental/cpo/cpo_trainer.py +++ b/trl/experimental/cpo/cpo_trainer.py @@ -119,6 +119,7 @@ class CPOTrainer(_BaseTrainer): _tag_names = ["trl", "cpo"] _name = "CPO" + loss_is_scaled_for_ga = False _paper = { "title": "Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation", "id": "2401.08417", @@ -413,11 +414,6 @@ def make_inputs_require_grad(module, input, output): preprocess_logits_for_metrics=preprocess_logits_for_metrics, ) - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - # Add tags for models that have been loaded with the correct transformers version if hasattr(self.model, "add_model_tags"): self.model.add_model_tags(self._tag_names) diff --git a/trl/experimental/orpo/orpo_trainer.py b/trl/experimental/orpo/orpo_trainer.py index fb1434ac085..4763b2d7493 100644 --- a/trl/experimental/orpo/orpo_trainer.py +++ b/trl/experimental/orpo/orpo_trainer.py @@ -131,6 +131,7 @@ class ORPOTrainer(_BaseTrainer): _tag_names = ["trl", "orpo"] _name = "ORPO" + loss_is_scaled_for_ga = False _paper = { "title": "ORPO: Monolithic Preference Optimization without Reference Model", "id": "2403.07691", @@ -395,11 +396,6 @@ def make_inputs_require_grad(module, input, output): preprocess_logits_for_metrics=preprocess_logits_for_metrics, ) - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - # Add tags for models that have been loaded with the correct transformers version if hasattr(self.model, "add_model_tags"): self.model.add_model_tags(self._tag_names) diff --git a/trl/experimental/sdft/sdft_trainer.py b/trl/experimental/sdft/sdft_trainer.py index dac922845ab..1b0e2a14c47 100644 --- a/trl/experimental/sdft/sdft_trainer.py +++ b/trl/experimental/sdft/sdft_trainer.py @@ -208,6 +208,7 @@ class SDFTTrainer(_BaseTrainer): _tag_names = ["trl", "sdft"] _name = "SDFT" + loss_is_scaled_for_ga = True config_cls = SDFTConfig # docstyle-ignore _paper = { @@ -383,7 +384,6 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - compute_loss_func="non-None value to disable scaling", ) self._last_loaded_step = -1 if self.use_vllm else 0 @@ -436,7 +436,6 @@ def __init__( self.model.add_model_tags(self._tag_names) self._setup_teacher_model() - self.model_accepts_loss_kwargs = False def _set_signature_columns_if_needed(self): if self._signature_columns is None: diff --git a/trl/experimental/sdpo/sdpo_trainer.py b/trl/experimental/sdpo/sdpo_trainer.py index 5d90686ea50..a34474912c3 100644 --- a/trl/experimental/sdpo/sdpo_trainer.py +++ b/trl/experimental/sdpo/sdpo_trainer.py @@ -332,6 +332,7 @@ class SDPOTrainer(_BaseTrainer): config_cls = SDPOConfig _tag_names = ["trl", "sdpo"] _name = "SDPO" + loss_is_scaled_for_ga = True # docstyle-ignore _paper = { "title": "Reinforcement Learning via Self-Distillation", @@ -511,7 +512,6 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - compute_loss_func="non-None value to disable scaling", ) self._last_loaded_step = -1 if self.use_vllm else 0 @@ -564,7 +564,6 @@ def __init__( self.model.add_model_tags(self._tag_names) self._setup_teacher_model() - self.model_accepts_loss_kwargs = False self.importance_sampling_level = args.importance_sampling_level self.scale_rewards = args.scale_rewards diff --git a/trl/experimental/server_distillation/server_distillation_trainer.py b/trl/experimental/server_distillation/server_distillation_trainer.py index 645efe39713..262539565de 100644 --- a/trl/experimental/server_distillation/server_distillation_trainer.py +++ b/trl/experimental/server_distillation/server_distillation_trainer.py @@ -336,9 +336,9 @@ def compute_loss(self, model, inputs, return_outputs=False, num_items_in_batch=N labels=trimmed_labels, ) - # The base trainer disables Trainer's built-in grad-accum loss scaling (via `compute_loss_func`) because it - # normalizes by the global completion-token count. The server path normalizes locally with `batchmean` and does - # not consume `num_items_in_batch`, so it must re-apply that scaling itself. + # The base trainer sets `loss_is_scaled_for_ga = True` because it normalizes by the global completion-token + # count. The server path normalizes locally with `batchmean` and does not consume `num_items_in_batch`, so it + # must re-apply that scaling itself. if self.model.training: loss = loss / self.current_gradient_accumulation_steps diff --git a/trl/experimental/ssd/ssd_trainer.py b/trl/experimental/ssd/ssd_trainer.py index 93d957116f2..917c454648d 100644 --- a/trl/experimental/ssd/ssd_trainer.py +++ b/trl/experimental/ssd/ssd_trainer.py @@ -81,6 +81,7 @@ class SSDTrainer(_BaseTrainer): _tag_names = ["trl", "ssd"] _name = "SSD" + loss_is_scaled_for_ga = True config_cls = SSDConfig # docstyle-ignore _paper = { @@ -227,7 +228,6 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - compute_loss_func="non-None value to disable scaling", ) if args.disable_dropout: @@ -235,8 +235,6 @@ def __init__( self.model.add_model_tags(self._tag_names) - self.model_accepts_loss_kwargs = False - if self.use_vllm: from ...generation.vllm_generation import VLLMGeneration diff --git a/trl/trainer/base_trainer.py b/trl/trainer/base_trainer.py index c35b4248e26..70d58e09c9a 100644 --- a/trl/trainer/base_trainer.py +++ b/trl/trainer/base_trainer.py @@ -16,9 +16,11 @@ from pathlib import Path import torch +import transformers from accelerate.utils import is_peft_model from datasets import Dataset from huggingface_hub.utils import send_telemetry +from packaging.version import Version from transformers import CONFIG_MAPPING, Trainer, is_wandb_available from .. import __version__ @@ -66,9 +68,16 @@ class _BaseTrainer(Trainer): _name = "Base" _paper = {} _template_file = None + # Whether `compute_loss` already scales the loss for gradient accumulation, see `Trainer.loss_is_scaled_for_ga` + loss_is_scaled_for_ga = None def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) + # `Trainer.loss_is_scaled_for_ga` requires transformers 5.19; older versions only read these two attributes + if Version(transformers.__version__) < Version("5.19.0.dev0") and self.loss_is_scaled_for_ga is not None: + self.model_accepts_loss_kwargs = False + if self.loss_is_scaled_for_ga: + self.compute_loss_func = "non-None value to disable scaling" self._send_telemetry() def _send_telemetry(self): diff --git a/trl/trainer/distillation_trainer.py b/trl/trainer/distillation_trainer.py index ed73490289a..f92e6f3e008 100644 --- a/trl/trainer/distillation_trainer.py +++ b/trl/trainer/distillation_trainer.py @@ -378,6 +378,7 @@ class DistillationTrainer(_BaseTrainer): _tag_names = ["trl", "distillation"] _name = "Distillation" + loss_is_scaled_for_ga = True _paper = { "title": "On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes", "id": "2306.13649", @@ -705,19 +706,8 @@ def __init__( processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - # In Trainer, `training_step` scales the loss by `gradient_accumulation_steps` only if `compute_loss_func` - # is None. Here, loss scaling instead depends on the total number of completion tokens across the global - # accumulated batch. To control scaling ourselves, we must disable Trainer's built-in scaling. The simplest - # (though a bit hacky) way is to set `compute_loss_func` to any non-None value, which bypasses that behavior - # without rewriting `training_step`. - compute_loss_func="non-None value to disable scaling", ) - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - self._dist = DistributedBackend(self.accelerator) # Add tags to the model diff --git a/trl/trainer/dpo_trainer.py b/trl/trainer/dpo_trainer.py index ed7bbc09251..51bf0927531 100644 --- a/trl/trainer/dpo_trainer.py +++ b/trl/trainer/dpo_trainer.py @@ -492,6 +492,7 @@ class DPOTrainer(_BaseTrainer): _tag_names = ["trl", "dpo"] _name = "DPO" + loss_is_scaled_for_ga = False _paper = { "title": "Direct Preference Optimization: Your Language Model is Secretly a Reward Model", "id": "2305.18290", @@ -952,11 +953,6 @@ def __init__( else: self._tp_size = 1 - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - # Add tags to the model self.model.add_model_tags(self._tag_names) diff --git a/trl/trainer/grpo_trainer.py b/trl/trainer/grpo_trainer.py index 9ee59882234..d8f54fff25d 100644 --- a/trl/trainer/grpo_trainer.py +++ b/trl/trainer/grpo_trainer.py @@ -283,6 +283,7 @@ class GRPOTrainer(_BaseTrainer): _tag_names = ["trl", "grpo"] _name = "GRPO" + loss_is_scaled_for_ga = True _paper = { "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models", "id": "2402.03300", @@ -930,12 +931,6 @@ def get_reward(environments, _env_type=env_type, **kwargs): processing_class=processing_class, callbacks=callbacks, optimizers=optimizers, - # In Trainer, `training_step` scales the loss by `gradient_accumulation_steps` only if `compute_loss_func` - # is None. For DAPO, loss scaling instead depends on the total number of completions tokens across the - # global accumulated batch. To control scaling ourselves, we must disable Trainer's built-in scaling. The - # simplest (though a bit hacky) way is to set `compute_loss_func` to any non-None value, which bypasses - # that behavior without rewriting `training_step`. - compute_loss_func="non-None value to disable scaling", ) # Reference model @@ -1112,10 +1107,6 @@ def cast_outputs_to_original_dtype(module, args, output): # Keep training-specific generation kwargs to overwrite model's original generation config self.generation_kwargs = generation_kwargs - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False self._dist = DistributedBackend(self.accelerator) # Add tags to the model diff --git a/trl/trainer/kto_trainer.py b/trl/trainer/kto_trainer.py index 5c4b3bac127..04e0f0ad51e 100644 --- a/trl/trainer/kto_trainer.py +++ b/trl/trainer/kto_trainer.py @@ -554,6 +554,7 @@ class KTOTrainer(_BaseTrainer): _tag_names = ["trl", "kto"] _name = "KTO" + loss_is_scaled_for_ga = False _paper = { "title": "KTO: Model Alignment as Prospect Theoretic Optimization", "id": "2402.01306", @@ -961,11 +962,6 @@ def __init__( else: self._tp_size = 1 - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - # Add tags to the model self.model.add_model_tags(self._tag_names) diff --git a/trl/trainer/reward_trainer.py b/trl/trainer/reward_trainer.py index 316411e7434..1c6efa2486d 100644 --- a/trl/trainer/reward_trainer.py +++ b/trl/trainer/reward_trainer.py @@ -329,6 +329,7 @@ class RewardTrainer(_BaseTrainer): _tag_names = ["trl", "reward-trainer"] _name = "Reward" + loss_is_scaled_for_ga = False _template_file = "rm_model_card.md" def __init__( @@ -619,11 +620,6 @@ def __init__( else: self._tp_size = 1 - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False - # Add tags to the model self.model.add_model_tags(self._tag_names) diff --git a/trl/trainer/rloo_trainer.py b/trl/trainer/rloo_trainer.py index c8738bb5046..d6ca2f174b4 100644 --- a/trl/trainer/rloo_trainer.py +++ b/trl/trainer/rloo_trainer.py @@ -217,6 +217,7 @@ class RLOOTrainer(_BaseTrainer): _tag_names = ["trl", "rloo"] _name = "RLOO" + loss_is_scaled_for_ga = False _paper = { "title": "Back to Basics: Revisiting REINFORCE-Style Optimization for Learning from Human Feedback in LLMs", "id": "2402.14740", @@ -750,10 +751,6 @@ def __init__( # Keep training-specific generation kwargs to overwrite model's original generation config self.generation_kwargs = generation_kwargs - # Gradient accumulation requires scaled loss. Normally, loss scaling in the parent class depends on whether the - # model accepts loss-related kwargs. Since we compute our own loss, this check is irrelevant. We set - # self.model_accepts_loss_kwargs to False to enable scaling. - self.model_accepts_loss_kwargs = False self._dist = DistributedBackend(self.accelerator) # Add tags to the model From c070eb22333283161de947c6ad6faae8251df3a8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Quentin=20Gallou=C3=A9dec?= Date: Fri, 2 Oct 2026 18:08:50 +0000 Subject: [PATCH 2/4] Put loss_is_scaled_for_ga first in the trainer class bodies --- .../async_distillation/async_distillation_trainer.py | 3 ++- trl/experimental/async_grpo/async_grpo_trainer.py | 3 ++- trl/experimental/cpo/cpo_trainer.py | 3 ++- trl/experimental/orpo/orpo_trainer.py | 3 ++- trl/experimental/sdft/sdft_trainer.py | 3 ++- trl/experimental/sdpo/sdpo_trainer.py | 3 ++- trl/experimental/ssd/ssd_trainer.py | 3 ++- trl/trainer/base_trainer.py | 5 +++-- trl/trainer/distillation_trainer.py | 3 ++- trl/trainer/dpo_trainer.py | 3 ++- trl/trainer/grpo_trainer.py | 3 ++- trl/trainer/kto_trainer.py | 3 ++- trl/trainer/reward_trainer.py | 3 ++- trl/trainer/rloo_trainer.py | 3 ++- 14 files changed, 29 insertions(+), 15 deletions(-) diff --git a/trl/experimental/async_distillation/async_distillation_trainer.py b/trl/experimental/async_distillation/async_distillation_trainer.py index eb60c837885..d674b742a93 100644 --- a/trl/experimental/async_distillation/async_distillation_trainer.py +++ b/trl/experimental/async_distillation/async_distillation_trainer.py @@ -931,9 +931,10 @@ class AsyncDistillationTrainer(_BaseTrainer): `rollout_worker` updates the policy itself). """ + loss_is_scaled_for_ga = True + _tag_names = ["trl", "async-distillation"] _name = "AsyncDistillation" - loss_is_scaled_for_ga = True _paper = { "title": "On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes", "id": "2306.13649", diff --git a/trl/experimental/async_grpo/async_grpo_trainer.py b/trl/experimental/async_grpo/async_grpo_trainer.py index ca6b9ad7bf5..26c600d542b 100644 --- a/trl/experimental/async_grpo/async_grpo_trainer.py +++ b/trl/experimental/async_grpo/async_grpo_trainer.py @@ -1024,9 +1024,10 @@ class AsyncGRPOTrainer(_BaseTrainer): implementation to disable trainer-side weight sync. """ + loss_is_scaled_for_ga = True + _tag_names = ["trl", "async-grpo"] _name = "AsyncGRPO" - loss_is_scaled_for_ga = True _paper = { "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models", "id": "2402.03300", diff --git a/trl/experimental/cpo/cpo_trainer.py b/trl/experimental/cpo/cpo_trainer.py index 02821186c5e..0ebd23cfd4c 100644 --- a/trl/experimental/cpo/cpo_trainer.py +++ b/trl/experimental/cpo/cpo_trainer.py @@ -80,6 +80,8 @@ class CPOTrainer(_BaseTrainer): + loss_is_scaled_for_ga = False + r""" Initialize CPOTrainer. @@ -119,7 +121,6 @@ class CPOTrainer(_BaseTrainer): _tag_names = ["trl", "cpo"] _name = "CPO" - loss_is_scaled_for_ga = False _paper = { "title": "Contrastive Preference Optimization: Pushing the Boundaries of LLM Performance in Machine Translation", "id": "2401.08417", diff --git a/trl/experimental/orpo/orpo_trainer.py b/trl/experimental/orpo/orpo_trainer.py index 4763b2d7493..8ce7039ed73 100644 --- a/trl/experimental/orpo/orpo_trainer.py +++ b/trl/experimental/orpo/orpo_trainer.py @@ -92,6 +92,8 @@ def log1mexp(x: torch.FloatTensor) -> torch.FloatTensor: class ORPOTrainer(_BaseTrainer): + loss_is_scaled_for_ga = False + r""" Initialize ORPOTrainer. @@ -131,7 +133,6 @@ class ORPOTrainer(_BaseTrainer): _tag_names = ["trl", "orpo"] _name = "ORPO" - loss_is_scaled_for_ga = False _paper = { "title": "ORPO: Monolithic Preference Optimization without Reference Model", "id": "2403.07691", diff --git a/trl/experimental/sdft/sdft_trainer.py b/trl/experimental/sdft/sdft_trainer.py index 1b0e2a14c47..da122eac757 100644 --- a/trl/experimental/sdft/sdft_trainer.py +++ b/trl/experimental/sdft/sdft_trainer.py @@ -206,9 +206,10 @@ def build( class SDFTTrainer(_BaseTrainer): """Trainer for SDFT-style on-policy self-distillation with explicit teacher prompts.""" + loss_is_scaled_for_ga = True + _tag_names = ["trl", "sdft"] _name = "SDFT" - loss_is_scaled_for_ga = True config_cls = SDFTConfig # docstyle-ignore _paper = { diff --git a/trl/experimental/sdpo/sdpo_trainer.py b/trl/experimental/sdpo/sdpo_trainer.py index a34474912c3..154b09f1883 100644 --- a/trl/experimental/sdpo/sdpo_trainer.py +++ b/trl/experimental/sdpo/sdpo_trainer.py @@ -329,10 +329,11 @@ class SDPOTrainer(_BaseTrainer): next-token predictions back into the policy. """ + loss_is_scaled_for_ga = True + config_cls = SDPOConfig _tag_names = ["trl", "sdpo"] _name = "SDPO" - loss_is_scaled_for_ga = True # docstyle-ignore _paper = { "title": "Reinforcement Learning via Self-Distillation", diff --git a/trl/experimental/ssd/ssd_trainer.py b/trl/experimental/ssd/ssd_trainer.py index 917c454648d..54dadfd6238 100644 --- a/trl/experimental/ssd/ssd_trainer.py +++ b/trl/experimental/ssd/ssd_trainer.py @@ -79,9 +79,10 @@ class SSDTrainer(_BaseTrainer): ``prompt`` column. """ + loss_is_scaled_for_ga = True + _tag_names = ["trl", "ssd"] _name = "SSD" - loss_is_scaled_for_ga = True config_cls = SSDConfig # docstyle-ignore _paper = { diff --git a/trl/trainer/base_trainer.py b/trl/trainer/base_trainer.py index 70d58e09c9a..8e99b7ef858 100644 --- a/trl/trainer/base_trainer.py +++ b/trl/trainer/base_trainer.py @@ -64,12 +64,13 @@ class _BaseTrainer(Trainer): + # Whether `compute_loss` already scales the loss for gradient accumulation, see `Trainer.loss_is_scaled_for_ga` + loss_is_scaled_for_ga = None + _tag_names = [] _name = "Base" _paper = {} _template_file = None - # Whether `compute_loss` already scales the loss for gradient accumulation, see `Trainer.loss_is_scaled_for_ga` - loss_is_scaled_for_ga = None def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) diff --git a/trl/trainer/distillation_trainer.py b/trl/trainer/distillation_trainer.py index f92e6f3e008..78d41fcad85 100644 --- a/trl/trainer/distillation_trainer.py +++ b/trl/trainer/distillation_trainer.py @@ -376,9 +376,10 @@ class DistillationTrainer(_BaseTrainer): use and that it has been fine-tuned for tool calling. """ + loss_is_scaled_for_ga = True + _tag_names = ["trl", "distillation"] _name = "Distillation" - loss_is_scaled_for_ga = True _paper = { "title": "On-Policy Distillation of Language Models: Learning from Self-Generated Mistakes", "id": "2306.13649", diff --git a/trl/trainer/dpo_trainer.py b/trl/trainer/dpo_trainer.py index 51bf0927531..0928ce389ca 100644 --- a/trl/trainer/dpo_trainer.py +++ b/trl/trainer/dpo_trainer.py @@ -490,9 +490,10 @@ class DPOTrainer(_BaseTrainer): PEFT configuration used to wrap the model. If `None`, the model is not wrapped. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "dpo"] _name = "DPO" - loss_is_scaled_for_ga = False _paper = { "title": "Direct Preference Optimization: Your Language Model is Secretly a Reward Model", "id": "2305.18290", diff --git a/trl/trainer/grpo_trainer.py b/trl/trainer/grpo_trainer.py index d8f54fff25d..936e180e922 100644 --- a/trl/trainer/grpo_trainer.py +++ b/trl/trainer/grpo_trainer.py @@ -281,9 +281,10 @@ class GRPOTrainer(_BaseTrainer): any time without prior notice. """ + loss_is_scaled_for_ga = True + _tag_names = ["trl", "grpo"] _name = "GRPO" - loss_is_scaled_for_ga = True _paper = { "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models", "id": "2402.03300", diff --git a/trl/trainer/kto_trainer.py b/trl/trainer/kto_trainer.py index 04e0f0ad51e..55d958e731a 100644 --- a/trl/trainer/kto_trainer.py +++ b/trl/trainer/kto_trainer.py @@ -552,9 +552,10 @@ class KTOTrainer(_BaseTrainer): PEFT configuration used to wrap the model. If `None`, the model is not wrapped. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "kto"] _name = "KTO" - loss_is_scaled_for_ga = False _paper = { "title": "KTO: Model Alignment as Prospect Theoretic Optimization", "id": "2402.01306", diff --git a/trl/trainer/reward_trainer.py b/trl/trainer/reward_trainer.py index 1c6efa2486d..e55627e02d0 100644 --- a/trl/trainer/reward_trainer.py +++ b/trl/trainer/reward_trainer.py @@ -327,9 +327,10 @@ class RewardTrainer(_BaseTrainer): to ensure that the reward head is properly trained. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "reward-trainer"] _name = "Reward" - loss_is_scaled_for_ga = False _template_file = "rm_model_card.md" def __init__( diff --git a/trl/trainer/rloo_trainer.py b/trl/trainer/rloo_trainer.py index d6ca2f174b4..5d218afe578 100644 --- a/trl/trainer/rloo_trainer.py +++ b/trl/trainer/rloo_trainer.py @@ -215,9 +215,10 @@ class RLOOTrainer(_BaseTrainer): PEFT configuration used to wrap the model. If `None`, the model is not wrapped. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "rloo"] _name = "RLOO" - loss_is_scaled_for_ga = False _paper = { "title": "Back to Basics: Revisiting REINFORCE-Style Optimization for Learning from Human Feedback in LLMs", "id": "2402.14740", From 1d027f0d96152e7475bc21fcbab3d6d9f6753164 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Quentin=20Gallou=C3=A9dec?= Date: Fri, 2 Oct 2026 18:08:59 +0000 Subject: [PATCH 3/4] Keep the CPO and ORPO docstrings first --- trl/experimental/cpo/cpo_trainer.py | 4 ++-- trl/experimental/orpo/orpo_trainer.py | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/trl/experimental/cpo/cpo_trainer.py b/trl/experimental/cpo/cpo_trainer.py index 0ebd23cfd4c..5683fb0ec31 100644 --- a/trl/experimental/cpo/cpo_trainer.py +++ b/trl/experimental/cpo/cpo_trainer.py @@ -80,8 +80,6 @@ class CPOTrainer(_BaseTrainer): - loss_is_scaled_for_ga = False - r""" Initialize CPOTrainer. @@ -119,6 +117,8 @@ class CPOTrainer(_BaseTrainer): metric values. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "cpo"] _name = "CPO" _paper = { diff --git a/trl/experimental/orpo/orpo_trainer.py b/trl/experimental/orpo/orpo_trainer.py index 8ce7039ed73..7b9eedc5a9e 100644 --- a/trl/experimental/orpo/orpo_trainer.py +++ b/trl/experimental/orpo/orpo_trainer.py @@ -92,8 +92,6 @@ def log1mexp(x: torch.FloatTensor) -> torch.FloatTensor: class ORPOTrainer(_BaseTrainer): - loss_is_scaled_for_ga = False - r""" Initialize ORPOTrainer. @@ -131,6 +129,8 @@ class ORPOTrainer(_BaseTrainer): metric values. """ + loss_is_scaled_for_ga = False + _tag_names = ["trl", "orpo"] _name = "ORPO" _paper = { From b181e8b6ad93d556d89333b29e2f175de9a645dd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Quentin=20Gallou=C3=A9dec?= Date: Thu, 8 Oct 2026 16:40:40 +0000 Subject: [PATCH 4/4] Compare against the released transformers 5.19.0 --- trl/trainer/base_trainer.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/trl/trainer/base_trainer.py b/trl/trainer/base_trainer.py index 5ca9ff0fa6f..f9dab55e75a 100644 --- a/trl/trainer/base_trainer.py +++ b/trl/trainer/base_trainer.py @@ -74,7 +74,7 @@ class _BaseTrainer(Trainer): def __init__(self, *args, **kwargs): super().__init__(*args, **kwargs) # `Trainer.loss_is_scaled_for_ga` requires transformers 5.19; older versions only read these two attributes - if Version(transformers.__version__) < Version("5.19.0.dev0") and self.loss_is_scaled_for_ga is not None: + if Version(transformers.__version__) < Version("5.19.0") and self.loss_is_scaled_for_ga is not None: self.model_accepts_loss_kwargs = False if self.loss_is_scaled_for_ga: self.compute_loss_func = "non-None value to disable scaling"