DPO — Preference Optimization without a Reward Model#
Validates the core conclusions of dpo.tex:
The same data, two fewer components: on the same preference pairs, DPO trains no reward model and runs no PPO, reaching a true score comparable to Chapter 12’s RLHF pipeline;
The implicit reward, the same disease: DPO is equivalent to fitting an implicit reward \(\hat{r}(y)=\beta\log\frac{\pi(y)}{\pi_{\mathrm{ref}}(y)}\) — it extrapolates monotonously outside the coverage just like Chapter 12’s explicit reward model, making the same mistake;
A loosening anchor: smaller β drifts faster; even at the sweet β, prolonged training slowly drifts the policy out of the coverage — training length is DPO’s hidden hyperparameter; early stopping is not optional.
The task is exactly Chapter 12’s (vocabulary 24, sequence length 12, concave emphasis bonus); preference data likewise comes from SFT sampling labeled by a noisy BT model. The control arm replicates Chapter 12’s pipeline: reward model + PPO (β=0.5).
Output figures:
fig1_dpo_vs_rlhf.pdffig2_implicit_reward.pdffig3_beta_anchor.pdf
Estimated runtime: about 12 minutes on GPU; about 1 hour on CPU (3 seeds).
import numpy as np
import torch
import torch.nn as nn
import torch.nn.functional as F
import matplotlib
import matplotlib.pyplot as plt
matplotlib.rcParams.update({
'font.family': 'serif',
'font.serif': ['Times New Roman', 'DejaVu Serif'],
'pdf.fonttype': 42,
'ps.fonttype': 42,
'axes.labelsize': 11,
'xtick.labelsize': 9,
'ytick.labelsize': 9,
'legend.fontsize': 9,
})
BLUE = '#2166AC'
RED = '#D6604D'
GRAY = '#808080'
FIGSIZE = (7.2, 4.8)
OUTDIR = '.'
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
VOCAB = 24
BOS = VOCAB
SEQ_LEN = 12
E_TOKEN = 12
T_BASE = 1.5
D_MODEL = 64
SEEDS = (42, 43, 44)
TOKEN_VALUES = np.linspace(-1.0, 1.0, VOCAB)
TOKEN_VALUES_T = torch.tensor(TOKEN_VALUES, dtype=torch.float32, device=DEVICE)
PEAK_C = 3
BONUS_UP = 0.3
BONUS_DOWN = 0.6
def bonus(count):
up = BONUS_UP * torch.clamp(count.float(), max=PEAK_C)
down = BONUS_DOWN * torch.clamp(count.float() - PEAK_C, min=0)
return up - down
def true_score(seqs):
"""Ground-truth quality: mean token value + concave emphasis bonus."""
vals = TOKEN_VALUES_T[seqs].mean(dim=1)
return vals + bonus((seqs == E_TOKEN).sum(dim=1))
def set_seed(seed):
np.random.seed(seed)
torch.manual_seed(seed)
def style_axes(ax):
ax.spines['top'].set_visible(False)
ax.spines['right'].set_visible(False)
ax.grid(True, linestyle='--', linewidth=0.5, alpha=0.7)
print(f'Device: {DEVICE}')
Device: cuda
class TinyLM(nn.Module):
"""Char-level causal Transformer LM (~120k params), same as chapter 12."""
def __init__(self):
super().__init__()
self.emb = nn.Embedding(VOCAB + 1, D_MODEL)
self.pos = nn.Embedding(SEQ_LEN + 1, D_MODEL)
layer = nn.TransformerEncoderLayer(D_MODEL, 4, 128, batch_first=True,
dropout=0.0, norm_first=True)
self.encoder = nn.TransformerEncoder(layer, 2, enable_nested_tensor=False)
self.head = nn.Linear(D_MODEL, VOCAB)
mask = torch.triu(torch.full((SEQ_LEN + 1, SEQ_LEN + 1), float('-inf')), 1)
self.register_buffer('mask', mask)
def logits(self, inp):
L = inp.size(1)
h = self.emb(inp) + self.pos(torch.arange(L, device=inp.device))
h = self.encoder(h, mask=self.mask[:L, :L])
return self.head(h)
@torch.no_grad()
def generate(self, batch):
inp = torch.full((batch, 1), BOS, dtype=torch.long, device=DEVICE)
for _ in range(SEQ_LEN):
logits = self.logits(inp)[:, -1]
nxt = torch.multinomial(F.softmax(logits, dim=-1), 1)
inp = torch.cat([inp, nxt], dim=1)
return inp[:, 1:]
def log_probs(self, seqs):
inp = torch.cat([torch.full((seqs.size(0), 1), BOS, dtype=torch.long,
device=DEVICE), seqs[:, :-1]], dim=1)
logits = self.logits(inp)
return torch.log_softmax(logits, dim=-1).gather(2, seqs.unsqueeze(2)).squeeze(2)
class RewardModel(nn.Module):
"""Chapter-12 explicit reward model, used only by the RLHF baseline."""
def __init__(self):
super().__init__()
self.emb = nn.Embedding(VOCAB, D_MODEL)
self.net = nn.Sequential(nn.Linear(D_MODEL, D_MODEL), nn.ReLU(),
nn.Linear(D_MODEL, 1))
self.mu = 0.0
self.sigma = 1.0
def raw(self, seqs):
return self.net(self.emb(seqs).mean(dim=1)).squeeze(1)
def forward(self, seqs):
return (self.raw(seqs) - self.mu) / self.sigma
@torch.no_grad()
def calibrate(self, base_seqs):
r = self.raw(base_seqs)
self.mu = r.mean().item()
self.sigma = r.std().item() + 1e-8
def sample_base_corpus(n):
probs = F.softmax(torch.tensor(TOKEN_VALUES / T_BASE), dim=0).numpy()
return torch.tensor(np.random.choice(VOCAB, size=(n, SEQ_LEN), p=probs),
dtype=torch.long, device=DEVICE)
def train_sft(steps=800, batch=256, lr=3e-3):
lm = TinyLM().to(DEVICE)
opt = torch.optim.Adam(lm.parameters(), lr=lr)
for _ in range(steps):
loss = -lm.log_probs(sample_base_corpus(batch)).mean()
opt.zero_grad(); loss.backward(); opt.step()
return lm
def make_pref_pairs(lm, n_pairs, noise=0.1):
a = lm.generate(n_pairs)
b = lm.generate(n_pairs)
sa, sb = true_score(a), true_score(b)
a_wins = torch.bernoulli(torch.sigmoid((sa - sb) / noise)).bool()
return (torch.where(a_wins.unsqueeze(1), a, b),
torch.where(a_wins.unsqueeze(1), b, a))
def train_rm(winners, losers, base, steps=600, batch=128, lr=1e-3):
rm = RewardModel().to(DEVICE)
opt = torch.optim.Adam(rm.parameters(), lr=lr)
n = winners.size(0)
for _ in range(steps):
idx = torch.randint(0, n, (batch,), device=DEVICE)
loss = -F.logsigmoid(rm.raw(winners[idx]) - rm.raw(losers[idx])).mean()
opt.zero_grad(); loss.backward(); opt.step()
rm.calibrate(base)
return rm
def ppo_finetune(sft, rm, beta, iters=150, batch=256, k_epochs=4, mb=64,
clip=0.2, lr=3e-5):
"""Chapter-12 RLHF baseline: PPO-clip against the explicit reward model."""
policy = TinyLM().to(DEVICE)
policy.load_state_dict(sft.state_dict())
ref = TinyLM().to(DEVICE)
ref.load_state_dict(sft.state_dict())
ref.eval()
opt = torch.optim.Adam(policy.parameters(), lr=lr)
truth_hist = []
for it in range(iters):
with torch.no_grad():
seqs = policy.generate(batch)
lp_old = policy.log_probs(seqs)
lp_ref = ref.log_probs(seqs)
shaped = rm(seqs) - beta * (lp_old - lp_ref).sum(dim=1)
adv = (shaped - shaped.mean()) / (shaped.std() + 1e-8)
for _ in range(k_epochs):
perm = torch.randperm(batch, device=DEVICE)
for s in range(0, batch, mb):
idx = perm[s:s + mb]
lp_new = policy.log_probs(seqs[idx])
ratio = torch.exp((lp_new - lp_old[idx]).sum(dim=1))
s1 = ratio * adv[idx]
s2 = torch.clamp(ratio, 1 - clip, 1 + clip) * adv[idx]
loss = -torch.min(s1, s2).mean()
opt.zero_grad(); loss.backward()
nn.utils.clip_grad_norm_(policy.parameters(), 1.0)
opt.step()
truth_hist.append(true_score(seqs).mean().item())
return truth_hist
def train_dpo(sft, winners, losers, beta, steps=3000, batch=128, lr=1e-4,
eval_every=50):
"""Direct Preference Optimization on fixed preference pairs.
No reward model, no sampling during training; the policy is evaluated
by generation only for logging.
"""
policy = TinyLM().to(DEVICE)
policy.load_state_dict(sft.state_dict())
ref = TinyLM().to(DEVICE)
ref.load_state_dict(sft.state_dict())
ref.eval()
opt = torch.optim.Adam(policy.parameters(), lr=lr)
n = winners.size(0)
history = {'step': [], 'truth': [], 'count_e': [], 'kl': []}
for step in range(1, steps + 1):
idx = torch.randint(0, n, (batch,), device=DEVICE)
w, l = winners[idx], losers[idx]
with torch.no_grad():
ref_w = ref.log_probs(w).sum(1)
ref_l = ref.log_probs(l).sum(1)
pol_w = policy.log_probs(w).sum(1)
pol_l = policy.log_probs(l).sum(1)
margin = (pol_w - ref_w) - (pol_l - ref_l)
loss = -F.logsigmoid(beta * margin).mean()
opt.zero_grad(); loss.backward()
nn.utils.clip_grad_norm_(policy.parameters(), 1.0)
opt.step()
if step % eval_every == 0:
with torch.no_grad():
gen = policy.generate(512)
history['step'].append(step)
history['truth'].append(true_score(gen).mean().item())
history['count_e'].append(
(gen == E_TOKEN).float().sum(dim=1).mean().item())
history['kl'].append(
(policy.log_probs(gen) - ref.log_probs(gen)).sum(1).mean().item())
return policy, ref, history
@torch.no_grad()
def implicit_reward_curve(policy, ref, beta, n=512):
"""Implicit reward beta*log(pi/pi_ref) for sequences with exactly c emphasis tokens."""
rows = []
for c in range(SEQ_LEN + 1):
seqs = sample_base_corpus(n)
seqs = seqs.masked_fill(seqs == E_TOKEN, E_TOKEN + 1)
for i in range(n):
pos = torch.randperm(SEQ_LEN, device=DEVICE)[:c]
seqs[i, pos] = E_TOKEN
imp = beta * (policy.log_probs(seqs).sum(1) - ref.log_probs(seqs).sum(1))
rows.append((imp.mean().item(), true_score(seqs).mean().item()))
return rows
print('=== SFT + preference data, per seed (same recipe as chapter 12) ===')
runs = {}
for seed in SEEDS:
set_seed(seed)
sft = train_sft()
base = sft.generate(2000)
winners, losers = make_pref_pairs(sft, 4000)
runs[seed] = {'sft': sft, 'base': base,
'winners': winners, 'losers': losers}
print(f' seed {seed}: base truth {true_score(base).mean():+.3f}')
=== SFT + preference data, per seed (same recipe as chapter 12) ===
seed 42: base truth +0.389
seed 43: base truth +0.384
seed 44: base truth +0.386
Figure 1 — DPO vs RLHF: the same preference data#
The control arm is Chapter 12’s full pipeline: train a reward model (BT + whitening), then PPO with a KL penalty (β=0.5, final true score about +0.84). The DPO arm optimizes the policy directly on the same 4000 preference pairs — no reward model, no sampling-based training. DPO with β=2.0 reaches an RLHF-comparable true score (about +0.8 ~ +1.0) at about 1000–2000 steps — two roads to the same objective, DPO saving two components (3 seeds, shading \(\pm 1\sigma\)).
print('=== Experiment 1: DPO (beta=2.0) vs the chapter-12 RLHF pipeline ===')
dpo_hists = {}
rlhf_finals = []
for seed in SEEDS:
set_seed(seed + 2000)
r = runs[seed]
_, _, dpo_hists[(seed, 2.0)] = train_dpo(r['sft'], r['winners'], r['losers'],
beta=2.0)
h = dpo_hists[(seed, 2.0)]
set_seed(seed + 5000)
rm = train_rm(r['winners'], r['losers'], r['base'])
rlhf_truth = ppo_finetune(r['sft'], rm, beta=0.5)
rlhf_finals.append(np.mean(rlhf_truth[-10:]))
print(f" seed {seed}: DPO peak {max(h['truth']):+.3f} | "
f"DPO final {h['truth'][-1]:+.3f} | RLHF final {rlhf_finals[-1]:+.3f}")
rlhf_mean = float(np.mean(rlhf_finals))
base_truth = float(np.mean([true_score(runs[s]['base']).mean().item()
for s in SEEDS]))
print(f'RLHF reference (mean of final-10): {rlhf_mean:+.3f} | SFT: {base_truth:+.3f}')
=== Experiment 1: DPO (beta=2.0) vs the chapter-12 RLHF pipeline ===
seed 42: DPO peak +0.942 | DPO final +0.588 | RLHF final +0.850
seed 43: DPO peak +0.989 | DPO final +0.924 | RLHF final +0.846
seed 44: DPO peak +1.045 | DPO final +1.035 | RLHF final +0.811
RLHF reference (mean of final-10): +0.835 | SFT: +0.386
def stack(key, beta):
return np.array([dpo_hists[(seed, beta)][key] for seed in SEEDS])
fig, ax = plt.subplots(figsize=FIGSIZE)
steps = np.array(dpo_hists[(SEEDS[0], 2.0)]['step'])
truth = stack('truth', 2.0)
ax.plot(steps, truth.mean(0), color=BLUE, linewidth=1.5, label='DPO ($\\beta=2.0$)')
ax.fill_between(steps, truth.mean(0) - truth.std(0), truth.mean(0) + truth.std(0),
color=BLUE, alpha=0.15)
ax.axhline(rlhf_mean, color=RED, linestyle='--', linewidth=1.2,
label=f'RLHF pipeline (RM + PPO), final {rlhf_mean:+.2f}')
ax.axhline(base_truth, color='black', linestyle=':', linewidth=1.0,
label='SFT baseline')
style_axes(ax)
ax.set_xlabel('DPO training step')
ax.set_ylabel('True score')
ax.set_title('Fig 1. DPO matches the RLHF pipeline on the same pairs\n'
'(3 seeds, mean $\\pm$ std)', pad=8)
ax.legend(loc='lower right')
fig.tight_layout()
fig.savefig(f'{OUTDIR}/fig1_dpo_vs_rlhf.pdf', bbox_inches='tight')
plt.show()
print('Saved fig1_dpo_vs_rlhf.pdf')
Saved fig1_dpo_vs_rlhf.pdf
Figure 2 — The implicit reward: the same disease#
DPO’s loss is equivalent to fitting an implicit reward \(\hat{r}(y)=\beta\log\frac{\pi(y)}{\pi_{\mathrm{ref}}(y)}\). Grouping the trained policy’s implicit reward by emphasis-token count \(c_E\) (this figure takes β=0.1, the collapsing group of Figure 3): within the coverage it tracks the true score (the preference is learned); outside, the true score collapses while the implicit reward keeps rising monotonically — just like Chapter 12’s explicit reward model’s extrapolation (its Figure 1, right). Skipping the reward-model component does not skip the fact that rewards are trustworthy only within the data coverage (3 seeds, shading \(\pm 1\sigma\)).
print('=== Experiment 2: DPO with beta=0.1, then measure the implicit reward ===')
imp_curves = []
for seed in SEEDS:
set_seed(seed + 3000)
r = runs[seed]
policy, ref, dpo_hists[(seed, 0.1)] = train_dpo(r['sft'], r['winners'],
r['losers'], beta=0.1)
h = dpo_hists[(seed, 0.1)]
set_seed(seed + 4000)
imp_curves.append(implicit_reward_curve(policy, ref, beta=0.1))
print(f" seed {seed}: final truth {h['truth'][-1]:+.3f} | "
f"final c_E {h['count_e'][-1]:.1f} | final KL {h['kl'][-1]:.1f}")
imp_curves = np.array(imp_curves) # (seeds, c, [implicit, true])
=== Experiment 2: DPO with beta=0.1, then measure the implicit reward ===
seed 42: final truth -4.457 | final c_E 12.0 | final KL 38.0
seed 43: final truth -4.230 | final c_E 11.6 | final KL 37.2
seed 44: final truth -2.821 | final c_E 9.5 | final KL 29.6
fig, ax = plt.subplots(figsize=FIGSIZE)
cs = np.arange(SEQ_LEN + 1)
imp_mu, imp_sd = imp_curves[:, :, 0].mean(0), imp_curves[:, :, 0].std(0)
ax.plot(cs, imp_mu, color=BLUE, marker='o', markersize=3, linewidth=1.5,
label='implicit reward $\\beta\\log(\\pi/\\pi_{ref})$')
ax.fill_between(cs, imp_mu - imp_sd, imp_mu + imp_sd, color=BLUE, alpha=0.15)
ax.set_ylabel('Implicit reward', color=BLUE)
ax.set_xlabel('Emphasis-token count $c_E$')
ax2 = ax.twinx()
ax2.plot(cs, imp_curves[0, :, 1], color=RED, marker='s', markersize=3,
linewidth=1.5, label='true score')
ax2.set_ylabel('True score', color=RED)
ax.axvspan(0, PEAK_C, color=GRAY, alpha=0.15)
ax.text(PEAK_C / 2, ax.get_ylim()[1] * 0.86, 'data\ncoverage',
ha='center', fontsize=8, color=GRAY)
style_axes(ax)
ax2.spines['top'].set_visible(False)
ax.set_title('Fig 2. The implicit reward extrapolates like the explicit one\n'
'(3 seeds, mean $\\pm$ std)', pad=8)
h1, l1 = ax.get_legend_handles_labels()
h2, l2 = ax2.get_legend_handles_labels()
ax.legend(h1 + h2, l1 + l2, loc='center right')
fig.tight_layout()
fig.savefig(f'{OUTDIR}/fig2_implicit_reward.pdf', bbox_inches='tight')
plt.show()
print('Saved fig2_implicit_reward.pdf')
Saved fig2_implicit_reward.pdf
Figure 3 — β and training length: the loosening anchor#
DPO has no explicit KL penalty; β plays the anchor via the implicit reward’s definition — but this anchor is soft: as long as the loss is unsaturated, it keeps amplifying the \(\log\pi/\pi_{\mathrm{ref}}\) gap. β=0.1 pushes the generation distribution out of coverage within a few hundred steps (\(c_E \to 12\), true score collapsing); β=0.5 collapses more slowly but equally; β=2.0 reaches the sweet spot (about +0.9) and then drifts slowly too. DPO never samples its own generations, yet token-level generalization still carries it out of the coverage — β and training steps jointly set the drift distance; early stopping is DPO’s hidden hyperparameter (3 seeds, shading \(\pm 1\sigma\)).
print('=== Experiment 3: beta sweep (0.5 remaining; 0.1 and 2.0 reused) ===')
for seed in SEEDS:
set_seed(seed + 6000)
r = runs[seed]
_, _, dpo_hists[(seed, 0.5)] = train_dpo(r['sft'], r['winners'], r['losers'],
beta=0.5)
for beta in (0.1, 0.5, 2.0):
truths = np.array([dpo_hists[(s, beta)]['truth'][-1] for s in SEEDS])
ces = np.array([dpo_hists[(s, beta)]['count_e'][-1] for s in SEEDS])
print(f' beta {beta}: final truth {truths.mean():+.3f} ± {truths.std():.3f} | '
f'final c_E {ces.mean():.1f}')
=== Experiment 3: beta sweep (0.5 remaining; 0.1 and 2.0 reused) ===
beta 0.1: final truth -3.836 ± 0.723 | final c_E 11.0
beta 0.5: final truth -4.050 ± 0.295 | final c_E 11.4
beta 2.0: final truth +0.849 ± 0.190 | final c_E 3.4
fig, ax = plt.subplots(figsize=FIGSIZE)
steps = np.array(dpo_hists[(SEEDS[0], 2.0)]['step'])
for beta, color, label in [(0.1, RED, r'$\beta=0.1$'),
(0.5, GRAY, r'$\beta=0.5$'),
(2.0, BLUE, r'$\beta=2.0$')]:
truth = stack('truth', beta)
ax.plot(steps, truth.mean(0), color=color, linewidth=1.5, label=label)
ax.fill_between(steps, truth.mean(0) - truth.std(0),
truth.mean(0) + truth.std(0), color=color, alpha=0.15)
ax.axhline(base_truth, color='black', linestyle=':', linewidth=1.0,
label='SFT baseline')
style_axes(ax)
ax.set_xlabel('DPO training step')
ax.set_ylabel('True score')
ax.set_title('Fig 3. The DPO anchor loosens with training\n'
'(3 seeds, mean $\\pm$ std)', pad=8)
ax.legend(loc='lower left')
fig.tight_layout()
fig.savefig(f'{OUTDIR}/fig3_beta_anchor.pdf', bbox_inches='tight')
plt.show()
print('Saved fig3_beta_anchor.pdf')
Saved fig3_beta_anchor.pdf
Summary#
DPO is an analytic shortcut to RLHF: the KL-regularized reward maximization has a closed-form optimal policy; substituting the implicit reward back into the BT likelihood removes both the reward model and PPO — the same preference data, a comparable peak true score (Figure 1);
Skipping components does not skip the disease: the implicit reward extrapolates monotonically outside the coverage exactly like the explicit model (Figure 2) — Chapter 12’s lesson applies to DPO verbatim;
The anchor is soft: DPO has no hard KL constraint; β is only a speed bump; even without sampling, token-level generalization pushes the generation distribution out of the coverage — too small a β collapses fast, a moderate one drifts slowly; training steps must be treated as a hyperparameter (Figure 3);
The two chapters combined: from RLHF to DPO, what changes is the implementation path, not “the preference data’s coverage sets the optimizable boundary”. The next chapter, GRPO/RLVR, uses a rule-checked reward on toy arithmetic so the proxy cannot diverge from the true objective on that task.