So far we've covered training agents with verifiable rewards using GRPO and search. But training doesn't end at deployment. The most sophisticated agents continuously improve by learning from their own real-world trajectories. This is online learning in the agentic context.
Code repo: RL_Agents
Notebook: post05_online_learning.ipynb
Traditional RL breaks neatly into two phases:
Online learning breaks this boundary:
This is inspired by a technique called STaR (Self-Taught Reasoner): the model bootstraps its own training data from its own successful outputs.
STaR is simple and powerful:
The code maps each GRPO component to a concrete bandit experiment so you can watch the update mechanics in action.
from __future__ import annotations
import math, random
import numpy as np
import pandas as pd
def ppo_clipped_objective(
old_prob: float, new_prob: float, advantage: float, eps: float = 0.2,
) -> float:
"""PPO clipped surrogate: L = min(r*A, clip(r, 1-eps, 1+eps)*A)"""
ratio = new_prob / max(old_prob, 1e-9)
clipped_ratio = min(max(ratio, 1 - eps), 1 + eps)
return min(ratio * advantage, clipped_ratio * advantage)
def grpo_relative_advantages(rewards: list[float]) -> list[float]:
"""A_i = r_i - mean(r_1...r_k) — no critic, only group baseline."""
baseline = sum(rewards) / max(len(rewards), 1)
return [round(r - baseline, 3) for r in rewards]
def beam_search_to_target(
target: int = 15, width: int = 3, depth: int = 4,
) -> pd.DataFrame:
"""Beam search with {+1, +3, *2}, keeping top-width beams by distance."""
beams = [(0, "start")]
ops = [("+1", 1), ("+3", 3), ("*2", 2)]
for _ in range(depth):
candidates: list[tuple[int, str]] = []
for value, trace in beams:
for name, op in ops:
nv = value + op if name != "*2" else value * op
candidates.append((nv, f"{trace} -> {name}={nv}"))
candidates.sort(key=lambda x: abs(target - x[0]))
beams = candidates[:width]
frame = pd.DataFrame(beams, columns=["value", "trace"])
frame["distance_to_target"] = (frame["value"] - target).abs()
return frame.sort_values("distance_to_target")
class _ResponseBandit:
"""
4-arm bandit: arm 0=weak(0.20), arm 1=mediocre(0.50),
arm 2=good(0.75), arm 3=excellent(0.92).
"""
TRUE_REWARDS = np.array([0.20, 0.50, 0.75, 0.92])
N_ARMS = 4
def __init__(self, noise: float = 0.04, seed: int = 5) -> None:
self.noise = noise
self.rng = np.random.default_rng(seed)
def pull(self, arm: int) -> float:
r = self.TRUE_REWARDS[arm] + self.rng.normal(0.0, self.noise)
return float(np.clip(r, 0.0, 1.0))
class GRPOTrainer:
"""
GRPO on a 4-arm bandit.
Each step: sample K candidates, score them, compute relative advantages,
apply PPO-clipped REINFORCE update to logits.
"""
K_GROUP = 4
EPS = 0.2
def __init__(self, bandit: _ResponseBandit,
lr: float = 0.15, seed: int = 42) -> None:
self.bandit = bandit
self.lr = lr
self.logits = np.zeros(bandit.N_ARMS)
self.rng = np.random.default_rng(seed)
def _probs(self) -> np.ndarray:
e = np.exp(self.logits - self.logits.max())
return e / e.sum()
def step(self) -> dict:
old_probs = self._probs()
arms = [int(self.rng.choice(self.bandit.N_ARMS, p=old_probs))
for _ in range(self.K_GROUP)]
rewards = [self.bandit.pull(a) for a in arms]
advantages = grpo_relative_advantages(rewards)
clip_count = 0
for arm, adv in zip(arms, advantages):
new_probs = self._probs()
ratio = new_probs[arm] / max(old_probs[arm], 1e-9)
clipped_ratio = np.clip(ratio, 1 - self.EPS, 1 + self.EPS)
use_clipped = (adv > 0 and ratio > 1 + self.EPS) or \
(adv < 0 and ratio < 1 - self.EPS)
clipped_adv = float(clipped_ratio * adv if use_clipped else ratio * adv)
if use_clipped:
clip_count += 1
grad = -new_probs.copy()
grad[arm] += 1.0
self.logits += self.lr * clipped_adv * grad
return {
"mean_reward": round(float(np.mean(rewards)), 3),
"best_arm_prob": round(float(self._probs()[-1]), 3),
"clip_frac": round(clip_count / self.K_GROUP, 2),
}
def train_grpo(n_steps: int = 100, seed: int = 42) -> pd.DataFrame:
"""Run GRPO on the response-quality bandit. Completes in < 0.1 s."""
bandit = _ResponseBandit(seed=seed)
trainer = GRPOTrainer(bandit=bandit, lr=0.15, seed=seed)
history = []
window = []
for step in range(1, n_steps + 1):
stats = trainer.step()
window.append(stats["mean_reward"])
if len(window) > 10:
window.pop(0)
history.append({
"step": step,
"mean_reward": stats["mean_reward"],
"best_arm_prob": stats["best_arm_prob"],
"clip_frac": stats["clip_frac"],
"reward_ma10": round(float(np.mean(window)), 3),
})
return pd.DataFrame(history)
# Run GRPO
df = train_grpo(n_steps=100)
print(f"Final best-arm probability: {df['best_arm_prob'].iloc[-1]:.1%}")
print(f"Clip fraction last 10 steps: {df['clip_frac'].tail(10).mean():.0%}")
print(df.tail(5)[["step", "mean_reward", "best_arm_prob", "clip_frac"]])
# Beam search reference
beams = beam_search_to_target(target=15, width=3, depth=4)
print("\nBeam search to reach 15:")
print(beams[["value", "distance_to_target", "trace"]].to_string(index=False))
When you run this code, you'll see that:
This is the essence of online learning: the agent becomes its own data engine.
The code above is adapted from the RL_Agents repository module:
git clone https://github.com/Pulkit12dhingra/RL_Agents
cd RL_Agents
uv sync
jupyter notebook notebooks/post05_online_learning.ipynb
STaR works because:
STaR has one major failure mode: distribution collapse. Here's the danger:
Suppose your model learns a narrow strategy that works for 60% of tasks. In iteration 1, you filter to those 60% and fine-tune. The model becomes overspecialized to that narrow strategy. Its success rate on the remaining 40% drops to 50%. In iteration 2, you filter to 30% of tasks (60% × 50%). Soon, the model only works on a tiny subset of the original distribution.
You've optimized yourself into a corner: high success on a narrow band, low success everywhere else.
Here's what a mature deployed agent system looks like:
User Query
↓
Model (inference, possibly with search)
↓
Output generated
↓
Verifier checks success/failure
↓
Trajectory logged to database
↓
[Async Process] Successful trajectories collected in batches
↓
Fine-tuning job runs nightly/weekly
↓
Updated model deployed
↓
(Loop)
The agent is learning continuously from real user interactions. No additional labeling required. The verifier (same one used for RLVR) automates the labeling.
In practice, online learning can yield dramatic improvements:
The key: this improvement comes entirely from the model learning its own successful traces, not from additional human annotation.
Online learning closes the loop:
This virtuous cycle is what powers the most capable agents. They're not static. They improve with every deployment day.