{"slug": "on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics", "title": "On-Policy or Off-Policy Learning? A Systematic Study of Distillation Dynamics", "summary": "A systematic study of distillation dynamics finds that existing comparisons between supervised fine-tuning and reinforcement learning vary too many factors at once, making it impossible to isolate the contribution of rollout policy to on-policy learning's claimed benefits of reduced catastrophic forgetting, sparser parameter updates, and improved generalisation. The research examines on-policy versus off-policy learning to disentangle those factors.", "body_md": "On-policy learning has been argued to reduce catastrophic forgetting, produce sparser parameter updates, and improve generalisation. However, existing comparisons between supervised fine-tuning and reinforcement learning vary many factors simultaneously, making the contribution of rollout policy dif", "url": "https://wpnews.pro/news/on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics", "canonical_source": "https://aiflash.com/news/129930/", "published_at": "2026-10-02 16:30:01+00:00", "updated_at": "2026-10-02 16:37:12.101851+00:00", "lang": "en", "topics": ["machine-learning", "ai-research", "large-language-models", "artificial-intelligence"], "entities": [], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics", "markdown": "https://wpnews.pro/news/on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics.md", "text": "https://wpnews.pro/news/on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics.txt", "jsonld": "https://wpnews.pro/news/on-policy-or-off-policy-learning-a-systematic-study-of-distillation-dynamics.jsonld"}}