{"slug": "on-policy-parameter-update-direction-underlies-generalization-in-llm-post", "title": "On-Policy Parameter Update Direction Underlies Generalization in LLM Post-Training", "summary": "A study of on-policy post-training paradigms in large language models finds that the on-policy parameter update direction underlies generalization, arguing prior work treated these update behaviors only as byproducts rather than as optimization principles. The research examines parameter update behavior during on-policy post-training to explain the strong generalization these paradigms achieve.", "body_md": "The strong generalization performance of on-policy post-training paradigms has motivated studies of their parameter update behaviors. However, these studies treat the observed behaviors only as byproducts in on-policy training, overlooking their potential to serve as optimization principles for impr", "url": "https://wpnews.pro/news/on-policy-parameter-update-direction-underlies-generalization-in-llm-post", "canonical_source": "https://aiflash.com/news/131123/", "published_at": "2026-10-05 03:30:16+00:00", "updated_at": "2026-10-05 03:43:02.609681+00:00", "lang": "en", "topics": ["large-language-models", "machine-learning", "ai-research", "artificial-intelligence"], "entities": [], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/on-policy-parameter-update-direction-underlies-generalization-in-llm-post", "markdown": "https://wpnews.pro/news/on-policy-parameter-update-direction-underlies-generalization-in-llm-post.md", "text": "https://wpnews.pro/news/on-policy-parameter-update-direction-underlies-generalization-in-llm-post.txt", "jsonld": "https://wpnews.pro/news/on-policy-parameter-update-direction-underlies-generalization-in-llm-post.jsonld"}}