{"slug": "qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon", "title": "QwenGyre: An Elastic Reinforcement Learning Framework for Training xLong-Horizon Agents", "summary": "QwenGyre is an elastic reinforcement learning framework for training extreme-long-horizon LLM agents, where a single execution can span hours, hundreds of model-environment interactions, and nearly 1M tokens per rollout. The framework targets the two fundamental challenges of applying online reinforcement learning to such executions.", "body_md": "Large language model (LLM) agents increasingly undertake extreme-long (xlong) horizon tasks, where a single execution can span hours, hundreds of model--environment interactions, and nearly 1M tokens per rollout. Applying online reinforcement learning (RL) to such executions poses two fundamental ch", "url": "https://wpnews.pro/news/qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon", "canonical_source": "https://aiflash.com/news/128330/", "published_at": "2026-09-29 07:00:58+00:00", "updated_at": "2026-09-29 07:17:47.021058+00:00", "lang": "en", "topics": ["ai-agents", "large-language-models", "machine-learning", "ai-research", "artificial-intelligence"], "entities": ["QwenGyre"], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon", "markdown": "https://wpnews.pro/news/qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon.md", "text": "https://wpnews.pro/news/qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon.txt", "jsonld": "https://wpnews.pro/news/qwengyre-an-elastic-reinforcement-learning-framework-for-training-xlong-horizon.jsonld"}}