{"slug": "composite-bench", "title": "Composite-Bench", "summary": "Composite-Bench, a long-horizon browser-based computer-use benchmark built from deterministic enterprise-web environments, scores AI agents on pass@1 against certified optimal solutions across five headline slices and a sixth number-partitioning slice with a separate reward ladder. Claude Fable 5 ranks first with a score of 74 on the top imported result.", "body_md": "Long-horizon browser-based computer-use benchmark built from deterministic enterprise-web environments. Its five headline slices score pass@1 against certified optimal solutions, while a sixth number-partitioning slice reports a separate reward ladder for deliberately intractable tasks. Category: Agentic. Imported rows: 8. Top imported result: Claude Fable 5, rank 1, 74.", "url": "https://wpnews.pro/news/composite-bench", "canonical_source": "https://benchmarklist.com/benchmarks/composite_bench/", "published_at": "2026-07-24 00:00:00+00:00", "updated_at": "2026-07-24 00:37:44.297202+00:00", "lang": "en", "topics": ["ai-agents", "ai-research"], "entities": ["Composite-Bench", "Claude Fable 5"], "alternates": {"html": "https://wpnews.pro/news/composite-bench", "markdown": "https://wpnews.pro/news/composite-bench.md", "text": "https://wpnews.pro/news/composite-bench.txt", "jsonld": "https://wpnews.pro/news/composite-bench.jsonld"}}