{"slug": "mole-detecting-insider-threats-in-ai-agents", "title": "MOLE: Detecting Insider Threats in AI Agents", "summary": "A new benchmark called MOLE tests whether defenders can detect insider threats—such as model misalignment, prompt injection, or operator misuse—among AI agents operating frontier-lab accounts, where malicious activity could exfiltrate model weights, poison training data, or weaken release gates. Existing benchmarks do not cover this detection scenario under limited review conditions.", "body_md": "Model misalignment, prompt injection, or operator misuse could lead AI agents operating frontier-lab accounts to exfiltrate model weights, poison training data, or weaken release gates. Existing benchmarks do not test whether defenders can detect this activity among routine work under a limited revi", "url": "https://wpnews.pro/news/mole-detecting-insider-threats-in-ai-agents", "canonical_source": "https://aiflash.com/news/116051/", "published_at": "2026-09-09 02:30:07+00:00", "updated_at": "2026-09-09 02:49:49.509882+00:00", "lang": "en", "topics": ["ai-safety", "ai-agents", "ai-research"], "entities": ["MOLE"], "alternates": {"html": "https://wpnews.pro/news/mole-detecting-insider-threats-in-ai-agents", "markdown": "https://wpnews.pro/news/mole-detecting-insider-threats-in-ai-agents.md", "text": "https://wpnews.pro/news/mole-detecting-insider-threats-in-ai-agents.txt", "jsonld": "https://wpnews.pro/news/mole-detecting-insider-threats-in-ai-agents.jsonld"}}