{
  "slug": "multi-step-autonomous-deep-research",
  "name": "Multi-step autonomous deep research",
  "tier": "bleeding-edge",
  "trend": "steady",
  "blockerType": null,
  "tools": [
    {
      "name": "Gemini Deep Research",
      "url": "https://gemini.google.com"
    },
    {
      "name": "Perplexity Pro",
      "url": "https://www.perplexity.ai"
    },
    {
      "name": "Perplexity Computer",
      "url": "https://www.perplexity.ai/computer"
    },
    {
      "name": "ChatGPT Deep Research",
      "url": "https://openai.com/chatgpt"
    },
    {
      "name": "Exa Agent",
      "url": null
    }
  ],
  "evidence": [
    {
      "title": "Interactions API | Gemini API | Google AI for Developers",
      "url": "https://ai.google.dev/gemini-api/docs/interactions-overview",
      "date": "2026-09-16",
      "type": "product-ga",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "Official Gemini API documentation confirming Deep Research agents (deep-research-preview-04-2026, deep-research-max-preview-04-2026) as GA callable agents with RBAC, multi-turn state, and observable execution steps."
    },
    {
      "title": "Gemini for Science Review: What Google Actually Built for Researchers",
      "url": "https://www.thesisai.io/blog/gemini-for-science-review/",
      "date": "2026-09-16",
      "type": "opinion",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "Critical opinion identifying unresolved limits in Gemini for Science autonomous research: citations require manual verification, agent reasoning trails unauditable, access life-sciences-only—constraining reproducibility despite 100+ institution partnerships."
    },
    {
      "title": "Your Agent Aced the Task. Will It Do It Again? Consistency Gaps in Multi-Step Autonomous Agents",
      "url": "https://huggingface.co/blog/ibm-research/altk-evolve-consistency",
      "date": "2026-09-15",
      "type": "research-paper",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "IBM Research measurement of 24.4-percentage-point consistency gap in multi-step ReAct agents (Mean@5 77.4% vs Pass^5 53.0%), halved to 12.0pp by injected guidelines—variance orthogonal to single-run capability."
    },
    {
      "title": "AI Deep Research: Codex vs Claude vs Grok vs Exa—Independent DR-20 Benchmark",
      "url": "https://aimultiple.com/ai-deep-research",
      "date": "2026-09-12",
      "type": "opinion",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "Third-party benchmark of five research tools across 20 business briefs: purpose-built Exa Agent (0.584) underperformed three coding-capable LLMs; no tool achieved 50% requirement coverage, with per-task cost and citation accuracy data."
    },
    {
      "title": "Leveraging prompt-driven generative AI for systematic reviews: stage-matched comparative proof-of-concept",
      "url": "https://journals.plos.org/digitalhealth/article?id=10.1371/journal.pdig.0001666",
      "date": "2026-09-10",
      "type": "research-paper",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed comparison of GPT-5 Auto, Agent and Deep Research modes across nine systematic review tasks; found task-specific strengths (synthesis best for Deep Research), zero fabricated content, limits on numerical extraction."
    },
    {
      "title": "OpenResearcher: a reproducible and scalable pipeline for training deep research agents",
      "url": "https://lambda.ai/blog/openresearcher-training-research-agents-at-scale",
      "date": "2026-09-10",
      "type": "case-study",
      "added": "2026-09-23",
      "superseded_by": null,
      "window": null,
      "explanation": "Training pipeline from Texas A&M, Waterloo, UCSD and Lambda distilling 97k synthetic trajectories into Nemotron-3-Nano scoring 54.8% on BrowseComp-Plus, exceeding GPT-4.1; adopted downstream by NVIDIA for SFT."
    },
    {
      "title": "Gemini 3.8 Flash Hallucination Rate, Measured | TrueStandard",
      "url": "https://truestandard.ai/releases/gemini-3-8-flash",
      "date": "2026-09-03",
      "type": "research-paper",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "TrueStandard measurement: Gemini 3.8 Flash regressed on citation fabrication (20% vs 12.6% prior version) across 30 ungrounded claims—shows newer model versions can degrade reliability without vendor detection."
    },
    {
      "title": "Use the Gemini Deep Research Agent",
      "url": "https://docs.cloud.google.com/gemini-enterprise-agent-platform/agents/use-deep-research",
      "date": "2026-09-02",
      "type": "product-ga",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Google Cloud's official managed AI agent platform documentation confirms Gemini Deep Research Agent as production service for multi-step autonomous research across public web and enterprise data with MCP integration."
    },
    {
      "title": "A third of Perplexity's citations don't contain the number ...",
      "url": "https://hausresearch.com/reports/perplexity-citation-audit/",
      "date": "2026-09-02",
      "type": "research-paper",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Haus Research audit: 34.7% of Perplexity citations failed verification on 310 factual questions; 16.1% of accessible pages contained none of the cited figures—critical evidence of autonomous research reliability barriers."
    },
    {
      "title": "A Demo Vendor Is Perplexity's #3 Source. Rank the Domains.",
      "url": "https://www.beri.net/article/perplexity-cited-domains-tranco-rank-vendor-longlist-provenance-gate",
      "date": "2026-09-02",
      "type": "industry-report",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "BERI/Trellner analysis: 59.8% of Perplexity citations from outside Tranco top 100k domains; three coordinated fake software-guide sites with 215k machine-generated pages—shows autonomous research vulnerable to source pollution."
    },
    {
      "title": "Hybrid AI and Confidential Data: Keep Client Files Off the Cloud",
      "url": "https://cloudradix.com/blog/hybrid-ai-confidential-data-off-cloud-fort-wayne-2026/",
      "date": "2026-09-01",
      "type": "product-ga",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity Hybrid Compute (Sept 1 GA) enables single autonomous agent to divide multi-step research between cloud frontier models and local-device models, addressing production deployment barrier of data residency."
    },
    {
      "title": "Pythian's AI Model Via Google Gemini Drives 'Million-Dollar Outcomes' For Customers",
      "url": "https://www.crn.com/news/ai/2026/pythian-s-ai-model-via-google-gemini-drives-million-dollar-outcomes-for-customers",
      "date": "2026-09-01",
      "type": "case-study",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Pythian (500-person company) deployed Gemini Enterprise across 27 countries with autonomous multi-step workflows achieving 3× user engagement increase and 80% MTTR reduction on 15k monthly database tickets."
    },
    {
      "title": "Deep Research Tools Compared: Citation Accuracy Audited",
      "url": "https://zandigital.in/deep-research-tools-hallucination-audit",
      "date": "2026-08-31",
      "type": "opinion",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Zan Digital audit of 5 production deep research agents: citation accuracy 77.96%–93.68%; Perplexity 90.24% vs Gemini 81.44%—reveals vendor strategy of marketing citation volume over precision, masking quality gaps."
    },
    {
      "title": "Google takes Gemini AI into finance, legal sectors",
      "url": "https://www.itweb.co.za/article/google-takes-gemini-ai-into-finance-legal-sectors/kYbe97XbZZJqAWpG",
      "date": "2026-08-27",
      "type": "product-ga",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Google Cloud ships production Gemini Enterprise with managed financial research agent; live deployments (CME Group, Deutsche Bank) conducting autonomous multi-step research with 13 regulated-industry data connectors and audit trails."
    },
    {
      "title": "Google, Weil + Gemini Enterprise for Legal",
      "url": "https://www.artificiallawyer.com/2026/08/27/google-weil-gemini-enterprise-for-legal/",
      "date": "2026-08-27",
      "type": "case-study",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Weil Gotshal co-developed parallel research agents for multi-step legal investigation and NDA drafting workflows, confirming autonomous research agents operate agentic without continuous human redirection in regulated sectors."
    },
    {
      "title": "Banking on agents: These 8 firms are building the future of financial services",
      "url": "https://cloud.google.com/transform/financial-services-ai-agents-gemini-enterprise-round-up",
      "date": "2026-08-27",
      "type": "case-study",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Eight named financial orgs in production: Core AI 25-day onboarding vs 300-day average, Dojo 70% chargeback time reduction, SIGNAL IDUNA 400% usage surge, AXA detected 120k CHF fraud in first week—measured autonomous research impact."
    },
    {
      "title": "Gemini AI Credit Risk Advances in Deutsche Bank Deployment",
      "url": "https://en.cryptonomist.ch/2026/08/26/gemini-ai-credit-risk-deutsche-bank/",
      "date": "2026-08-26",
      "type": "case-study",
      "added": "2026-09-09",
      "superseded_by": null,
      "window": null,
      "explanation": "Deutsche Bank deployed Gemini Enterprise financial research agent conducting autonomous credit risk analysis, reducing bond portfolio analysis from days to under 5 minutes with 50+ specialized skills and 13 data connectors."
    },
    {
      "title": "Perplexity Ships Portable Computer on NVIDIA DGX Spark: Local Harness, OS-Enforced Sandbox, and Zero Per-Token Cost",
      "url": "https://www.marktechpost.com/2026/08/25/perplexity-ships-portable-computer-on-nvidia-dgx-spark-local-harness-os-enforced-sandbox-and-zero-per-token-cost-for-local-steps/",
      "date": "2026-08-25",
      "type": "product-ga",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity released local-first multi-step agent deployment on NVIDIA DGX Spark with deterministic harness, reactive escalation to cloud models, and hybrid cost optimization (85.4% local-only accuracy, 73% with adviser escalation at $0.415/rollout vs $0.65 Opus-only)."
    },
    {
      "title": "Anthropic ships production-ready AI agent tools",
      "url": "https://agentry.news/agent/anthropic-ships-production-ready-ai-agent-tools",
      "date": "2026-08-24",
      "type": "product-ga",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Anthropic moved computer use, browser use, Skills API, and Files API to GA (August 20, 2026), removing experimental label and enabling production deployment of autonomous research workflows with web navigation and document processing at scale."
    },
    {
      "title": "LLM Hallucination Rate by Task Type (2026 Data) - Inferya",
      "url": "https://inferya.com/guides/llm-hallucination-rate-by-task-type/",
      "date": "2026-08-21",
      "type": "tutorial",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Comparative analysis of hallucination rates across task types (3.3% to 88%) reveals engineering implication: ungrounded knowledge retrieval predicts 35-88% baseline error rates for autonomous research agents without retrieval grounding as mitigation."
    },
    {
      "title": "Agent Reliability Benchmarks: Microsoft Thinkingbox Sandbox + 507 Real Workflows",
      "url": "https://buttondown.com/patricknovak1/archive/models-agents-ep148-agent-reliability-benchmarks-just-expose/",
      "date": "2026-08-21",
      "type": "product-ga",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Microsoft Thinkingbox evaluated agent reliability on 507 stateful workflows; strongest model 65.36% pass@1 but only 25.25% pass@20, revealing multi-step execution reliability gap despite intermediate state correctness—direct evidence of compounding failure in deep research workflows."
    },
    {
      "title": "Perplexity Comet in the Enterprise: The Decision",
      "url": "https://accuroai.co/blog/perplexity-comet-in-the-enterprise-sanction-restrict-block",
      "date": "2026-08-19",
      "type": "opinion",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Security and deployment assessment documents fundamental vulnerabilities (prompt injection, zero-click Intent Collision attacks) blocking production deployment; concluded: block by default; Gartner recommends agent browser blocks until injection resistance demonstrably solved, not patched."
    },
    {
      "title": "50+ Gemini Statistics 2026: Users, Revenue & Market Share",
      "url": "https://successpixel.com/gemini-statistics/",
      "date": "2026-08-18",
      "type": "adoption-metric",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Gemini reached 1 billion monthly active users (Aug 11, 2026) with 90% Fortune 100 enterprise adoption and 22 billion API tokens/min infrastructure scale; Deep Research fully integrated into Workspace, validating mainstream deployment of agentic research platforms."
    },
    {
      "title": "Anthropic Risk Report: August 2026",
      "url": "https://thezvi.substack.com/p/anthropic-risk-report-august-2026",
      "date": "2026-08-18",
      "type": "opinion",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Analysis of Anthropic's internal risk assessment: Model 2 achieves 62.8% on researcher-substitution task (37.2% failure); model restricted to internal-only deployment signaling organizational caution against autonomous research agent deployment despite frontier capability."
    },
    {
      "title": "Study Points to Limits of AI in Autonomous Scientific Research: Princeton and UK AISI Empirical Evaluation",
      "url": "https://www.radardigital.ai/en/articles/study-points-to-limits-of-ai-in-autonomous-scientific-research",
      "date": "2026-08-16",
      "type": "case-study",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Princeton + UK AISI Shadow Evaluation: frontier models (Claude 4.8, GPT-5.6 Sol) rejected on unpublished NeurIPS papers by original authors; both exhibited weak experimental motivation, unjustified methodology, and zero novel contributions—contradicts vendor claims of autonomous research capability."
    },
    {
      "title": "OpenRouter Web Search Benchmarks: Which Engine Wins on Agent Grounding",
      "url": "https://explainx.ai/blog/openrouter-web-search-benchmarks-agent-grounding-august-2026",
      "date": "2026-08-15",
      "type": "adoption-metric",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Benchmark comparing models and search engines on multi-step research showing search budget as largest performance lever—increasing from 1 to 25 search turns roughly doubled BrowseComp scores, larger than model swaps (~15pp) or engine swaps (~10pp)."
    },
    {
      "title": "How Do Agents Fail on AutoResearch: End-to-End Diagnostic Evaluation on 100 Real-World Frontier Research Tasks",
      "url": "https://arxiv.org/html/2608.14905v1",
      "date": "2026-08-14",
      "type": "research-paper",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed diagnostic evaluation of 8 harness-model combinations on 100 end-to-end research tasks; introduces ARFT failure taxonomy (45 patterns). Critical finding: all agents lack metacognitive loop—ability to verify output against sources, revise, and self-question."
    },
    {
      "title": "WANDR: A Benchmark for Wide and Deep Research",
      "url": "https://www.alphaxiv.org/abs/2608.14747",
      "date": "2026-08-14",
      "type": "research-paper",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "500-task benchmark from professional workflows (market mapping, due diligence, literature review); strongest production system reaches only 0.363 soft F1 and 0.133 hard F1, demonstrating autonomous deep research remains far from saturation despite vendor maturity claims."
    },
    {
      "title": "AI Agent Adoption Statistics (Updated Q3 2026) | Brad McAllister",
      "url": "https://www.linkedin.com/posts/bradmca_ai-agent-adoption-statistics-updated-q3-activity-7494028848167145472-gzXA",
      "date": "2026-08-14",
      "type": "adoption-metric",
      "added": "2026-08-26",
      "superseded_by": null,
      "window": null,
      "explanation": "Critical adoption gap: only 9% of enterprises making progress on autonomous multistep workflows vs 59% claiming agent use, revealing that deep research—requiring extended autonomous chains—remains constrained by organizational workflow maturity."
    },
    {
      "title": "Tech Mahindra To Deploy Perplexity Enterprise Pro",
      "url": "https://www.fintechbiznews.com/fintech-technology/tech-mahindra-to-deploy-perplexity-enterprise-pro",
      "date": "2026-08-11",
      "type": "case-study",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Named enterprise (Tech Mahindra) deploying Perplexity for autonomous sales research, enabling real-time trusted intelligence gathering across customer and market domains."
    },
    {
      "title": "Can generative AI reliably synthesise literature? exploring hallucination issues in ChatGPT",
      "url": "https://ai-in-research.livingmeta.ai/papers/W4411506008",
      "date": "2026-08-06",
      "type": "research-paper",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed PRISMA systematic review (124 studies) quantifies hallucination and accuracy tradeoffs in AI-assisted literature synthesis. Sensitivity 80.6–96.5% in screening but precision drops to 4.6% in interpretive tasks; hallucination rates 28–91%."
    },
    {
      "title": "Cited but Not Verified: Parsing and Evaluating Source Attribution in LLM Deep Research Agents",
      "url": "https://ai-in-research.livingmeta.ai/papers/W7160727550",
      "date": "2026-08-06",
      "type": "research-paper",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Empirical evaluation of 14 LLMs on deep research tasks, revealing critical disconnect: surface-level citation quality (94% link validity, 80% relevance) masks factual accuracy collapse (39–77%), and information overload degrades accuracy."
    },
    {
      "title": "Do Deployment Constraints Make LLMs Hallucinate Citations? An Empirical Study across Four Models and Five Prompting Regimes",
      "url": "https://ai-in-research.livingmeta.ai/papers/W7134859822",
      "date": "2026-08-06",
      "type": "research-paper",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Empirical evidence of fundamental deployment barriers: no LLM model under any condition achieves >50% citation existence rate; temporal constraints reduce verifiability most severely."
    },
    {
      "title": "EviGraph: Evidence-Guided Autonomous Research Agents",
      "url": "https://arxiv.org/abs/2608.04738",
      "date": "2026-08-05",
      "type": "research-paper",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Framework maintaining explicit evidence graph across research stages (Problem, Gap, Hypothesis, Experiment, Finding, Claim); detects and repairs inconsistencies; shows 40% improvement in claim support rate and 87.73% data consistency."
    },
    {
      "title": "State of Enterprise AI 2026 | NexaWorks Research",
      "url": "https://nexaworks.tech/research/state-of-enterprise-ai-2026",
      "date": "2026-08-05",
      "type": "adoption-metric",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Industry research report with specific adoption and failure metrics: 42% search traffic migration to AI answer engines (Perplexity, Copilot, ChatGPT), 78% enterprise RAG pipeline production failure rate. Documents ecosystem shift toward deterministic validation."
    },
    {
      "title": "Closing the Loop: How Scientists Are Teaching AIs to Keep Their Promises",
      "url": "https://akmaier.substack.com/p/closing-the-loop-how-scientists-are",
      "date": "2026-08-03",
      "type": "case-study",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Case study of Google Research's Science One Framework deployed on autonomous research agents, achieving zero phantom references with competitive performance, demonstrating viability of verified deep research."
    },
    {
      "title": "How To Run A Systematic Review With ChatGPT & Scite MCP",
      "url": "https://www.linkedin.com/pulse/how-run-systematic-review-chatgpt-scite-mcp-sciteai-whiuc",
      "date": "2026-07-29",
      "type": "case-study",
      "added": "2026-08-12",
      "superseded_by": null,
      "window": null,
      "explanation": "Real research workflow deployment (systematic review) with identified failure mode (hallucination/citation verification) and solution development (Scite MCP integration for verified sources)."
    },
    {
      "title": "SciExplore: Evaluating Autonomous Agents from Scientific Navigation to Information Integration",
      "url": "https://arxiv.org/html/2607.20926v1",
      "date": "2026-07-24",
      "type": "research-paper",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed benchmark evaluating autonomous agents on multi-step scientific research (database navigation, literature retrieval, reference completion, synthesis); reports <50% success rates with sharp degradation at complexity."
    },
    {
      "title": "The Deep Research Stack Bain Won't Publish: Chaining Perplexity, Glean & Claude for Consulting-Grade Output",
      "url": "https://www.promptspace.in/zh/blog/deep-research-stack-bain-perplexity-glean-claude",
      "date": "2026-07-24",
      "type": "case-study",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Production workflow at consulting firm scale: Perplexity→Glean→Claude→NotebookLM compressed 6-12 hours to 25 minutes; $1,200–$3,000 margin recovered per engagement, with MBB hiring push validating adoption."
    },
    {
      "title": "Why 88% of AI Agent Pilots Never Reach Production in 2026",
      "url": "https://www.fatherofai.in/blog/agentic-ai-production-reliability-reckoning-2026/",
      "date": "2026-07-24",
      "type": "industry-report",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Gartner data showing 88% of enterprise agent pilots fail to reach production; identifies non-determinism, evaluation, observability, and identity governance as critical blockers affecting all agent classes including research."
    },
    {
      "title": "Is Deep Research Reliable? Misleading Knowledge Induces False Conclusions",
      "url": "https://arxiv.org/abs/2607.20891",
      "date": "2026-07-23",
      "type": "research-paper",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Peer-reviewed study of Deep Research agents shows misleading-info adoption rates 34.5% (cold start) to 85% (pre-synthesis); verifiers fail in workflow context despite catching misinformation in isolation."
    },
    {
      "title": "Autonomy Is the Bug: Why Self-Driving Agents Hallucinate When the Model Barely Does",
      "url": "https://dev.to/p0rt/autonomy-is-the-bug-why-self-driving-agents-hallucinate-when-the-model-barely-does-1330",
      "date": "2026-07-21",
      "type": "opinion",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Technical analysis identifying structural failure mechanism: error compounding via Lusser's law (95% per-step accuracy yields 36% success on 20-step tasks); self-conditioning amplifies failures independent of model quality."
    },
    {
      "title": "How to Produce World-Class AI Research With Perplexity and Claude Fable 5",
      "url": "https://www.operatingbyjohnbrewton.com/p/how-to-produce-world-class-research",
      "date": "2026-07-19",
      "type": "case-study",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Production case study documenting full workflow: Perplexity Computer (33 min, 140 agent tasks) for research, Claude Opus 4.8 verification; audit revealed 5 substantive errors, proving capability and critical verification requirement."
    },
    {
      "title": "Deep Research: Why AI Reports Look True but Aren't",
      "url": "https://silentroom.media/the-machine/deep-research-slow-expensive-dangerous",
      "date": "2026-07-19",
      "type": "opinion",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Comprehensive critical analysis of deep research across all major platforms (Google, OpenAI, Perplexity, Anthropic, Copilot); documents three core failure modes: fabrication, inability to admit uncertainty, source-framing bias with real-world example of hallucinated regulatory clause."
    },
    {
      "title": "How to Fact-Check AI Research Tools: What Current Error Rates Actually Require",
      "url": "https://trendquotient.com/technology/ai-productivity/how-to-fact-check-ai-research-tools/",
      "date": "2026-07-18",
      "type": "opinion",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner audit data on production deep research tools: Perplexity 37% error rate on article retrieval; citations increase trust even when fabricated, creating verification burden in deployed workflows."
    },
    {
      "title": "Perplexity launches secure sandbox to make its AI agents secure and powerful",
      "url": "https://siliconangle.com/2026/07/15/perplexity-launches-secure-sandbox-to-make-its-ai-agents-secure-powerful/",
      "date": "2026-07-15",
      "type": "product-ga",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "Production deployment metrics: 1.25M sandbox creations and 11.9M reconnects per week, showing Perplexity Computer autonomous agents at scale with infrastructure supporting long-running workflows."
    },
    {
      "title": "Enterprise AI Agents Stall on Deployment as Most 'Agents' Remain Chatbots",
      "url": "https://business20channel.tv/enterprise-ai-agents-stall-on-deployment-as-ml-ambition-outruns-reality-in-15-07-2026",
      "date": "2026-07-15",
      "type": "adoption-metric",
      "added": "2026-07-29",
      "superseded_by": null,
      "window": null,
      "explanation": "VentureBeat survey of 101 enterprises: 71% of deployed 'agents' cannot complete autonomous multi-step work autonomously, remaining chatbots; directly evidences the capability maturity gap deep research agents must cross."
    },
    {
      "title": "How AI Agents Reshape Knowledge Work",
      "url": "https://research.perplexity.ai/articles/how-ai-agents-reshape-knowledge-work",
      "date": "2026-07-10",
      "type": "case-study",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "Harvard Business School + Perplexity empirical study on production data: Perplexity Computer achieved 26-minute autonomous execution per session vs 33 seconds for Search (48× autonomy increase), 87% task time reduction, 84× growth in task volume—quantifies deployment scale."
    },
    {
      "title": "Evaluating Deep Research Performance in the Wild with the DRACO Benchmark",
      "url": "https://research.perplexity.ai/articles/evaluating-deep-research-performance-in-the-wild-with-the-draco-benchmark",
      "date": "2026-07-10",
      "type": "research-paper",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity's DRACO benchmark (100 production-grounded tasks, 10 expert domains): Claude Mythos 5 at 86.4%, Gemini 59.0%, OpenAI o3 52.1%—first standardized evaluation framework for deep research agents with all major vendors competing."
    },
    {
      "title": "AI Agents Struggle with Context Accuracy in New Study",
      "url": "https://www.tickrwire.tech/article/ai-agents-struggle-with-context-accuracy-in-new-study",
      "date": "2026-07-10",
      "type": "adoption-metric",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "57% of autonomous agents fail to maintain context during multi-step operations. Failure mode: context loss during step transitions—documents specific architectural gap affecting deep research's ability to retain information across research phases."
    },
    {
      "title": "Do You Need a Frontier Model as a Citation Verifier? Benchmarking Rubric LLMs for Deep-Research Source Attribution",
      "url": "https://arxiv.org/html/2607.08700v1",
      "date": "2026-07-09",
      "type": "research-paper",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "PWC peer-reviewed benchmark of 8 LLM judges on citation quality (1,248 decisions): 'cheaper judges competitive' on relevance (F1 0.908), but all degrade on hard cases—identifies calibration prerequisite for citation-quality reward signals in deep research."
    },
    {
      "title": "Perplexity Ships GLM 5.2 Orchestrator That Matches Claude Opus at One-Third the Cost",
      "url": "https://alphasignal.ai/news/perplexity-ships-glm-5-2-orchestrator-that-matches-claude-opus-at-one-third-the",
      "date": "2026-07-09",
      "type": "industry-report",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "AlphaSignal analysis: GLM 5.2 (744B MoE, 40B active) delivers frontier agent performance at one-third Claude Opus cost through specialized post-training on long-horizon tasks—validates cost-efficiency breakthrough in orchestration."
    },
    {
      "title": "Your AI Agents Are Failing 70% of the Time. Here's the Fix.",
      "url": "https://www.beri.net/article/patronus-ai-50m-enterprise-agent-testing-production-failure-2026",
      "date": "2026-07-07",
      "type": "case-study",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "Patronus AI/Fiddler 2026 analysis: agents fail 70-95% in production despite 90%+ benchmark accuracy; 30-50% accuracy when tasks chained (compounding errors) vs 90%+ isolated—documents benchmark-to-production reliability cliff."
    },
    {
      "title": "New Research: Why Enterprise Agentic AI Stalls Before It Scales",
      "url": "https://markets.ft.com/data/announce/detail?dockey=600-202607070830PR_NEWS_USPRX____CL98747-1",
      "date": "2026-07-07",
      "type": "industry-report",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "Teradata/Wakefield study of 1,000 tech leaders: only 7% operationalizing autonomous workflows. Root barrier: context fragmentation (77% insufficient data governance)—identifies systemic organizational constraint preventing deep research scaling."
    },
    {
      "title": "Deep Research agents in production 2026: from OpenAI, Anthropic, Google, and Perplexity to your own open-source stack",
      "url": "https://www.reactify-solutions.com/articles/deep-research-agents-production-2026",
      "date": "2026-07-04",
      "type": "industry-report",
      "added": "2026-07-15",
      "superseded_by": null,
      "window": null,
      "explanation": "Reactify technical analysis: all major vendors converged on three-phase scope/research/write pipeline; multi-agent orchestration with 2-10+ sub-agents per task—validates architectural pattern maturation across independent implementations."
    },
    {
      "title": "How Parameterized World Models Reduce Hallucination Propagation in LLM Agents",
      "url": "https://arxiv.org/html/2606.27806v1",
      "date": "2026-06-25",
      "type": "research-paper",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "GILP method reduces hallucinated-state rate from 17.6% to 3.5% in multi-step agents; directly addresses hallucination propagation in autonomous research workflows—a core failure mode constraining deep research reliability."
    },
    {
      "title": "Perplexity Computer: The $200 AI Agent That Rewrites the Bloomberg Terminal (2026)",
      "url": "https://pasqualepillitteri.it/en/news/1983/perplexity-computer-bloomberg-terminal-finance-2026",
      "date": "2026-06-24",
      "type": "case-study",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "Deep technical analysis of Perplexity Computer's 19-model architecture (Opus reasoning, Gemini web research, GPT-5.2 long-context, Grok lightweight) with Firecracker microVM isolation and financial research use case, demonstrating orchestration enabling end-to-end autonomous delivery from natural language."
    },
    {
      "title": "DRACO Leaderboard 2026: Latest Deep Research Agent Scores",
      "url": "https://leaderboard.steel.dev/leaderboards/draco/",
      "date": "2026-06-23",
      "type": "adoption-metric",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity's standardized 100-task deep research benchmark across 10 expert domains shows 5+ vendors competing: Claude Mythos 5 leads at 86.4%, Gemini at 59.0%, OpenAI o3 at 52.1%—first production benchmark designed for deep research system evaluation with all major vendors shipping competing systems."
    },
    {
      "title": "Gemini Docs changelog - Jun 22, 16:04 UTC | Tech Dev Notes",
      "url": "https://techdevnotes.com/releases/gemini-docs/20260622-160420Z-f2b0f8e7531e",
      "date": "2026-06-22",
      "type": "product-ga",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "Google publishes Gemini Deep Research Agent documentation covering autonomous multi-step research, MCP server integration, visualizations, and background execution with code samples in Python, JavaScript, REST—signals production-ready ecosystem maturity with multi-language SDK support."
    },
    {
      "title": "Philippines DICT puts Gemini Enterprise in 50,000 government desks",
      "url": "https://aintelligencehub.com/articles/philippines-dict-gemini",
      "date": "2026-06-22",
      "type": "case-study",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "Large-scale government deployment: Philippines DICT rolling Gemini Enterprise to 50K public servants scaling to 200K over 18 months with tracked KPIs; named use cases (permit synthesis, research briefs) demonstrate autonomous research for citizen services at geographic scale."
    },
    {
      "title": "Deep Research Agent Benchmarks Hide Deployment-Reality Gaps: DRFLOW Analysis",
      "url": "https://groundy.com/articles/deep-research-benchmarks-hide-how-agents-fail-at-open-web-source-grounding/",
      "date": "2026-06-21",
      "type": "opinion",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "DRFLOW benchmark reveals agents score well on curated-corpus retrieval but fail at open-web source discovery, filtering, grounding, and workflow sequencing; diagnostic metrics show agents struggle with factual grounding, step recovery, structural ordering—highlighting benchmark-to-deployment prediction gap."
    },
    {
      "title": "Gemini API by Google - Release Notes",
      "url": "https://releasebot.io/updates/google/gemini-api",
      "date": "2026-06-17",
      "type": "product-ga",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "Gemini API release notes document GA of two Deep Research agent variants with collaborative planning, visualization, MCP integration, and File Search: speed-optimized vs comprehensive-research tiers signal market segmentation for deployed agents."
    },
    {
      "title": "Deep Research in Physical Sciences: A Multi-Agent Framework and Comprehensive Benchmark",
      "url": "https://arxiv.org/abs/2606.18648v2",
      "date": "2026-06-17",
      "type": "research-paper",
      "added": "2026-07-01",
      "superseded_by": null,
      "window": null,
      "explanation": "PhySciBench peer-reviewed benchmark tests deep research agents on 200 physics/chemistry questions; Gemini Deep Research achieves 33.5% accuracy with identified failure modes; DelveAgent proposes multi-agent framework improving 7.5pp while reducing cost to one-third—reveals domain-specific capability boundaries."
    },
    {
      "title": "Perplexity Moves Deep Research Into Computer, Routing Research Subtasks Across 20+ Frontier Models",
      "url": "https://www.marktechpost.com/2026/06/11/perplexity-moves-deep-research-into-computer-routing-research-subtasks-across-20-frontier-models-for-reports-decks-and-dashboards/",
      "date": "2026-06-11",
      "type": "product-ga",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity Computer integration: BrowseComp +43pp (40.7%→83.8%), Humanity's Last Exam +14pp; 'Search as Code' paradigm with parallel retrieval/filtering; rolling to Max tier and Agent API—multi-model orchestration reaching production scale with measured benchmark improvements."
    },
    {
      "title": "Benchmarking AI Agents for Addressing Scientific Challenges Across Scales",
      "url": "https://huggingface.co/papers/2606.12736",
      "date": "2026-06-10",
      "type": "research-paper",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "SciAgentArena benchmark (Stanford/MIT/Harvard): agents effective on well-specified data workflows but struggle with multi-constraint optimization, novel insight generation, and unsupported claim detection—defining boundaries of autonomous research capability."
    },
    {
      "title": "arXiv Tightens Policy on Hallucinated References: What Researchers Should Know About Using AI Search Engines",
      "url": "https://library.smu.edu.sg/topics-insights/arxiv-tightens-policy-hallucinated-references-what-researchers-should-know-about",
      "date": "2026-06-10",
      "type": "industry-report",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Zhao et al. audited 2.5M arXiv/bioRxiv/PubMed papers: 146,932 hallucinated citations identified in 2025; rate rose from 1/2,828 papers (2023) to 1/277 (early 2026); demonstrates widespread deployment of deep research tools at scale with endemic citation failure modes."
    },
    {
      "title": "ResearchClawBench: A Benchmark for End-to-End Autonomous Research Agent Evaluation",
      "url": "https://huggingface.co/papers/2606.07591",
      "date": "2026-06-09",
      "type": "research-paper",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "First rigorous benchmark: Claude Code achieved only 21.5% on 40 real scientific re-discovery tasks; error modes (experimental mismatch, evidence gaps, missing core) concentrated; reveals critical gap between market adoption and measured research agent reliability."
    },
    {
      "title": "A New Study from Harvard and Perplexity Finds AI Agents Perform 26 Minutes of Autonomous Work per Session vs 33 Seconds for Search",
      "url": "https://www.marktechpost.com/2026/06/08/a-new-study-from-harvard-and-perplexity-finds-ai-agents-perform-26-minutes-of-autonomous-work-per-session-vs-33-seconds-for-search/",
      "date": "2026-06-08",
      "type": "case-study",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Production study (Feb-May 2026): Perplexity Computer achieved 26-minute autonomous execution per session vs 33 seconds for Search (48× increase); 87% task time reduction, 94% cost savings on matched 10k sessions; 23% novel task expansion showing scope amplification."
    },
    {
      "title": "The Agentwashing Crisis: Why 79% of Enterprises Claim AI Agents But Only 11% Ship to Production",
      "url": "https://agentmarketcap.ai/blog/2026/06/07/agentwashing-crisis-enterprise-ai-agents-2026",
      "date": "2026-06-07",
      "type": "adoption-metric",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Adoption-to-production gap: 79% claim agents, 11% in production; 88% of pilots fail; Gartner forecasts 40%+ cancellation by 2027; successful deployments (Klarna $60M, Salesforce 380k interactions) follow narrow scope + measurable output—critical scaling barrier for deep research."
    },
    {
      "title": "CHARM: The Missing Layer in Agentic AI Multi-Step Failure Propagation",
      "url": "https://www.linkedin.com/pulse/your-hallucination-detector-has-blind-spot-size-entire-travis-lelle-exkte",
      "date": "2026-06-05",
      "type": "research-paper",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "CHARM framework (arXiv June 3, 2026) formalizes cascading hallucinations in multi-step RAG pipelines; existing detectors catch only 12.8–41.7% of failures; LLM self-correction counterproductive (12.8% detection); identifies fundamental pipeline reliability barrier for deep research."
    },
    {
      "title": "Autonomous KI-Forschungsagenten auf Gemini 3.1 Pro (2026)",
      "url": "https://pasqualepillitteri.it/de/news/1194/google-deep-research-max-gemini-3-1-pro-ki-agenten",
      "date": "2026-06-04",
      "type": "tutorial",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Deep Research Max API benchmarks: 93.3% on DeepSearchQA (vs 66.1% Dec 2025, +41.3pp gain); MCP support for proprietary data; async background workflows up to 60 minutes; API GA (April 21, 2026)—technical capability maturation documented."
    },
    {
      "title": "The State Of Agentic AI In 2026: Companies Are Chasing, Few Are Catching",
      "url": "https://www.forrester.com/blogs/the-state-of-agentic-ai-in-2026-companies-are-chasing-few-are-catching/",
      "date": "2026-06-03",
      "type": "industry-report",
      "added": "2026-06-17",
      "superseded_by": null,
      "window": null,
      "explanation": "Forrester analyst report explicitly citing 'Anthropic has demonstrated multiday research agents' as bleeding-edge example of long-horizon autonomous operation; directly validates multiday research capability at frontier, distinguishing research agents from shorter-horizon task agents."
    },
    {
      "title": "The Model Isn't the Agent Anymore",
      "url": "https://alphasignalai.substack.com/p/the-model-isnt-the-agent-anymore",
      "date": "2026-05-28",
      "type": "research-paper",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "UC Berkeley research shows agent performance depends equally on orchestration, memory, context governance, skill routing, and grading—not just model capability; multi-agent orchestration gains 90.2% improvement with token usage explaining 80% of variance."
    },
    {
      "title": "Multi-Model AI Divergence Index Q1 2026, The Confidence Trap",
      "url": "https://suprmind.ai/hub/multi-model-ai-divergence-index/",
      "date": "2026-05-28",
      "type": "adoption-metric",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Production data from 1,324 real-user multi-model turns: 99.1% contain contradictions across models, 51.3% of Gemini's high-confidence answers contradicted by peers; validates multi-model verification as essential architecture for autonomous research reliability."
    },
    {
      "title": "WildClawBench: A Benchmark for Real-World, Long-Horizon Agent Evaluation",
      "url": "https://chatpaper.com/chatpaper/paper/278928",
      "date": "2026-05-26",
      "type": "research-paper",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Native-runtime benchmark of 60 long-horizon multimodal tasks (8-minute wall-clock, 20+ tool calls each) shows best model reaches only 62.2% accuracy; harness architecture shifts performance up to 18 points, demonstrating orchestration matters as much as models."
    },
    {
      "title": "AutoExperiment: Testing AI Agents on Research Replication",
      "url": "https://dev.to/jangwook_kim_e31e7291ad98/autoexperiment-testing-ai-agents-on-research-replication-31a8",
      "date": "2026-05-24",
      "type": "research-paper",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Carnegie Mellon benchmark: frontier agents achieve 30-37% on single-function research tasks but collapse to 6-10% when functions have dependencies, revealing core limitation—agents fail on cross-function data-flow reasoning even with error observation loops."
    },
    {
      "title": "AutoResearchClaw and Verification Boundaries for Research Agents",
      "url": "https://bemiagent.com/agents/autoresearchclaw-verification-boundaries-research-agents",
      "date": "2026-05-22",
      "type": "research-paper",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Autonomous research architecture achieving 0.648 accuracy (vs AI Scientist v2 0.419) through verified numeric registry and citation verification; identifies key failure modes: confirmation bias, stateless runs, ungrounded writing; proposes verification boundaries as production control."
    },
    {
      "title": "State of AI Agents 2026: 200+ Data Points - Digital Applied",
      "url": "https://www.digitalapplied.com/blog/state-of-ai-agents-2026-200-data-points",
      "date": "2026-05-22",
      "type": "adoption-metric",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Enterprise adoption snapshot: 88% use AI somewhere but only 23% scale agentic systems enterprise-wide; Gartner forecasts 40%+ of agentic projects cancelled by 2027 due to escalating costs and unclear ROI—quantifies bleeding-edge adoption barriers."
    },
    {
      "title": "How I Built a Sales Research Pipeline Entirely Inside Google's Ecosystem",
      "url": "https://cashandcache.substack.com/p/i-used-gemini-to-find-sales-leads",
      "date": "2026-05-22",
      "type": "case-study",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Real workflow validation: Gemini Deep Research (7-10 min/company) → NotebookLM → Google Sheets producing 0-16 qualification scores with cited evidence; tested on three companies with measurable scoring differentiation and outreach decisions."
    },
    {
      "title": "Google Deep Research & Deep Research Max Review 2026",
      "url": "https://www.agent-finder.co/reviews/google-deep-research-deep-research-max",
      "date": "2026-05-20",
      "type": "opinion",
      "added": "2026-06-03",
      "superseded_by": null,
      "window": null,
      "explanation": "Technical review documenting Gemini Deep Research capabilities: synthesizes 30-60 sources with inline citations, MCP integration, native chart generation, iterative reasoning (4-7 search iterations). Max timing 3-10 minutes; gap narrowing with specialized platforms like Elicit."
    },
    {
      "title": "Latest updates on Perplexity AI: What's New in 2026? - Toolkitly",
      "url": "https://www.toolkitly.com/latest-updates/perplexity-ai",
      "date": "2026-05-14",
      "type": "product-ga",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "Product updates showing data warehouse integrations (Snowflake, Databricks for live SQL queries), 35 finance-specific workflows, GPT-5.5 default orchestration model—evidence of infrastructure maturity."
    },
    {
      "title": "Perplexity AI Features and Capabilities in 2026: The Complete Guide",
      "url": "https://www.secondtalent.com/resources/perplexity-ai-features-capabilities-2026/",
      "date": "2026-05-12",
      "type": "adoption-metric",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "Adoption metrics (45M+ MAU, 1B+ monthly queries, $450M ARR, 8% market share) plus case study: VC firm reduced deep research analysis time from 4 hours to 10 minutes per company."
    },
    {
      "title": "How Perplexity Works: Deep Research, Spaces, Pages, Model Council, Comet, and More",
      "url": "https://suprmind.ai/hub/perplexity/features/",
      "date": "2026-05-12",
      "type": "tutorial",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "Technical specification of Perplexity's iterative multi-step research loop (30 parallel searches, query decomposition, source evaluation, synthesis threshold) with cost structure ($0.82 per complex query)."
    },
    {
      "title": "Perplexity Release Notes - May 2026 Latest Updates",
      "url": "https://releasebot.io/updates/perplexity-ai",
      "date": "2026-05-12",
      "type": "product-ga",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "GA announcements: Personal Computer (Mac OS agentic agent) expanded from Max-only ($200/mo) to Pro ($20/mo), Finance connectors (Morningstar, PitchBook, FactSet), Teams integration, multi-step task approval gates."
    },
    {
      "title": "Perplexity Computer: Multi-Model Agent Orchestration Guide",
      "url": "https://zenvanriel.com/ai-engineer-blog/perplexity-computer-multi-model-agent-orchestration/",
      "date": "2026-05-07",
      "type": "opinion",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "Technical analysis showing 19-model orchestration enables autonomous deep research at enterprise scale, with parallel sub-agent execution for hours-long unattended workflows."
    },
    {
      "title": "Evaluating AI Agents In 2026: Benchmarks For Teams",
      "url": "https://www.adaline.ai/blog/evaluating-ai-agents-in-2026",
      "date": "2026-05-07",
      "type": "industry-report",
      "added": "2026-05-20",
      "superseded_by": null,
      "window": null,
      "explanation": "Benchmark analysis for multi-step research agents (OpenAI Deep Research 51.5% BrowseComp, GAIA 74.5%) showing shift from model-answer metrics to multi-step execution traces and tool-use quality."
    },
    {
      "title": "The Research Reality Gap: Why AI Agents are Scraping the Floor",
      "url": "https://micheallanham.substack.com/p/the-research-reality-gap-why-ai-agents",
      "date": "2026-05-04",
      "type": "research-paper",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "AutoResearchBench quantifies frontier model failures on 1,000 multi-step research tasks: Claude Opus 4.6 achieves 9.39% accuracy, GPT-5.4 at 7.44%, Gemini 3.1 Pro at 7.93%—fundamental closure and state-tracking defects in current research agents."
    },
    {
      "title": "45 AI Agent Statistics You Need to Know in 2026",
      "url": "https://www.ringly.io/blog/ai-agent-statistics-2026",
      "date": "2026-05-02",
      "type": "adoption-metric",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Market snapshot: 51% of enterprises have agents in production; 40%+ of agentic projects project to be cancelled by 2027; deep research consolidates on 3 platforms with Perplexity at $450M ARR."
    },
    {
      "title": "A Strategic OSINT Assessment of AI Agent Deployment, Workforce Economics and Geopolitical Differentiation in 2026",
      "url": "https://www.debugliesintel.com/the-autonomous-enterprise-a-strategic-osint-assessment-of-ai-agent-deployment-workforce-economics-and-geopolitical-differentiation-in-2026/",
      "date": "2026-05-01",
      "type": "industry-report",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Strategic analysis: R&D/research agents 'transformational at long time horizons'; documents real deployment challenges (integration complexity, autonomy failures in long-horizon reasoning) and workforce economics of autonomous research scaling."
    },
    {
      "title": "The Duel: ChatGPT Deep Research vs Google Gemini Deep Search",
      "url": "https://mike.schwede.ch/en/practical-knowledge/chatgpt-deep-research-vs-gemini-deep-search",
      "date": "2026-04-29",
      "type": "case-study",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner side-by-side: Gemini retrieved 211 sources vs ChatGPT's 48; Gemini completed in minutes vs ChatGPT's 14 minutes; Gemini showed transparent reasoning plan—real deployment metrics validating capability differentials."
    },
    {
      "title": "Equip your agent with deep research in one line of code | SPARKIT",
      "url": "https://sparkit.science/blog/deep-research-tools-compared",
      "date": "2026-04-28",
      "type": "case-study",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "All 5 deep research tools (Perplexity, ChatGPT, Gemini, Elicit, SPARKIT) identified correct biomedical answer; critical gap: ChatGPT/Gemini lack production APIs, 100x latency variance, Gemini returns 6,000+ words vs Perplexity's 40."
    },
    {
      "title": "85% of Enterprises Run AI Agents, Only 5% Trust Them for Production",
      "url": "https://saassentinel.com/2026/04/26/85-of-enterprises-run-ai-agents-only-5-trust-them-for-production/",
      "date": "2026-04-26",
      "type": "adoption-metric",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Cisco RSA 2026: 85% pilot AI agents, 5% production deployment; 88% experienced AI security incidents; real CEO examples show agents autonomously overriding policy, committing code without approval—governance barriers constraining deep research at scale."
    },
    {
      "title": "Enterprise AI Agents: 5 Proven Reasons 88% Fail Production in 2026",
      "url": "https://www.velsof.com/ai-automation/enterprise-ai-agents-fail-production-2026/",
      "date": "2026-04-25",
      "type": "industry-report",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Dynatrace 2026 Pulse: 88% of enterprise agents never reach production; 80% of integration time spent on legacy system connectors, evaluation frameworks, and governance design—core barriers for production deep research deployment."
    },
    {
      "title": "AI deep research agents struggle with nuanced academic subjects despite bold claims",
      "url": "https://www.newstex.com/blog/ai-tools-struggle-with-nuanced-academic-subjects-despite-bold-claims",
      "date": "2026-04-24",
      "type": "opinion",
      "added": "2026-05-06",
      "superseded_by": null,
      "window": null,
      "explanation": "Historian's independent testing: all three agents (ChatGPT, Perplexity, Gemini) exhibit citation hallucinations, false source attribution, and imprecision on specialized academic questions—critical limits of autonomous research on nuanced topics."
    },
    {
      "title": "Deep Research Max: a step change for autonomous research agents",
      "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/next-generation-gemini-deep-research/",
      "date": "2026-04-21",
      "type": "product-ga",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Google launches Deep Research Max with Gemini 3.1 Pro, introducing MCP support for proprietary data integration and native chart generation, positioning enterprise autonomous research for asynchronous due diligence workflows."
    },
    {
      "title": "Perplexity Computer | Agentic AI Platform, Pricing, How It Works 2026",
      "url": "https://www.objectwire.org/tech/perplexity/news/perplexity-computer-agentic-ai-platform-explained-2026",
      "date": "2026-04-21",
      "type": "product-ga",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Comprehensive technical documentation of Perplexity Computer's multi-model orchestration (Claude Opus reasoning, Gemini Deep Research, GPT-4 drafting) with sub-agent parallelization and background workflows running hours/days unattended."
    },
    {
      "title": "AI Slop Loop: Fake SEO Content Is Poisoning LLM Results",
      "url": "https://noticemesenpai.com/news/lily-ray-fake-google-update-ai-slop-loop-llm/",
      "date": "2026-04-16",
      "type": "case-study",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "SEO researcher demonstrates Perplexity autonomous research vulnerable to poisoned sources through synthetic data amplification: fabricated article cited as fact within 24 hours, revealing model collapse failure mode at scale."
    },
    {
      "title": "ArXiv: HORIZON — Where and Why AI Agents Fail on Long-Horizon Tasks",
      "url": "https://24-ai.news/en/vijest/2026-04-15/arxiv-horizon-agenti-dugi-zadaci/",
      "date": "2026-04-15",
      "type": "research-paper",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "HORIZON benchmark systematically diagnoses multi-step agent failures: cumulative error degradation, context loss after 20+ actions, faulty error recovery—showing short-horizon benchmarks do not predict long-horizon reliability."
    },
    {
      "title": "A Gemini Deep Research Failure Mode: Refusal, Topic Drift, and Fabricated Charts",
      "url": "https://dev.to/gys/a-gemini-deep-research-failure-mode-refusal-topic-drift-and-fabricated-charts-1dgd",
      "date": "2026-04-14",
      "type": "opinion",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Independent analysis documents pipeline desynchronization failures: safety refusal bugs triggered by escaped Markdown, topic drift recovery, and fabricated infographics (synthetic data in charts unrelated to report content)."
    },
    {
      "title": "Stanford's 2026 AI Index: Agents Score Half as Well as PhD Experts on Complex Multistep Workflows",
      "url": "https://newclawtimes.com/articles/stanford-hai-2026-ai-index-agents-50-percent-phd-experts-china-parity/",
      "date": "2026-04-14",
      "type": "industry-report",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Stanford HAI 2026 AI Index Report documents best agents achieve ~50% of PhD specialist performance on complex workflows; PaperArena benchmark shows 39% accuracy for autonomous research agents (reasoning, tool use, paper interaction)."
    },
    {
      "title": "Autonomous Research, Broken Reasoning, Smarter Agents",
      "url": "https://awesomeagents.ai/science/autonomous-research-broken-reasoning-smarter-agents/",
      "date": "2026-04-13",
      "type": "research-paper",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "AlphaLab arXiv paper (2604.08590) demonstrates GPT-5.2 and Claude Opus 4.6 autonomously conducting multi-phase research with quantified outcomes: 4.4x GPU kernel speedup, 22% validation loss reduction, $150-200 per campaign cost."
    },
    {
      "title": "Deep Research Agents: Why Most Implementations Loop Forever or Stop Too Early",
      "url": "https://tianpan.co/blog/2026-04-12-deep-research-agents-orchestrating-multi-step-search-that-converges",
      "date": "2026-04-12",
      "type": "opinion",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Deep technical analysis of multi-step research agent architecture: convergence detection strategies, cost data ($2-5 per query), source credibility failure (50-90% unsupported citations), production scaling patterns with parallel execution."
    },
    {
      "title": "Google's FACTS Benchmark Reveals Top AI Model Is Only 69% Accurate",
      "url": "https://hyper.ai/en/stories/0cda5fd52f8626cb63c0f83e62e86fbb",
      "date": "2026-04-10",
      "type": "research-paper",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Google DeepMind's FACTS benchmark (fact-QA, web search, document extraction, visual interpretation) shows Gemini 3 Pro achieves only 69% accuracy, revealing fundamental reliability gap for research-dependent tasks."
    },
    {
      "title": "Perplexity Hits $450M ARR After 50% Monthly Revenue Surge Following Agentic AI Pivot",
      "url": "https://newclawtimes.com/articles/perplexity-450-million-arr-50-percent-revenue-surge-agentic-ai/",
      "date": "2026-04-09",
      "type": "adoption-metric",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Perplexity Computer reaches $450M ARR (50% MoM growth March 2026) with 100M+ users and tens of thousands of enterprises deploying autonomous workflows (document review, campaign planning, tax filing generation)."
    },
    {
      "title": "Perplexity revenue surges 50% as AI startup shifts from search to autonomous AI agents",
      "url": "https://techstartups.com/2026/04/08/perplexity-revenue-surges-50-as-ai-startup-shifts-from-search-to-autonomous-ai-agents/",
      "date": "2026-04-08",
      "type": "adoption-metric",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Case study: Perplexity Computer executing multi-step autonomous workflows with measured ROI—$225K annual marketing stack replaced in one weekend; 100M+ user base validates production deployment scale."
    },
    {
      "title": "Perplexity's AI Pivot: From Search to Enterprise Agents - i10X",
      "url": "https://i10x.ai/news/perplexitys-pivot-from-ai-search-to-enterprise-agents",
      "date": "2026-04-08",
      "type": "industry-report",
      "added": "2026-04-22",
      "superseded_by": null,
      "window": null,
      "explanation": "Independent analyst: strategic pivot from commodity search to high-value workflow automation where agentic systems command premium pricing by replacing labor, not augmenting it—forcing incumbents to accelerate agent GTM."
    },
    {
      "title": "Enterprise AI adoption in 2026: Why 79% face challenges despite high investment",
      "url": "https://writer.com/blog/enterprise-ai-adoption-2026/",
      "date": "2026-04-07",
      "type": "adoption-metric",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Independent survey (2,400 respondents): 97% deployed AI agents, only 29% see ROI; 67% suffered data breaches via unapproved AI tools, 36% lack governance plans—governance gaps directly prevent scaling of autonomous research agents."
    },
    {
      "title": "The Year of the Autonomous Worker: How Agentic AI Redefined the Enterprise in 2026",
      "url": "https://shshell.com/blog/agentic-ai-enterprise-2026",
      "date": "2026-04-03",
      "type": "opinion",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Practitioner analysis documents deployed autonomous research workflow: Sovereign Legal Agents autonomously conduct M&A due diligence across centuries of records, achieving 90% time reduction (months to single afternoon) with production-scale execution metrics."
    },
    {
      "title": "ChatGPT Got It Wrong. Gemini Made Up a Reason. Claude Gave Up.",
      "url": "https://compoundingai.substack.com/p/chatgpt-got-it-wrong-gemini-made",
      "date": "2026-04-02",
      "type": "case-study",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Systematic comparison of AI systems on financial document research: ChatGPT hallucinated, Gemini fabricated reasoning, Claude refused—demonstrating autonomous research architecturally unfit for mission-critical analysis without structured source tracking."
    },
    {
      "title": "Google Unveils Gemini Enterprise to Expand AI Integration in the Workplace",
      "url": "https://hyper.ai/en/stories/64b2ae9c43b1297e7bc08eefddfaaf46",
      "date": "2026-03-30",
      "type": "case-study",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Google launches Gemini Enterprise with Deep Research as core pre-built autonomous agent; named deployments show measurable adoption: Macquarie Bank 38% engagement lift, JCOM analyzing 100k+ customer conversations monthly with autonomous agents."
    },
    {
      "title": "How AI Agents Actually Work in Production: Lessons From 306 Real-World Deployments",
      "url": "https://www.libertify.com/interactive-library/ai-agents-production-deployment-study/",
      "date": "2026-03-28",
      "type": "research-paper",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Large empirical MAP study across 306 practitioners and 26 industries reveals critical constraint: 80% of production agents use predefined workflows, not open-ended autonomy—key limitation explaining why deep research remains bounded capability."
    },
    {
      "title": "Google Gemini Deep Research Now Integrates with Gmail, Drive, Docs, and Chat",
      "url": "https://hyper.ai/en/stories/999b1db53ebe3b4722ae160844a3b04e",
      "date": "2026-03-27",
      "type": "product-ga",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Gemini Deep Research expands to access Workspace data (Gmail, Drive, Docs, Chat), enabling autonomous research blending public web searches with private organizational knowledge—core capability maturation for enterprise deep research."
    },
    {
      "title": "Gemini 3 As An Agentic International Think Tank In A Box",
      "url": "https://blog.gdeltproject.org/gemini-3-as-an-agentic-international-think-tank-in-a-box-five-reports-deeply-analyzing-iran-based-on-yesterdays-iranian-tv-news-coverage/",
      "date": "2026-03-27",
      "type": "case-study",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "GDELT Project proof-of-concept: Gemini 3.1 Pro autonomously analyzes TV transcripts, identifies research topics, generates five synthetic think tank reports with full domain synthesis—100% automated, $2.14 cost, zero human intervention except prompt."
    },
    {
      "title": "Deloitte Tech Trends 2026: Only 11% of Enterprises Run Agentic AI in Production",
      "url": "https://virtualassistantva.com/news/deloitte-tech-trends-2026-agentic-ai-silicon-workforce-only-11-percent-production",
      "date": "2026-03-26",
      "type": "industry-report",
      "added": "2026-04-08",
      "superseded_by": null,
      "window": null,
      "explanation": "Deloitte authoritative analysis quantifies critical adoption gap: only 11% of enterprises run agentic AI in production; 40% of projects fail due to process-design barriers, directly explaining why deep research agents remain bleeding-edge despite product maturity."
    },
    {
      "title": "Google's Gemini Deep Research Goes Full Workspace Integration",
      "url": "https://www.techbuzz.ai/articles/google-s-gemini-deep-research-goes-full-workspace-integration",
      "date": "2026-03-18",
      "type": "product-ga",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "Gemini Deep Research achieves Workspace GA, integrating Gmail, Drive, and Chat to combine internal documents with web sources in unified research workflows for enterprise teams."
    },
    {
      "title": "ChatGPT's Scientific Judgement Questioned As Study Exposes Reliability Gaps",
      "url": "https://www.azoai.com/news/20260317/ChatGPTe28099s-Scientific-Judgement-Questioned-As-Study-Exposes-Reliability-Gaps.aspx",
      "date": "2026-03-17",
      "type": "research-paper",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "WSU study of 700+ business hypotheses: ChatGPT achieves 76.5% accuracy but only 41% consistency across 10 runs, revealing unreliability for scientific claim verification—critical gap for research-grade deep research agents."
    },
    {
      "title": "Data Points: Perplexity Computer hits desktop, mobile, Pro",
      "url": "https://charonhub.deeplearning.ai/perplexity-computer-hits-desktop-mobile-pro/",
      "date": "2026-03-16",
      "type": "news-coverage",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "Perplexity Computer expands to Personal Computer (local Mac Mini execution), Enterprise (Snowflake/Salesforce integration), and Comet Enterprise browser; launches four new APIs (Search, Agent, Embeddings, Sandbox) orchestrating 20 models."
    },
    {
      "title": "Perplexity Computer is INSANE: How to 10x Your Productivity with Model-Agnostic Automation",
      "url": "https://www.bizrescuepro.com/perplexity-computer-is-insane-how-to-10x-your-productivity-with-model-agnostic-automation/",
      "date": "2026-03-16",
      "type": "case-study",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "Production deployment testing: Perplexity Computer automates five concrete workflows (weekly briefing, YouTube research, investor analysis, competitive monitoring, legal research) with measured speed gains and cost reductions."
    },
    {
      "title": "Agentic AI: 27% Stuck Between Pilot and Production",
      "url": "https://byteiota.com/agentic-ai-27-stuck-between-pilot-and-production/",
      "date": "2026-03-11",
      "type": "adoption-metric",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "CrewAI survey: 81% of enterprises claim to scale AI agents but only 11% deployed in production (27-point gap); 38% stuck in pilots; Gartner predicts 40% of agentic projects cancelled by 2027 due to costs, ROI, and governance barriers."
    },
    {
      "title": "DeepFact: Co-Evolving Benchmarks and Agents for Deep Research Factuality",
      "url": "https://arxiv.org/abs/2603.05912",
      "date": "2026-03-06",
      "type": "research-paper",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "Google researchers address deep research report (DRR) factuality via co-evolving benchmarks; PhD-level expert accuracy improves from 60.8% (static labels) to 81%+ (iterative audit-then-score), confirming benchmark brittleness as reliability barrier."
    },
    {
      "title": "Perplexity AI API Access and Developer Use Cases Overview",
      "url": "https://www.datastudios.org/post/perplexity-ai-api-access-and-developer-use-cases-overview-platform-structure-key-capabilities-and",
      "date": "2026-03-03",
      "type": "industry-report",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-Q1",
      "explanation": "DataStudios comprehensive API analysis identifies three core Perplexity APIs: Sonar (web-grounded answers), Agentic Research (multi-step orchestrated workflows with explicit reasoning control), Search (ranked results with filtering)."
    },
    {
      "title": "Perplexity Unveils Enterprise-Focused AI Agent System Powered by Multi-Model Architecture",
      "url": "https://theaiinsider.tech/2026/02/28/perplexity-unveils-enterprise-focused-ai-agent-system-powered-by-multi-model-architecture/",
      "date": "2026-02-28",
      "type": "product-ga",
      "added": "2026-03-25",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Perplexity launches Computer, a production autonomous research system orchestrating 19 AI models with autonomous task execution, dynamic subagent generation, and enterprise deployment at $200/month Max tier."
    },
    {
      "title": "ChatGPT Deep Research: Guide to AI Agents & RAG",
      "url": "https://intuitionlabs.ai/articles/chatgpt-deep-research-guide-ai-agents-rag",
      "date": "2026-02-26",
      "type": "opinion",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Comprehensive analysis of ChatGPT Deep Research: ~700M weekly users conducting research, but hallucination rates remain high (90% in complex synthesis), underscoring need for human oversight."
    },
    {
      "title": "When Agent Capability Gains Stop Predicting Reliability",
      "url": "https://promptedllc.com/research/when-agent-capability-gains-stop-predicting-reliability",
      "date": "2026-02-20",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Analyst report cites Princeton research: 18 months of capability improvements yield zero reliability gains for production agents, highlighting the decoupling of model advancement from deployment readiness."
    },
    {
      "title": "Perplexity AI Stats Feb 2026: Uses, Users, Market Share, and More",
      "url": "https://fatjoe.com/blog/perplexity-ai-stats/",
      "date": "2026-02-20",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Perplexity usage data: 33M monthly active users, 780M queries in 2025; 20.8% of queries target research/learning, demonstrating mainstream adoption of AI-powered deep research workflows."
    },
    {
      "title": "Gemini Apps' release updates and improvements",
      "url": "https://gemini.google/to/release-notes/?hl=en-GB",
      "date": "2026-02-19",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "Google releases Gemini 3.1 Pro and upgrades Gemini 3 Deep Think for complex science and research tasks, signaling continued platform investment in autonomous multi-step research capabilities."
    },
    {
      "title": "Autonomous Deepmind AI Generates Publishable Math Papers",
      "url": "https://www.nextbigfuture.com/2026/02/autonomous-deepmind-ai-generates-publishable-math-papers-next-accelerate-science-research.html",
      "date": "2026-02-11",
      "type": "case-study",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "DeepMind's Aletheia multi-step research agent achieves 91.9% on IMO-ProofBench, autonomously solves mathematical problems, and co-authors published papers, demonstrating specialized deep research deployment."
    },
    {
      "title": "Research reveals that even the best AI with web search turned on experiences false beliefs in about 30% of cases",
      "url": "https://gigazine.net/gsc_news/en/20260210-ai-hallucination-halluhard/",
      "date": "2026-02-10",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-02",
      "explanation": "HalluHard benchmark study shows state-of-the-art AI models with web search still hallucinate ~30% in multi-turn conversations, revealing critical reliability limitations for autonomous deep research workflows."
    },
    {
      "title": "Gartner Data: Enterprise AI Adoption Analysis 2026",
      "url": "https://aiseo.com.mx/en/gartner-enterprise-ai-adoption-analysis-2026/",
      "date": "2026-01-30",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Gartner Q4 2025 survey: 67% of enterprises report employees using ChatGPT, Perplexity, or Gemini for business research, signaling mainstream adoption of AI-powered research tools."
    },
    {
      "title": "The Next Phase of AI: Technology, Infrastructure, and Policy in 2025-2026",
      "url": "https://www.americanactionforum.org/insight/the-next-phase-of-ai-technology-infrastructure-and-policy-in-2025-2026/",
      "date": "2026-01-30",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Policy analysis: 62% of organizations experimented with agentic workflows in 2025, but 70-80% struggled to scale with only 5% achieving meaningful ROI, indicating adoption barriers."
    },
    {
      "title": "AI assistants make widespread errors about the news, new study finds",
      "url": "https://www.globalbankingandfinance.com/ai-news-ebu-zero/",
      "date": "2026-01-21",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "EBU/BBC study of 3,000 AI assistant responses: 45% contained significant errors; Gemini had 72% sourcing errors, highlighting accuracy and citation integrity risks in deep research workflows."
    },
    {
      "title": "Building multi-agent applications with Deep Agents",
      "url": "https://blog.langchain.com/building-multi-agent-applications-with-deep-agents/",
      "date": "2026-01-21",
      "type": "significant-repo",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "LangChain Deep Agents framework enables multi-step autonomous research through subagents and progressive skills, addressing context bloat and enabling complex task decomposition."
    },
    {
      "title": "Perplexity vs Gemini 3 Pro: My Honest Review After a Year of Daily Use",
      "url": "https://www.glbgpt.com/hub/perplexity-vs-gemini-3-pro/",
      "date": "2026-01-08",
      "type": "opinion",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Independent practitioner assessment: Perplexity excels in factual accuracy (93.9%) and citations (95%), while Gemini dominates multimodal reasoning, reflecting tool differentiation in deep research workflows."
    },
    {
      "title": "AI App Winners of 2025: Market Share Leaders and What to Expect in 2026",
      "url": "https://www.fsedigital.com/blog/ai-app-winners-2025-market-share-trends-2026/",
      "date": "2026-01-05",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2026-01",
      "explanation": "Market analysis: Perplexity recorded 370% YoY user growth with peak 14.1% market share, positioning itself as leading AI research tool alternative to traditional search."
    },
    {
      "title": "From large language models to AI agents in energy materials research",
      "url": "https://www.oaepublish.com/articles/aiagent.2025.03",
      "date": "2025-12-19",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Peer-reviewed analysis of multi-agent AI systems for scientific discovery in energy materials, addressing autonomous reasoning, hypothesis generation, and physics-informed agent design."
    },
    {
      "title": "Why AI Agents Failed in 2025: Deloitte's New Findings",
      "url": "https://eastgate-software.com/why-ai-agents-failed-in-2025-deloittes-new-findings/",
      "date": "2025-12-11",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Deloitte Tech Trends 2025: only 11% of organizations use AI agents in production; legacy architecture and weak data governance are primary adoption barriers; 93% of AI budgets fund technology, 7% fund readiness."
    },
    {
      "title": "Perplexity AI Agent Adoption Reaches 57%, Driven By Digital Sector",
      "url": "https://quantumzeitgeist.com/57-percent-ai-perplexity-agent-adoption-reaches-driven-digital-sector-higher/",
      "date": "2025-12-09",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Study of 100M+ Perplexity Comet agent interactions (Jul-Oct 2025): 57% of agentic queries target research/learning; adoption concentrated in knowledge-intensive sectors (digital tech 28%, academia, finance, marketing)."
    },
    {
      "title": "Gemini 3 Pro tops new AI reliability benchmark, but hallucination rates remain high",
      "url": "https://the-decoder.com/gemini-3-pro-tops-new-ai-reliability-benchmark-but-hallucination-rates-remain-high/",
      "date": "2025-11-19",
      "type": "news-coverage",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Artificial Analysis benchmark (6,000 questions): Gemini 3 Pro achieves 53% accuracy (leading peers), but 88% hallucination rate persists, indicating reliability limitations for multi-step research workflows."
    },
    {
      "title": "How to Generate Full Reports with Gemini Deep Research",
      "url": "https://skywork.ai/blog/ai-agent/gemini-research-reports/",
      "date": "2025-11-10",
      "type": "case-study",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Production deployment case study: Gemini Deep Research generates 6-8 page reports with 12-18 sources in 2-4 minutes, achieving 93% citation accuracy and 15-25% speed improvement over manual workflow."
    },
    {
      "title": "Survey finds slow adoption of autonomous AI agents in enterprises",
      "url": "https://itbrief.news/story/survey-finds-slow-adoption-of-autonomous-ai-agents-in-enterprises",
      "date": "2025-10-01",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q4",
      "explanation": "Gartner survey of 360 IT leaders (Oct 2025): only 15% are considering/piloting/deploying fully autonomous agents; 74% cite security concerns, 19% trust hallucination protection, governance barriers impede production."
    },
    {
      "title": "Survey and analysis of hallucinations in large language models",
      "url": "https://www.frontiersin.org/journals/artificial-intelligence/articles/10.3389/frai.2025.1622292/full",
      "date": "2025-09-30",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Peer-reviewed survey empirically analyzing hallucination attribution in LLMs; chain-of-thought prompting reduces hallucinations in prompt-sensitive scenarios, though intrinsic model limitations persist."
    },
    {
      "title": "Which Barriers Still Block Agentic AI Adoption?",
      "url": "https://www.vccafe.com/2025/09/24/which-barriers-still-block-agentic-ai-adoption/",
      "date": "2025-09-24",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "VC analysis citing MIT study: 95% of GenAI pilots fail to reach production; identifies reliability, data quality, and governance barriers to deep research agent deployment at scale."
    },
    {
      "title": "Unlock Perplexity AI Pro's Next-Gen Research Power - Global Adoption",
      "url": "https://aiintro.space/perplexity-ai-pro-2025-research-global-adoption/",
      "date": "2025-08-13",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Analysis of Perplexity Pro regional adoption showing 40% large enterprises (banking, law, pharma), 60% using for mission-critical research; production deployment across compliance, innovation, risk modeling use cases."
    },
    {
      "title": "Sử dụng tính năng Deep Research trong Các ứng dụng - Google Support",
      "url": "https://support.google.com/gemini/answer/15719111?hl=vi&co=GENIE.Platform%3DAndroid",
      "date": "2025-08-03",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Official Google documentation confirming Deep Research production deployment with Workspace integration, usage quotas, and enterprise-grade features across mobile and web platforms."
    },
    {
      "title": "Gemini AI - has some issues (including NSFW) - Google AI Studio",
      "url": "https://discuss.ai.google.dev/t/gemini-ai-has-some-issues-including-nsfw/94942",
      "date": "2025-07-21",
      "type": "case-study",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "User-documented hallucinations in Gemini Deep Research on current affairs; reports inconsistent responses and stance changes on factual events, providing negative signal on real-world reliability."
    },
    {
      "title": "Exploring the Dilemma of AI Use in Medical Research",
      "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC12288101/",
      "date": "2025-07-15",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q3",
      "explanation": "Peer-reviewed analysis of OpenAI's Deep Research tool in scientific workflow; documents paradox between efficiency gains and risks to citation integrity, critical appraisal, and research quality."
    },
    {
      "title": "Deep Research: A Survey of Autonomous Research Agents - arXiv",
      "url": "https://arxiv.org/html/2508.12752v1",
      "date": "2025-06-26",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "Comprehensive arXiv survey documents deep research agent advances in planning, reasoning, question development, and optimization; maps emerging category's technical foundations."
    },
    {
      "title": "Steerable Deep Research: Building Production-Ready Agentic Workflows with Controlled Autonomy",
      "url": "https://www.zenml.io/blog/steerable-deep-research-building-production-ready-agentic-workflows-with-controlled-autonomy",
      "date": "2025-06-17",
      "type": "case-study",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "By June 2025, production-ready deep research patterns emerged across platforms (OpenAI, Google, Perplexity, Claude), with steerable workflows enabling controlled autonomy in enterprise settings."
    },
    {
      "title": "The latest updates for Deep Research in Gemini",
      "url": "https://workspaceupdates.googleblog.com/2025/05/deep-research-updates-gemini-io-2025.html",
      "date": "2025-05-22",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "Google I/O 2025: Deep Research now supports Flash 2.5 experimental; Gemini Advanced users access 2.5 Pro experimental, accelerating deployment to enterprise via Workspace."
    },
    {
      "title": "Perplexity AI Statistics And Facts (2025) - ElectroIQ",
      "url": "https://electroiq.com/stats/perplexity-ai-statistics/",
      "date": "2025-05-06",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "Perplexity reached 15M active users by Q2 2025 with 50% growth over three months; valuation jumped to $9B, signaling mainstream adoption of AI-powered deep research."
    },
    {
      "title": "Unpacking Perplexity AI: Features, Performance, and...",
      "url": "https://seo.goover.ai/report/202504/go-public-report-en-9f7b12b9-b6c4-4015-8ae9-7fc46f912d74-0-0.html",
      "date": "2025-04-26",
      "type": "news-coverage",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "Perplexity pursues $500M-$1B funding round targeting $18B valuation; zero-click search adoption demonstrates strong market demand for instant AI-powered research insights."
    },
    {
      "title": "Gemini DeepResearch has an information lag problem, leading to bias in the research subject",
      "url": "https://discuss.ai.google.dev/t/gemini-deepresearch-has-an-information-lag-problem-leading-to-bias-in-the-research-subject/80028",
      "date": "2025-04-18",
      "type": "opinion",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q2",
      "explanation": "User reports document Gemini Deep Research limitations: knowledge cutoff information lag, model substitutions (GPT 4.1 vs 4o mini), leading to incomplete or biased research outputs."
    },
    {
      "title": "Deep Research Agents: A Systematic Examination and Roadmap",
      "url": "https://arxiv.org/html/2506.18096",
      "date": "2025-03-07",
      "type": "research-paper",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Comprehensive survey of deep research agents examining advances in reasoning, tool integration, retrieval-augmented generation, and agentic retrieval as category matures."
    },
    {
      "title": "Gemini Deep Research and experimental models now available to Google Workspace users in Gemini Advanced",
      "url": "https://workspaceupdates.googleblog.com/2025/02/deep-research-available-for-google-workspace-in-gemini-advanced.html?m=1",
      "date": "2025-02-20",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Google extends Deep Research to Workspace users in Gemini Advanced starting in English; signals enterprise deployment of multi-step research capabilities."
    },
    {
      "title": "Autonomous Research and the Job Market: Is AI the End of Traditional Analysts?",
      "url": "https://www.1950.ai/post/autonomous-research-and-the-job-market-is-ai-the-end-of-traditional-analysts",
      "date": "2025-02-20",
      "type": "opinion",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "OpenAI Deep Research reshaping knowledge-intensive research with high-quality reports at speed and cost efficiency; agentic RAG integration driving industry transformation."
    },
    {
      "title": "The Evolution of Research: A Deep Dive into Perplexity's New Deep Research",
      "url": "https://notes.scottyq.co.uk/2025/02/14/the-evolution-of-research-a/",
      "date": "2025-02-14",
      "type": "news-coverage",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Perplexity Deep Research generates comprehensive reports in 10-30 minutes with advanced language models and autonomous research; benchmarks 93.9% on SimpleQA vs competitors."
    },
    {
      "title": "OpenAI's new 'deep research' agent is still just a fallible tool – not a human-level expert",
      "url": "https://modernsciences.org/staging/4414/openai-deep-research-agent-ai-tool-limitations-february-2025/",
      "date": "2025-02-13",
      "type": "opinion",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "Critical analysis of OpenAI Deep Research limitations: autonomous web searching and source compilation work, but tool remains fallible, not expert-level, with reliability gaps."
    },
    {
      "title": "OpenAI Unveils Deep Research, an AI Agent That Conducts Multi-Step Research Tasks",
      "url": "https://www.maginative.com/article/openai-unveils-deep-research-an-ai-agent-that-conducts-multi-step-research-tasks",
      "date": "2025-02-03",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2025-Q1",
      "explanation": "OpenAI launches Deep Research in ChatGPT Pro, using o3 model for autonomous multi-step research; available to Pro users with rollout to Plus and Team users."
    },
    {
      "title": "Chapter 4: Key Use Cases And... (AI Agents Survey Results)",
      "url": "https://yougot.us/news/2024-12-28-AI-Agents-Survey-Results/",
      "date": "2024-12-28",
      "type": "adoption-metric",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Survey of 300+ practitioners shows 68% deployment of AI agents but only 32% see significant ROI, highlighting adoption breadth alongside challenges."
    },
    {
      "title": "Gemini Advanced rolls out 'Deep Research' as first agentic feature",
      "url": "https://9to5google.com/2024/12/20/gemini-advanced-deep-research/",
      "date": "2024-12-20",
      "type": "news-coverage",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Deep Research rolls out globally to Gemini Advanced users across 45+ languages and 150+ countries, demonstrating mainstream vendor commitment to multi-step research tooling."
    },
    {
      "title": "For artificial intelligence to be mission-critical, it must hallucinate less",
      "url": "https://breakingdefense.com/2024/12/for-artificial-intelligence-to-be-mission-critical-it-must-hallucinate-less/",
      "date": "2024-12-18",
      "type": "news-coverage",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Breaking Defense reports LLMs hallucinate in ~10% of responses; highlights accuracy barriers for deep research systems, though Primer's RAG-V reduces errors to 0.1%."
    },
    {
      "title": "Introducing Gemini 2.0: our new AI model for the agentic era",
      "url": "https://blog.google/innovation-and-ai/models-and-research/google-deepmind/google-gemini-ai-update-december-2024/",
      "date": "2024-12-11",
      "type": "product-ga",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Google announces Deep Research feature in Gemini Advanced, a multi-step research assistant using advanced reasoning and long context to explore complex topics and compile reports."
    },
    {
      "title": "What's next for AI? - AI agents and autonomous AI",
      "url": "https://www.deloitte.com/us/en/insights/topics/technology-management/tech-trends/2025/tech-trends-ai-agents-and-autonomous-ai.html",
      "date": "2024-12-10",
      "type": "industry-report",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Deloitte Tech Trends 2025 report signals mainstream analyst recognition of AI agents as key trend, with 70% of organizations using LLMs and shift toward more agentic capabilities."
    },
    {
      "title": "Perplexity Pro Search case study",
      "url": "https://huggingface.co/datasets/zenml/llmops-database/blame/main/markdown_data/perplexity-2024-c63bb046-1a69-4e05-a0ed-a6b82dff00ad.md",
      "date": "2024-12-04",
      "type": "case-study",
      "added": "2026-03-20",
      "superseded_by": null,
      "window": "2024-Q4",
      "explanation": "Perplexity Pro Search demonstrated 50% increase in query search volume with multi-step reasoning architecture separating planning and execution phases."
    }
  ],
  "tierHistory": [
    {
      "tier": "research",
      "from": "2024-12-01",
      "to": "2024-12-01"
    },
    {
      "tier": "bleeding-edge",
      "from": "2024-12-01",
      "to": null
    }
  ],
  "trendHistory": [
    {
      "trend": "steady",
      "blockerType": null,
      "from": "2026-09-26",
      "to": null
    }
  ],
  "description": "AI agents that conduct multi-step research autonomously — formulating queries, reading sources, following leads, and synthesising findings. Includes tools like Gemini Deep Research and Perplexity Pro; distinct from single-query retrieval which answers from a single search round.",
  "overview": "Multi-step autonomous deep research -- AI agents that formulate queries, read sources, follow leads across multiple rounds, and synthesise findings without human intervention -- has crossed from consumer experimentation into bounded production deployment. Three major platforms (Google, OpenAI, Perplexity) now offer enterprise-grade deep research features deployed at scale, with Perplexity Computer reaching $450M ARR and Google's Deep Research Max launching native multi-model orchestration for asynchronous enterprise workflows. Yet the practice remains fundamentally constrained by two unresolved gaps. First: reliability. Multi-turn research accuracy gaps persist across all platforms: Stanford HAI 2026 AI Index shows agents achieving only ~50% of PhD specialist performance on complex workflows; WildClawBench reveals best-in-class models reach only 62.2% on 60 realistic long-horizon tasks; AutoExperiment demonstrates frontier agents collapse from 30-37% accuracy to 6-10% when research tasks have cross-function dependencies; multi-model consensus breaks down, with 99.1% of real-world turns showing contradictions across frontier models and Gemini's single-model confidence suffering 51.3% contradiction rate from peers. A critical new finding (July 2026): verification failures are *workflow-context dependent*—verifier models that correctly identify misinformation in isolation adopt the same false information when embedded in multi-step research pipelines, with adoption rates escalating from 34.5% at cold-start to 85% immediately before synthesis, suggesting verification requires continuous monitoring rather than final-stage auditing. Error compounding is structural (Lusser's law): 95% per-step accuracy yields only 36% success on 20-step tasks, and self-conditioning causes agents to treat their own earlier errors as ground truth. A Princeton-backed analysis documented that eighteen months of model capability gains yielded zero reliability improvement for production agents. Second: organizational scaling. Only 23% of enterprises scale agentic systems enterprise-wide despite 88% using AI somewhere; 71% of deployed \"agents\" cannot complete autonomous multi-step work, remaining chatbots; 88% of agent pilots fail to reach production. Gartner forecasts 40%+ of agentic projects cancelled by 2027 due to governance and unclear ROI. Deep research agents are fully available and deployed in early-adopter teams -- but orchestration matters as much as models (multi-agent systems achieve 90.2% improvement over single-agent), and the majority of organizations have not yet solved the verification frameworks, context governance, and process redesign required for autonomous research at scale. For exploratory research and bounded decision support with human review, deep research delivers measurable acceleration. For mission-critical analysis, autonomous research remains a supervised tool requiring verification boundaries.",
  "currentLandscape": "By September 2026, deep research has consolidated around three major vendor platforms (Google, OpenAI, Perplexity) with production deployments reaching mainstream scale. Perplexity Computer, launched February 2026, orchestrates 20 AI models for multi-step autonomous research tasks with enterprise deployment at $200/month; expanded to local-first deployment (Portable Computer on NVIDIA DGX Spark, August 2026) with deterministic harness and reactive cloud escalation achieving 85.4% accuracy on local tasks and 73% with hybrid routing at one-third frontier cost. Perplexity Hybrid Compute (September 1, 2026) enables single agents to divide research workflows between cloud frontier models and on-device models without restarting, addressing production deployment barriers around data residency and privacy. Perplexity's SPACE sandbox infrastructure processes 1.25M creations and 11.9M reconnects weekly. Google's Deep Research Max (April 2026) adds Model Context Protocol (MCP) support and now reaches 1 billion monthly active users with 90% Fortune 100 enterprise adoption (September 2026), 22B API tokens/min infrastructure scale, Workspace integration enabling synthesis across internal documents and public web sources. Google Gemini Enterprise for Financial Services (August 2026) operates a managed financial research agent across regulated data (FactSet, Moody's, LSEG, MSCI, SEC Edgar) with early-stage customers including CME Group and Deutsche Bank executing autonomous credit risk analysis (reducing bond portfolio assessment from days to under 5 minutes) and legal research (Weil Gotshal deploying parallel research agents for multi-step investigation without continuous human redirection). Anthropic shipped production-grade infrastructure (computer use, browser use, Skills API, Files API reaching GA, August 2026), removing experimental designation for autonomous research task orchestration. Perplexity Computer reached $450M ARR with tens of thousands of enterprises executing multi-step workflows. OpenAI's Deep Research scales across ChatGPT's 700M+ weekly users. The practice has achieved mainstream infrastructure maturity and vendor competition at scale.\n\nEarly-stage production deployments achieve measurable results in specific use cases: legal teams conducting M&A due diligence autonomously across centuries of corporate records, reducing months to single afternoons (90% time reduction); sales research pipeline (Gemini Deep Research → NotebookLM → Google Sheets) producing 0-16 qualification scores with cited evidence in under 20 minutes per company; financial analysis (Skywork case study: 93% citation accuracy, 15-25% speed gains); media research (GDELT Project: autonomous analysis of 500+ TV transcripts generating think tank reports at $2.14 per run, unattended); academic research with frontier models (AlphaLab: multi-phase research with 4.4x GPU kernel speedup); database operations (Pythian: 3× engagement increase, 80% MTTR reduction on 15k monthly tickets). Infrastructure maturation: LangChain Deep Agents framework, standardized multi-model orchestration patterns, and verification boundaries enabling production control. Yet critical citation quality findings from September 2026 audits sharpen the reliability picture: Haus Research audit shows 34.7% of Perplexity citations fail verification on factual questions (16.1% of accessible pages contained none of the cited figures); BERI/Trellner analysis reveals 59.8% of Perplexity citations originate outside Tranco's top 100k domains with three coordinated fake software-guide domains supplying 215k machine-generated pages designed for AI consumption. Zan Digital's comparative audit shows citation accuracy ranges 77.96%–93.68% across platforms, with Gemini 81.44% vs Perplexity 90.24%—vendors market citation volume over precision, creating perception of credibility that masks accuracy collapse.\n\nYet critical reliability barriers persist and were sharply clarified in August-September 2026. AutoResearch (August 2026) evaluated 8 harness-model combinations on 100 end-to-end research tasks, identifying 45 failure patterns; central finding: all agents lack metacognitive loop—inability to verify output against sources, revise work, or self-question their reasoning path. WANDR benchmark (August 2026) of 500 professional research tasks shows strongest production system reaches only 0.363 soft F1 and 0.133 hard F1—demonstrating the practice remains far from saturation despite vendor maturity claims. Princeton and UK AISI Shadow Evaluation (August 2026) tested frontier models on real unpublished NeurIPS papers; both Claude 4.8 and GPT-5.6 Sol were rejected by original authors for weak experimental motivation, unjustified methodology, and zero novel contributions—contradicting vendor assertions of autonomous research capability. September 2026 evidence shows even newer model versions regress on reliability: TrueStandard measurement found Gemini 3.8 Flash fabricated 20% of DOI citations vs 12.6% for predecessor (Gemini 3.7), indicating versions can degrade reliability undetected by vendor testing. Earlier barriers remain structural: WildClawBench shows best-in-class reach only 62.2% on long-horizon tasks (18pp swing from orchestration alone). Multi-model consensus breaks down: 99.1% of production turns show contradictions; Gemini's high-confidence answers suffer 51.3% contradiction rate from peers. AutoExperiment demonstrates agents collapse from 30-37% to 6-10% accuracy when tasks have dependencies—agents fail on cross-function data-flow reasoning. Verification failures are pipeline-context dependent (July 2026): verifiers correctly reject misinformation in isolation but adopt it when embedded in research workflows (adoption rates 34.5% cold-start to 85% pre-synthesis). Source credibility is systematically exploited: 50-90% of citations remain unsupported; no LLM model under any condition achieves >50% citation-existence rate (August 2026 empirical study). Organizational barriers compound reliability gaps: only 9% of enterprises making progress on autonomous multistep workflows vs 59% claiming agent use (August 2026); 71% of deployed \"agents\" cannot complete autonomous multi-step work; 88% of agent pilots fail to reach production. Security vulnerabilities block deployment: Perplexity Comet suffers from zero-click Intent Collision attacks exploiting agent autonomy decision-making (August 2026), with Gartner recommending agent browser blocks by default. Deep research remains production-feasible for bounded use cases with human review (Tech Mahindra, Perplexity for sales), but autonomous research at enterprise scale is blocked by metacognitive deficits in models, pipeline reliability gaps, organizational scaling barriers, and fundamental vulnerabilities in agent autonomy. August 2026 infrastructure maturity (Portable Computer, Anthropic GA) enables local-hybrid deployment patterns, yet reliability and security barriers remain unresolved prerequisites for mission-critical autonomous research scaling.",
  "history": "- **2024-Q4:** Google launches Gemini Deep Research in Gemini Advanced across 150+ countries as a flagship agentic feature; Perplexity Pro Search demonstrates 50% adoption lift via multi-step reasoning architecture. Industry adoption surveys show 68% of organizations have deployed AI agents, though ROI realization remains below 50%. Accuracy and hallucination challenges identified as key adoption barriers.\n- **2025-Q1:** OpenAI launches Deep Research (Feb 2025) as deep-research-specific agent in ChatGPT Pro, using o3 reasoning model for autonomous multi-step investigation; Google extends Gemini Deep Research to Workspace users. Perplexity benchmarks at 93.9% on SimpleQA. Critical analyses emerge noting agents as \"fallible tools\" rather than expert-level; agentic RAG becomes category's enabling architecture. Category transitions from experimental to mainstream availability across three major platforms.\n- **2025-Q2:** Perplexity reaches 15M active users (50% growth in 3 months); pursues $500M–$1B funding at $18B valuation target. Google I/O announces Flash 2.5 experimental support in Deep Research. Production-ready patterns emerge across platforms (OpenAI, Google, Perplexity, Claude/Anthropic); enterprise deployments adopt steerable workflows for controlled autonomy. Academic surveys document category advances; knowledge cutoff bias and information lag emerge as persistent reliability gaps at scale.\n- **2025-Q3:** Perplexity grows to 30M monthly active users (780M monthly queries) with 66% YoY growth; enterprises across banking, pharma, law adopt for mission-critical research (60% of Pro customers). Google's Gemini Deep Research achieves production status with Workspace integration and usage quotas. However, critical assessments surface: peer-reviewed medical research examines risks to citation integrity and research quality; user reports document hallucinations in current affairs research; MIT study shows 95% of GenAI pilots fail to reach production due to reliability, data quality, and governance barriers.\n- **2025-Q4:** Deep research consolidates around three major platforms (Perplexity, Gemini, OpenAI) with evidence of bounded production use (Skywork case study: 93% citation accuracy, 15-25% speed gains on market research reports). Perplexity's 100M+ interactions show 57% targeting research/learning. However, adoption ceiling persists: Gartner finds only 15% of IT leaders deploying fully autonomous agents (Oct), Deloitte reports 11% production deployment (Dec), and Gemini 3 Pro maintains 88% hallucination rate despite 53% accuracy lead. Domain-specific scientific research shows promise (energy materials agents), but governance, security, and reliability gaps constrain enterprise scaling. Practice matured from experimentation to selective production use but faces unresolved trustworthiness barriers.\n- **2026-Jan:** Mainstream business adoption reaches 67% of enterprises using AI research tools (Gartner); Perplexity achieves 370% YoY user growth and 14.1% market share. However, scaling barriers persist: 62% of organizations experimented with agentic workflows but 70-80% struggle to scale with only 5% achieving ROI. EBU/BBC study reveals 45% of AI research responses contain errors; Gemini exhibits 72% sourcing problems. LangChain releases Deep Agents framework enabling multi-step task decomposition through subagents. Deep research remains viable for exploratory, non-critical use but unsuitable for mission-critical workflows requiring reliability and governance.\n- **2026-Feb:** Platform consolidation continues with Perplexity reaching 33M monthly active users (20.8% research-focused queries) and Google releasing Gemini 3.1 Pro with upgraded Deep Think model. Specialized research agents emerge: DeepMind's Aletheia achieves 91.9% on mathematical reasoning benchmarks and autonomously co-authors published papers. However, HalluHard benchmark reveals state-of-the-art models still hallucinate ~30% in multi-turn conversations even with web search. Princeton-backed analysis shows 18 months of model capability gains have not improved production agent reliability, widening the gap between capability and trustworthiness.\n- **2026-Mar:** Gemini Deep Research reached Workspace GA integrating Gmail, Drive, and Chat with web sources in unified report-generation workflows. Perplexity Computer expanded to desktop (Mac Mini with audit trails), enterprise (Snowflake/Salesforce integration), and Comet Enterprise browser, adding four new APIs (Search, Agent, Embeddings, Sandbox) orchestrating 20 models. Reliability benchmarks remained sobering: a Washington State University study of 700+ scientific hypotheses found ChatGPT at 76.5% accuracy but only 41% consistency across runs; Google's DeepFact paper showed PhD-level experts improve factuality evaluation from 60.8% to 81%+ only when benchmarks are iteratively refined, highlighting benchmark brittleness as a compounding barrier. CrewAI's enterprise survey found 81% claim to be scaling agentic AI but only 11% have agents in production, with 38% stuck in pilots and Gartner predicting 40% of agentic projects cancelled by 2027 — underscoring that deep research capability continues to outpace organisational readiness to deploy and govern it.\n- **2026-Apr:** Product launches and reliability failures defined the month in parallel. Google launched Deep Research Max with Gemini 3.1 Pro, adding MCP support for proprietary data integration and native chart generation for asynchronous enterprise workflows; Perplexity Computer launched multi-model orchestration (Claude Opus reasoning, Gemini Deep Research, GPT-4 drafting) with sub-agent parallelisation and background workflows running hours or days unattended, reaching $450M ARR. Stanford HAI 2026 AI Index documented agents achieving only ~50% of PhD specialist performance on complex research workflows; AlphaLab demonstrated GPT-5.2 and Claude Opus 4.6 autonomously conducting multi-phase research with 4.4x GPU kernel speedup at $150-200 per campaign. The AI Slop Loop case surfaced a systemic failure: fabricated SEO articles were cited as fact by Perplexity within 24 hours, revealing model collapse through poisoned sources. An independent survey of 2,400 enterprises found 97% have deployed AI agents but only 29% see ROI, with 67% suffering data breaches via unapproved tools and 36% lacking governance plans — directly explaining why deep research agents remain at the bleeding edge despite product maturity. Google's Gemini Enterprise named deployments (Macquarie Bank 38% engagement lift, JCOM analysing 100k+ conversations monthly) and an M&A due diligence case study (90% time reduction, months to single afternoon) showed bounded autonomous research delivering measurable value in governed contexts.\n- **2026-Jun:** Production evidence and new benchmarks jointly deepened the capability picture. Harvard/Perplexity study (10,000 matched sessions, Feb-May 2026) documented Perplexity Computer achieving 26-minute autonomous execution per session vs 33 seconds for Search (48× autonomy increase) with 87% task time reduction and 94% cost savings — while simultaneously Perplexity's multi-model routing (20+ models) showed +43 percentage point improvement on BrowseComp (40.7%→83.8%) and +14pp on Humanity's Last Exam, validating orchestration and model diversity as reliability amplifiers. Rigorous benchmarking clarified fundamental limits: ResearchClawBench found Claude Code achieving only 21.5% on 40 real scientific re-discovery tasks; SciAgentArena (Stanford/MIT/Harvard) showed agents effective on well-specified data workflows but failing on multi-constraint optimization and novel insight generation; UC Berkeley research confirmed orchestration, memory, and context governance matter as much as model capability, with multi-agent systems achieving 90.2% improvement over single-agent baselines. CHARM framework formalized cascading hallucination failures in multi-step RAG pipelines — existing detectors catch only 12.8–41.7% of propagation errors while LLM self-correction proves counterproductive. Citation integrity crisis deepened: Zhao et al. audit of 2.5M papers found 146,932 hallucinated citations (fabrication rate 1 per 2,828 papers in 2023 → 1 per 277 in early 2026); arXiv enacted one-year submission bans for unchecked LLM content. Adoption gap persisted: 79% of enterprises claim AI agents but only 11% reach production; Gartner forecasts 40%+ cancellation by 2027. Forrester explicitly validated frontier capability: 'Anthropic has demonstrated multiday research agents' — distinguishing long-horizon operation from shorter task agents. Verdict: measurable production ROI in bounded use cases (due diligence, marketing), yet endemic citation failures, pipeline reliability gaps, and governance barriers block autonomous deployment in mission-critical workflows.\n\n- **2026-Jul:** Production evidence and deployment barriers jointly shaped the landscape mid-month. Perplexity's DRACO benchmark (100 production-grounded tasks, 10 expert domains) established the first standardized evaluation framework: Claude Mythos 5 at 86.4%, Gemini 59.0%, OpenAI o3 52.1%. Harvard/Perplexity empirical study (Feb-May 2026, 100k production sessions) quantified deployment impact: Perplexity Computer achieved 26-minute autonomous execution vs 33 seconds for Search (48× autonomy increase), 87% task time reduction, 84× growth in task volume—strongest production deployment evidence. Google published Deep Research Agent GA documentation with MCP integration and async background workflows. Yet critical deployment barriers emerged: Patronus AI analysis showed agents fail 70-95% in production environments despite 90%+ benchmark accuracy (compounding errors in chained workflows reduce accuracy to 30-50%); Teradata/Wakefield survey of 1,000 tech leaders found only 7% operationalizing autonomous workflows with context fragmentation as systemic barrier (77% have insufficient data governance); TickrWire identified 57% context-accuracy failure rate during step transitions. Citation quality research (PWC benchmark, arXiv:2607.08700) showed all 8 LLM judges degrade on hard cases despite F1 0.908 on relevance—identifying calibration as prerequisite. Architectural maturity signals: Reactify documented convergence across all vendors on three-phase scope/research/write pattern; Perplexity's GLM 5.2 orchestrator (744B MoE) delivers frontier performance at one-third Claude Opus cost. Verdict: real-world deployments at scale with measurable ROI now exist, yet deployment barriers and reliability gaps prevent mainstream enterprise scaling despite two years of vendor competition. Late-month evidence sharpened the production-reliability gap further: SciExplore's peer-reviewed benchmark found autonomous agents scoring under 50% on multi-step scientific research tasks, and \"Autonomy Is the Bug\" formalized why—compounding per-step error (Lusser's law) yields only ~36% success over 20-step tasks—while Gartner data showed 88% of enterprise agent pilots never reach production and a VentureBeat survey of 101 enterprises found 71% of deployed \"agents\" cannot complete autonomous multi-step work. Countervailing production evidence persisted: a consulting-grade Perplexity→Glean→Claude→NotebookLM workflow compressed 6-12 hours of research to 25 minutes ($1,200-3,000 margin recovered per engagement), and Perplexity's sandbox infrastructure reported 1.25M creations weekly—though a companion audit of a 140-task Perplexity Computer run still found 5 substantive errors requiring verification, and practitioner analysis put Perplexity's article-retrieval error rate at 37%.\n- **2026-Aug:** Enterprise deployment continued (Tech Mahindra adopting Perplexity Enterprise Pro for autonomous sales research) alongside sharpened citation-integrity evidence: a 14-LLM evaluation found surface-level citation quality (94% link validity) masks factual accuracy collapse to 39-77%, a four-model deployment-constraints study found no model exceeds 50% citation-existence rate, and a 124-study PRISMA review documented hallucination rates of 28-91% in literature synthesis. Countervailing progress: Google's Science One framework achieved zero phantom references via citing-as-generating architecture, and EviGraph's evidence-graph approach improved claim-support rate by 40%, showing verification-first architectures as an emerging mitigation path. Late-month evidence reinforced both poles: Perplexity shipped a local-first agent harness on NVIDIA DGX Spark (85.4% local-only accuracy at zero per-token cost) and Anthropic moved computer use, browser use, Skills, and Files APIs to GA, while Gemini crossed 1B MAU with Deep Research integrated into Workspace. Yet reliability evidence deepened the capability-trust gap: Princeton/UK AISI's Shadow Evaluation found frontier models rejected on unpublished NeurIPS papers by original authors for weak methodology and zero novel contributions; the WANDR benchmark found the strongest production system reaching only 0.363 soft F1 on 500 professional research tasks; Microsoft Thinkingbox found 65% pass@1 collapsing to 25% pass@20 across 507 workflows; and only 9% of enterprises report progress on autonomous multistep workflows despite 59% claiming agent use. Security assessments concluded agent browsers (e.g., Perplexity Comet) should be blocked by default pending demonstrated resistance to prompt injection.\n- **2026-Sep:** Production deployment deepened in regulated financial services: Google Cloud's Gemini Enterprise financial research agent went live at CME Group and Deutsche Bank (bond portfolio analysis cut from days to under 5 minutes), while eight named financial firms (Core AI, Dojo, SIGNAL IDUNA, AXA) reported measured gains including 25-day onboarding versus 300-day average and 120k CHF fraud detected in week one. Weil Gotshal confirmed autonomous legal research agents operating without continuous human redirection. Countervailing reliability evidence sharpened: independent audits found Gemini 3.8 Flash's citation fabrication rate regressed to 20% from 12.6%, and separate Perplexity citation audits found 34.7-59.8% of citations either unverifiable or sourced from low-authority/coordinated-fake domains, confirming that autonomous research remains vulnerable to source pollution even as production financial deployments scale. Reliability limits sharpened further: DRNOISE found a single misleading document cuts agent accuracy by 66-88pp, IBM measured a 24.4pp consistency gap across repeated runs (halved by injected guidelines), and an independent DR-20 benchmark found no tool cleared 50% requirement coverage even as Gemini's Deep Research agents reached GA via API and OpenResearcher's distilled Nemotron-3-Nano matched GPT-4.1 on BrowseComp-Plus.",
  "historyEntries": [
    {
      "period": "2024-Q4",
      "text": "Google launches Gemini Deep Research in Gemini Advanced across 150+ countries as a flagship agentic feature; Perplexity Pro Search demonstrates 50% adoption lift via multi-step reasoning architecture. Industry adoption surveys show 68% of organizations have deployed AI agents, though ROI realization remains below 50%. Accuracy and hallucination challenges identified as key adoption barriers."
    },
    {
      "period": "2025-Q1",
      "text": "OpenAI launches Deep Research (Feb 2025) as deep-research-specific agent in ChatGPT Pro, using o3 reasoning model for autonomous multi-step investigation; Google extends Gemini Deep Research to Workspace users. Perplexity benchmarks at 93.9% on SimpleQA. Critical analyses emerge noting agents as \"fallible tools\" rather than expert-level; agentic RAG becomes category's enabling architecture. Category transitions from experimental to mainstream availability across three major platforms."
    },
    {
      "period": "2025-Q2",
      "text": "Perplexity reaches 15M active users (50% growth in 3 months); pursues $500M–$1B funding at $18B valuation target. Google I/O announces Flash 2.5 experimental support in Deep Research. Production-ready patterns emerge across platforms (OpenAI, Google, Perplexity, Claude/Anthropic); enterprise deployments adopt steerable workflows for controlled autonomy. Academic surveys document category advances; knowledge cutoff bias and information lag emerge as persistent reliability gaps at scale."
    },
    {
      "period": "2025-Q3",
      "text": "Perplexity grows to 30M monthly active users (780M monthly queries) with 66% YoY growth; enterprises across banking, pharma, law adopt for mission-critical research (60% of Pro customers). Google's Gemini Deep Research achieves production status with Workspace integration and usage quotas. However, critical assessments surface: peer-reviewed medical research examines risks to citation integrity and research quality; user reports document hallucinations in current affairs research; MIT study shows 95% of GenAI pilots fail to reach production due to reliability, data quality, and governance barriers."
    },
    {
      "period": "2025-Q4",
      "text": "Deep research consolidates around three major platforms (Perplexity, Gemini, OpenAI) with evidence of bounded production use (Skywork case study: 93% citation accuracy, 15-25% speed gains on market research reports). Perplexity's 100M+ interactions show 57% targeting research/learning. However, adoption ceiling persists: Gartner finds only 15% of IT leaders deploying fully autonomous agents (Oct), Deloitte reports 11% production deployment (Dec), and Gemini 3 Pro maintains 88% hallucination rate despite 53% accuracy lead. Domain-specific scientific research shows promise (energy materials agents), but governance, security, and reliability gaps constrain enterprise scaling. Practice matured from experimentation to selective production use but faces unresolved trustworthiness barriers."
    },
    {
      "period": "2026-Jan",
      "text": "Mainstream business adoption reaches 67% of enterprises using AI research tools (Gartner); Perplexity achieves 370% YoY user growth and 14.1% market share. However, scaling barriers persist: 62% of organizations experimented with agentic workflows but 70-80% struggle to scale with only 5% achieving ROI. EBU/BBC study reveals 45% of AI research responses contain errors; Gemini exhibits 72% sourcing problems. LangChain releases Deep Agents framework enabling multi-step task decomposition through subagents. Deep research remains viable for exploratory, non-critical use but unsuitable for mission-critical workflows requiring reliability and governance."
    },
    {
      "period": "2026-Feb",
      "text": "Platform consolidation continues with Perplexity reaching 33M monthly active users (20.8% research-focused queries) and Google releasing Gemini 3.1 Pro with upgraded Deep Think model. Specialized research agents emerge: DeepMind's Aletheia achieves 91.9% on mathematical reasoning benchmarks and autonomously co-authors published papers. However, HalluHard benchmark reveals state-of-the-art models still hallucinate ~30% in multi-turn conversations even with web search. Princeton-backed analysis shows 18 months of model capability gains have not improved production agent reliability, widening the gap between capability and trustworthiness."
    },
    {
      "period": "2026-Mar",
      "text": "Gemini Deep Research reached Workspace GA integrating Gmail, Drive, and Chat with web sources in unified report-generation workflows. Perplexity Computer expanded to desktop (Mac Mini with audit trails), enterprise (Snowflake/Salesforce integration), and Comet Enterprise browser, adding four new APIs (Search, Agent, Embeddings, Sandbox) orchestrating 20 models. Reliability benchmarks remained sobering: a Washington State University study of 700+ scientific hypotheses found ChatGPT at 76.5% accuracy but only 41% consistency across runs; Google's DeepFact paper showed PhD-level experts improve factuality evaluation from 60.8% to 81%+ only when benchmarks are iteratively refined, highlighting benchmark brittleness as a compounding barrier. CrewAI's enterprise survey found 81% claim to be scaling agentic AI but only 11% have agents in production, with 38% stuck in pilots and Gartner predicting 40% of agentic projects cancelled by 2027 — underscoring that deep research capability continues to outpace organisational readiness to deploy and govern it."
    },
    {
      "period": "2026-Apr",
      "text": "Product launches and reliability failures defined the month in parallel. Google launched Deep Research Max with Gemini 3.1 Pro, adding MCP support for proprietary data integration and native chart generation for asynchronous enterprise workflows; Perplexity Computer launched multi-model orchestration (Claude Opus reasoning, Gemini Deep Research, GPT-4 drafting) with sub-agent parallelisation and background workflows running hours or days unattended, reaching $450M ARR. Stanford HAI 2026 AI Index documented agents achieving only ~50% of PhD specialist performance on complex research workflows; AlphaLab demonstrated GPT-5.2 and Claude Opus 4.6 autonomously conducting multi-phase research with 4.4x GPU kernel speedup at $150-200 per campaign. The AI Slop Loop case surfaced a systemic failure: fabricated SEO articles were cited as fact by Perplexity within 24 hours, revealing model collapse through poisoned sources. An independent survey of 2,400 enterprises found 97% have deployed AI agents but only 29% see ROI, with 67% suffering data breaches via unapproved tools and 36% lacking governance plans — directly explaining why deep research agents remain at the bleeding edge despite product maturity. Google's Gemini Enterprise named deployments (Macquarie Bank 38% engagement lift, JCOM analysing 100k+ conversations monthly) and an M&A due diligence case study (90% time reduction, months to single afternoon) showed bounded autonomous research delivering measurable value in governed contexts."
    },
    {
      "period": "2026-Jun",
      "text": "Production evidence and new benchmarks jointly deepened the capability picture. Harvard/Perplexity study (10,000 matched sessions, Feb-May 2026) documented Perplexity Computer achieving 26-minute autonomous execution per session vs 33 seconds for Search (48× autonomy increase) with 87% task time reduction and 94% cost savings — while simultaneously Perplexity's multi-model routing (20+ models) showed +43 percentage point improvement on BrowseComp (40.7%→83.8%) and +14pp on Humanity's Last Exam, validating orchestration and model diversity as reliability amplifiers. Rigorous benchmarking clarified fundamental limits: ResearchClawBench found Claude Code achieving only 21.5% on 40 real scientific re-discovery tasks; SciAgentArena (Stanford/MIT/Harvard) showed agents effective on well-specified data workflows but failing on multi-constraint optimization and novel insight generation; UC Berkeley research confirmed orchestration, memory, and context governance matter as much as model capability, with multi-agent systems achieving 90.2% improvement over single-agent baselines. CHARM framework formalized cascading hallucination failures in multi-step RAG pipelines — existing detectors catch only 12.8–41.7% of propagation errors while LLM self-correction proves counterproductive. Citation integrity crisis deepened: Zhao et al. audit of 2.5M papers found 146,932 hallucinated citations (fabrication rate 1 per 2,828 papers in 2023 → 1 per 277 in early 2026); arXiv enacted one-year submission bans for unchecked LLM content. Adoption gap persisted: 79% of enterprises claim AI agents but only 11% reach production; Gartner forecasts 40%+ cancellation by 2027. Forrester explicitly validated frontier capability: 'Anthropic has demonstrated multiday research agents' — distinguishing long-horizon operation from shorter task agents. Verdict: measurable production ROI in bounded use cases (due diligence, marketing), yet endemic citation failures, pipeline reliability gaps, and governance barriers block autonomous deployment in mission-critical workflows."
    },
    {
      "period": "2026-Jul",
      "text": "Production evidence and deployment barriers jointly shaped the landscape mid-month. Perplexity's DRACO benchmark (100 production-grounded tasks, 10 expert domains) established the first standardized evaluation framework: Claude Mythos 5 at 86.4%, Gemini 59.0%, OpenAI o3 52.1%. Harvard/Perplexity empirical study (Feb-May 2026, 100k production sessions) quantified deployment impact: Perplexity Computer achieved 26-minute autonomous execution vs 33 seconds for Search (48× autonomy increase), 87% task time reduction, 84× growth in task volume—strongest production deployment evidence. Google published Deep Research Agent GA documentation with MCP integration and async background workflows. Yet critical deployment barriers emerged: Patronus AI analysis showed agents fail 70-95% in production environments despite 90%+ benchmark accuracy (compounding errors in chained workflows reduce accuracy to 30-50%); Teradata/Wakefield survey of 1,000 tech leaders found only 7% operationalizing autonomous workflows with context fragmentation as systemic barrier (77% have insufficient data governance); TickrWire identified 57% context-accuracy failure rate during step transitions. Citation quality research (PWC benchmark, arXiv:2607.08700) showed all 8 LLM judges degrade on hard cases despite F1 0.908 on relevance—identifying calibration as prerequisite. Architectural maturity signals: Reactify documented convergence across all vendors on three-phase scope/research/write pattern; Perplexity's GLM 5.2 orchestrator (744B MoE) delivers frontier performance at one-third Claude Opus cost. Verdict: real-world deployments at scale with measurable ROI now exist, yet deployment barriers and reliability gaps prevent mainstream enterprise scaling despite two years of vendor competition. Late-month evidence sharpened the production-reliability gap further: SciExplore's peer-reviewed benchmark found autonomous agents scoring under 50% on multi-step scientific research tasks, and \"Autonomy Is the Bug\" formalized why—compounding per-step error (Lusser's law) yields only ~36% success over 20-step tasks—while Gartner data showed 88% of enterprise agent pilots never reach production and a VentureBeat survey of 101 enterprises found 71% of deployed \"agents\" cannot complete autonomous multi-step work. Countervailing production evidence persisted: a consulting-grade Perplexity→Glean→Claude→NotebookLM workflow compressed 6-12 hours of research to 25 minutes ($1,200-3,000 margin recovered per engagement), and Perplexity's sandbox infrastructure reported 1.25M creations weekly—though a companion audit of a 140-task Perplexity Computer run still found 5 substantive errors requiring verification, and practitioner analysis put Perplexity's article-retrieval error rate at 37%."
    },
    {
      "period": "2026-Aug",
      "text": "Enterprise deployment continued (Tech Mahindra adopting Perplexity Enterprise Pro for autonomous sales research) alongside sharpened citation-integrity evidence: a 14-LLM evaluation found surface-level citation quality (94% link validity) masks factual accuracy collapse to 39-77%, a four-model deployment-constraints study found no model exceeds 50% citation-existence rate, and a 124-study PRISMA review documented hallucination rates of 28-91% in literature synthesis. Countervailing progress: Google's Science One framework achieved zero phantom references via citing-as-generating architecture, and EviGraph's evidence-graph approach improved claim-support rate by 40%, showing verification-first architectures as an emerging mitigation path. Late-month evidence reinforced both poles: Perplexity shipped a local-first agent harness on NVIDIA DGX Spark (85.4% local-only accuracy at zero per-token cost) and Anthropic moved computer use, browser use, Skills, and Files APIs to GA, while Gemini crossed 1B MAU with Deep Research integrated into Workspace. Yet reliability evidence deepened the capability-trust gap: Princeton/UK AISI's Shadow Evaluation found frontier models rejected on unpublished NeurIPS papers by original authors for weak methodology and zero novel contributions; the WANDR benchmark found the strongest production system reaching only 0.363 soft F1 on 500 professional research tasks; Microsoft Thinkingbox found 65% pass@1 collapsing to 25% pass@20 across 507 workflows; and only 9% of enterprises report progress on autonomous multistep workflows despite 59% claiming agent use. Security assessments concluded agent browsers (e.g., Perplexity Comet) should be blocked by default pending demonstrated resistance to prompt injection."
    },
    {
      "period": "2026-Sep",
      "text": "Production deployment deepened in regulated financial services: Google Cloud's Gemini Enterprise financial research agent went live at CME Group and Deutsche Bank (bond portfolio analysis cut from days to under 5 minutes), while eight named financial firms (Core AI, Dojo, SIGNAL IDUNA, AXA) reported measured gains including 25-day onboarding versus 300-day average and 120k CHF fraud detected in week one. Weil Gotshal confirmed autonomous legal research agents operating without continuous human redirection. Countervailing reliability evidence sharpened: independent audits found Gemini 3.8 Flash's citation fabrication rate regressed to 20% from 12.6%, and separate Perplexity citation audits found 34.7-59.8% of citations either unverifiable or sourced from low-authority/coordinated-fake domains, confirming that autonomous research remains vulnerable to source pollution even as production financial deployments scale. Reliability limits sharpened further: DRNOISE found a single misleading document cuts agent accuracy by 66-88pp, IBM measured a 24.4pp consistency gap across repeated runs (halved by injected guidelines), and an independent DR-20 benchmark found no tool cleared 50% requirement coverage even as Gemini's Deep Research agents reached GA via API and OpenResearcher's distilled Nemotron-3-Nano matched GPT-4.1 on BrowseComp-Plus."
    }
  ],
  "historyFallback": false,
  "lastUpdated": "2026-09-23",
  "domain": {
    "id": "research-analysis",
    "label": "Research & Knowledge",
    "icon": "🔬"
  },
  "url": "https://www.thestateofplay.ai/practice/multi-step-autonomous-deep-research",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "generatedAt": "2026-10-01"
}