{
  "name": "AX evidence register",
  "url": "https://agentexperience.tech/evidence/",
  "notice": "Licence to be confirmed before the source repository is published. To point at a record, cite \"agentexperience.tech, AX evidence register\" with the record ID. Quoted titles, claims and figures belong to their original authors; cite the original source for any number you use.",
  "schema": "https://agentexperience.tech/evidence.schema.json",
  "description": "Measured and observed findings about agent experience, each with its source, agent profile, evidence class and verification status.",
  "last_updated": "2026-10-08",
  "count": 40,
  "counts": {
    "evidence_class": {
      "preprint": 27,
      "independent-measurement": 4,
      "vendor-measurement": 5,
      "vendor-claim": 4
    },
    "finding_type": {
      "measurement": 24,
      "observation": 6,
      "negative-result": 4,
      "null-result": 2,
      "vendor-claim": 4
    },
    "verification": {
      "abstract-only": 27,
      "verified-live": 13
    }
  },
  "pattern_vocabulary": [
    "discovery",
    "selection",
    "description",
    "schema-enum",
    "error-recovery",
    "false-success",
    "exit-codes",
    "non-interactive",
    "dry-run",
    "idempotency",
    "approval",
    "confirmation",
    "auth-scopes",
    "secrets",
    "claim-later-onboarding",
    "context-budget",
    "dynamic-tools",
    "code-mode",
    "docs-for-agents",
    "llms-txt",
    "ard",
    "server-cards",
    "agents-md",
    "skills",
    "drift",
    "handoff",
    "evaluation",
    "measurement",
    "propensity",
    "cost",
    "prompt-injection",
    "webmcp"
  ],
  "records": [
    {
      "id": "EV-0001",
      "title": "An enum in the schema ends silent failures from example-only vocabularies",
      "claim": "SilentProbe (preprint) reports that a vocabulary a parameter description only exemplified (\"e.g.\") was missed on 88 of 88 attempts across twelve models, and that promoting it into the schema cut the failure to 0 of 89.",
      "patterns": [
        "schema-enum",
        "description",
        "false-success"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Attempts that missed the required vocabulary value",
          "baseline": "Vocabulary only exemplified in the description (\"e.g.\"): 88 of 88 missed",
          "treatment": "Vocabulary promoted into the schema: 0 of 89 missed",
          "direction": "decrease",
          "magnitude": "100% of attempts to 0%",
          "n": "88 and 89 attempts; twelve models across eight families"
        },
        {
          "metric": "Correct use when the full vocabulary is written out in the description",
          "baseline": null,
          "treatment": "88 to 91% correct",
          "direction": "not-applicable",
          "magnitude": "88 to 91%",
          "n": "Twelve models across eight families"
        },
        {
          "metric": "Agent behaviour after a silent failure, full agent loop",
          "baseline": null,
          "treatment": "Detected the failure in 12% of cases, repaired it in 0%, told the user a false negative in 41%, invented a figure in 12%",
          "direction": "not-applicable",
          "magnitude": "12% detected, 0% repaired, 41% false negative, 12% invented figure",
          "n": "Not stated in the abstract"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Twelve models across eight families. The abstract does not name them.",
        "harness": "The authors' agent loop, calling live commercial endpoints through one aggregation layer (Monid)",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "The paper states that the authors are affiliated with Monid, Inc., the aggregation layer used as the measurement instrument.",
      "source": {
        "url": "https://arxiv.org/abs/2609.00035",
        "title": "SilentProbe: Measuring Silent Failure in Production APIs Used as Agent Tools",
        "publisher": "arXiv",
        "authors": [
          "Zongrong Li",
          "Shengkun Ye",
          "Feiyou Guo",
          "Zuoyou Dang"
        ],
        "published": "2026-08-29",
        "identifier": "arXiv:2609.00035"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked. The competing-interest statement was read in the arXiv HTML full text on the same date.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Endpoints were reached through one aggregation layer run by an organisation the authors are affiliated with.",
        "The abstract does not name the twelve models, so the result cannot yet be tied to a model generation."
      ],
      "implications": "When a parameter accepts a closed set of values, put the set in the schema as an enum instead of giving examples in prose. A schema can be enforced by something other than the model; a description cannot.",
      "related": [
        "READ-05",
        "RECOVER-02"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0002",
      "title": "Constraints stated only in prose produce silent failures on live APIs",
      "claim": "SilentProbe (preprint) reports that only 7.5% of 2,501 public OpenAPI documents declare an enum, and that on live endpoints machine-checkable constraints returned an honest error in 111 of 111 cases while prose-only constraints failed silently in 44 of 61.",
      "patterns": [
        "schema-enum",
        "error-recovery",
        "false-success"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Share of OpenAPI documents that encode constraints",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "7.5% declare an enum; 15.2% declare any machine-checkable constraint; 40.1% state at least one constraint in prose that the schema does not encode",
          "n": "721,320 parameters across 2,501 independently published OpenAPI documents"
        },
        {
          "metric": "Responses to schema-derived invalid requests that fail silently (HTTP 200 with a parsable body)",
          "baseline": "Prose-only constraint: 44 of 61 failed silently",
          "treatment": "Machine-checkable constraint: 0 of 111 failed silently (111 of 111 honest errors)",
          "direction": "decrease",
          "magnitude": "About 72% silent to 0% silent (p = 2e-13)",
          "n": "219 perturbations against live endpoints of 27 vendors"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: a static audit of OpenAPI documents plus scripted requests to live endpoints.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "The paper states that the authors are affiliated with Monid, Inc., the aggregation layer through which the live endpoints were reached.",
      "source": {
        "url": "https://arxiv.org/abs/2609.00035",
        "title": "SilentProbe: Measuring Silent Failure in Production APIs Used as Agent Tools",
        "publisher": "arXiv",
        "authors": [
          "Zongrong Li",
          "Shengkun Ye",
          "Feiyou Guo",
          "Zuoyou Dang"
        ],
        "published": "2026-08-29",
        "identifier": "arXiv:2609.00035"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "The abstract says constraint form \"predicts\" honesty; it is an association across endpoints, not a controlled change to one API.",
        "Live probes went through one aggregation layer to 27 vendors."
      ],
      "implications": "An API that states its rules only in prose can answer a bad request with an empty success, which an agent cannot tell apart from \"no results\". Encode the rules in the schema so the server can return an error the agent can act on.",
      "related": [
        "READ-05",
        "RECOVER-01",
        "RECOVER-02"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0003",
      "title": "Error text that names the next tool lifts recovery",
      "claim": "A preprint testing five OpenAI models reports that an expired-credential error naming a terminal command left 45% of tasks recovered, naming the server's login tool instead raised recovery to 84%, and on rate limits naming the call to repeat raised recovery from 6% to 88%.",
      "patterns": [
        "error-recovery",
        "description"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Tasks recovered after an expired-credential error",
          "baseline": "Next step names a terminal command: 45% recovered",
          "treatment": "Next step names the server's login tool: 84% recovered",
          "direction": "increase",
          "magnitude": "+39 percentage points",
          "n": "Five OpenAI models on Berkeley Function Calling Leaderboard tasks; sample counts are not in the abstract"
        },
        {
          "metric": "Tasks recovered after a rate-limit error",
          "baseline": "GitHub's \"Wait before retrying.\": 6% recovered",
          "treatment": "Step names the call to repeat: 88% recovered",
          "direction": "increase",
          "magnitude": "+82 percentage points",
          "n": "As above"
        },
        {
          "metric": "Recovery lost to a developer-addressed step, by model generation",
          "baseline": "GPT-5.5: 18 points lost",
          "treatment": "GPT-6 Astra: 69 points lost",
          "direction": "increase",
          "magnitude": "Loss grew from 18 to 69 points with the newer model",
          "n": "As above"
        },
        {
          "metric": "Recovery on expired credentials with an agent-side fix",
          "baseline": "Terminal-command step left in place: 45%",
          "treatment": "One-sentence prompt deletes the step before the model reads it: 82%",
          "direction": "increase",
          "magnitude": "+37 percentage points",
          "n": "As above"
        },
        {
          "metric": "Prevalence of next steps in MCP server error messages",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "949 of 3,001 messages give a next step; on credential errors 62 of 67 steps ask for a terminal command, configuration change or web page; on rate limits 20 of 30 say to wait without naming the call",
          "n": "150 widely used MCP servers"
        }
      ],
      "agent_profile": {
        "models": [
          "GPT-5.5",
          "GPT-6 Astra"
        ],
        "models_note": "Five OpenAI models; the abstract names these two.",
        "harness": "Berkeley Function Calling Leaderboard tasks; the agents act only through the tools, with no terminal",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.35381",
        "title": "MCP Error Messages Written for Developers Hurt the Most Capable Agents Most",
        "publisher": "arXiv",
        "authors": [
          "Xiaonan Xu",
          "Wenjing Wu"
        ],
        "published": "2026-09-28",
        "version": "v2, revised 2026-09-29",
        "identifier": "arXiv:2609.35381"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "OpenAI models only, so transfer to other model families is untested.",
        "The agent has no terminal by construction, which is the condition under which developer-addressed steps fail."
      ],
      "implications": "Write error messages for a caller that can only call your tools: name the tool or call that fixes the problem, not a command, a settings page or \"wait\". Newer models followed the step more literally, so a wrong step cost more.",
      "related": [
        "RECOVER-02",
        "READ-04"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0004",
      "title": "Naming recovery tools is the active ingredient in failure receipts",
      "claim": "Outcome Monitors (preprint) reports that receipts naming a violated outcome and the public recovery tools raised ToolMaze completion from 10.9% to 28.1% across four models, and that removing the list of recovery tools eliminated the gain.",
      "patterns": [
        "error-recovery",
        "false-success"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "ToolMaze task completion with injected silent failures",
          "baseline": "No outcome monitor: 10.9%",
          "treatment": "Outcome monitor receipts: 28.1%",
          "direction": "increase",
          "magnitude": "+17.2 percentage points",
          "n": "Four models in two provider families, replicated in a third family"
        },
        {
          "metric": "Completion gain when the recovery-tool list is removed from the receipt",
          "baseline": "Receipt with recovery-tool list",
          "treatment": "Receipt without it: measured gain eliminated; restoring the list recovered it",
          "direction": "decrease",
          "magnitude": "Gain eliminated",
          "n": "Separate ToolMaze controls"
        },
        {
          "metric": "Completion when diagnostic detail or timing is varied",
          "baseline": null,
          "treatment": "No detectable difference",
          "direction": "no-change",
          "magnitude": "No detectable difference",
          "n": "Separate ToolMaze controls"
        },
        {
          "metric": "tau-bench retail completion",
          "baseline": "Without monitors",
          "treatment": "With monitors: +14.0 and +12.0 points on two tiers",
          "direction": "increase",
          "magnitude": "+14.0 and +12.0 percentage points",
          "n": "Two tiers"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Four models in two provider families, replicated in a third. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.19303",
        "title": "Outcome Monitors: Recovery Affordances for Silent Tool Failures",
        "publisher": "arXiv",
        "authors": [
          "Sugam Panthi",
          "Rabab Abdelfattah"
        ],
        "published": "2026-08-19",
        "identifier": "arXiv:2608.19303"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Failures were injected; detection outside the mined contract vocabulary fell to 46% on a suite from a published incident taxonomy."
      ],
      "implications": "When a tool result looks wrong, the most useful thing to send back is the name of the tool that can check or fix it. Extra diagnostic detail alone did not help in these tests.",
      "related": [
        "RECOVER-02",
        "RECOVER-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0005",
      "title": "Idempotency keys cut duplicate writes from 28% to 4%",
      "claim": "LIMBO (preprint) reports that offering an idempotency key on every write cut duplicate side effects from 28% to 4% of episodes because agents use keys when they exist, and that agents reported success in 90% of the episodes in which they had duplicated an effect.",
      "patterns": [
        "idempotency",
        "false-success",
        "error-recovery"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Episodes with a duplicated side effect",
          "baseline": "No idempotency key offered: 28%",
          "treatment": "Idempotency key on every write: 4%",
          "direction": "decrease",
          "magnitude": "28% to 4%",
          "n": "25,930 episodes; nine models; three production agent harnesses; twelve fault modes"
        },
        {
          "metric": "Duplicates by frontier models instructed to act exactly once",
          "baseline": "Lost acknowledgement, read-back possible: 0.5%",
          "treatment": "Request still in flight: 56%; transport delivered it twice: 74%",
          "direction": "increase",
          "magnitude": "0.5% when read-back can reveal the outcome, 56% and 74% when it cannot",
          "n": "As above"
        },
        {
          "metric": "Success reported in episodes where an effect was duplicated",
          "baseline": null,
          "treatment": "90% of such episodes",
          "direction": "not-applicable",
          "magnitude": "90%",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Nine recent models. The abstract does not name them.",
        "harness": "Three production agent harnesses (not named in the abstract); the paper reports the harness barely mattered",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.29095",
        "title": "Where Does Exactly-Once Live? Model, Harness, and Tool-Contract Effects on Duplicate Side Effects in LLM Agents",
        "publisher": "arXiv",
        "authors": [
          "Jiapeng Li"
        ],
        "published": "2026-09-24",
        "identifier": "arXiv:2609.29095"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Simulated services in a deterministic sandbox, graded against a ledger.",
        "Single author."
      ],
      "implications": "If a write can time out, accept an idempotency key and say in the tool description that a retry with the same key is safe. Do not rely on the agent's report that a write happened once; check your own ledger.",
      "related": [
        "ACT-03",
        "RECOVER-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0006",
      "title": "An evidence contract cuts false-success reports after tool failures",
      "claim": "Failure-Transparent Agents (preprint) reports that after a required tool failed, six models falsely reported success in 22.8% of responses by default, 9.3% with a transparency instruction and 0.8% with a structured evidence contract.",
      "patterns": [
        "false-success",
        "handoff",
        "evaluation"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "False-success rate after a tool failure",
          "baseline": "Baseline policy: 22.8%",
          "treatment": "Transparency instruction: 9.3%; structured evidence contract: 0.8%",
          "direction": "decrease",
          "magnitude": "22.8% to 0.8%",
          "n": "100 tasks; six models; three response policies; 3,600 human-annotated responses"
        },
        {
          "metric": "Fabricated-detail rate",
          "baseline": "28.3%",
          "treatment": "Transparency instruction: 14.3%; evidence contract: 0.8%",
          "direction": "decrease",
          "magnitude": "28.3% to 0.8%",
          "n": "As above"
        },
        {
          "metric": "Useful responses",
          "baseline": "74.9%",
          "treatment": "Transparency instruction: 89.2%; evidence contract: 98.8%",
          "direction": "increase",
          "magnitude": "74.9% to 98.8%",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Six models. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.35732",
        "title": "Failure-Transparent Agents: Benchmarking Post-Failure Reporting in Tool-Using Language Models",
        "publisher": "arXiv",
        "authors": [
          "Junru Zhu",
          "Shiming Xie",
          "Aime Lu Fan Chen",
          "Xiaoqing Ding",
          "Chunxin Tang",
          "Ruoyu Qi",
          "Yulang Fei"
        ],
        "published": "2026-09-28",
        "identifier": "arXiv:2609.35732"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Blocked-task benchmark with fixed failure traces; the paper describes the result as an association within it."
      ],
      "implications": "Ask for the agent's final report in a fixed structure that cites evidence for each claimed outcome, rather than asking it to be honest. Your own record of what happened should stay the source of truth.",
      "related": [
        "HANDBACK-01",
        "RECOVER-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0007",
      "title": "Browser agents declare success on most of their failures",
      "claim": "BreakingWeb (preprint) reports that controlled website changes cut browser-agent pass rates by 22.9% on average, against a 10.0% first-attempt loss for humans, and that 75% of the agents' failures ended with a declared success although the required change never happened.",
      "patterns": [
        "false-success",
        "evaluation"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Pass rate under a controlled environment change",
          "baseline": "Clean task",
          "treatment": "Intervention: agents lose 22.9% on average; humans lose 10.0% on a first attempt and 5.7% after one familiarisation attempt",
          "direction": "decrease",
          "magnitude": "22.9% average loss for agents",
          "n": "519 clean/intervention task pairs; seven self-hosted websites; 29 intervention families; six browser-use agents and three GUI-only agents"
        },
        {
          "metric": "Agent failures that end with a declared success",
          "baseline": null,
          "treatment": "75% of failures",
          "direction": "not-applicable",
          "magnitude": "75%",
          "n": "Failures of the six browser-use agents"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Six strong browser-use agents and three GUI-only agents. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.35814",
        "title": "Constructing Challenging Browser-Use Tasks by Controlled Environment Interventions",
        "publisher": "arXiv",
        "authors": [
          "Xunjian Yin",
          "Tianchen Guan",
          "Jinao Wang",
          "Weili Cao",
          "Daisy Xinlei Lin",
          "Royce Cheng-Yue",
          "Keagan Long",
          "Kyle Wong",
          "Bhuwan Dhingra",
          "Xiangjun Wang",
          "Shuyan Zhou"
        ],
        "published": "2026-09-20",
        "identifier": "arXiv:2609.35814"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Self-hosted sites with interventions designed to be detectable and recoverable."
      ],
      "implications": "Treat an agent's \"done\" as a claim to check. Give agents and the people they work for a way to confirm the change, such as a confirmation page, a record or an API read.",
      "related": [
        "READ-03",
        "HANDBACK-01"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0008",
      "title": "Declared completion and logged completion differ by up to 41 points",
      "claim": "WebPageBench (preprint) reports a gap of up to 41 points between the tasks web agents declared finished and the tasks the site's own event log confirmed, with one configuration declaring every task finished while meeting the conditions on 59%.",
      "patterns": [
        "false-success",
        "evaluation",
        "measurement"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Gap between declared and log-confirmed task completion",
          "baseline": "Declared finished",
          "treatment": "Confirmed by the event log",
          "direction": "decrease",
          "magnitude": "Up to 41 points; one configuration declared 100% and met the conditions on 59%",
          "n": "152 tasks; public leaderboard of 24 model-harness pairs"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Six browser/DOM harness configurations and five screenshot-only GUI-agent families. The abstract does not name the models.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.35026",
        "title": "WebPageBench: Event-Level Verification and Controlled UI-Variant Generation for Web Agents",
        "publisher": "arXiv",
        "authors": [
          "Anton Emelyanov",
          "Maria Tikhonova",
          "Zaven Martirosian",
          "Sergei Averkiev",
          "Alena Fenogenova"
        ],
        "published": "2026-09-28",
        "identifier": "arXiv:2609.35026"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Six instrumented mock sites with brand identifiers removed."
      ],
      "implications": "Measure success from your own system's events, not from what the agent says. Emit an event for each step that completes so it can be checked.",
      "related": [
        "READ-03",
        "HANDBACK-01"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0009",
      "title": "Tools as code match or beat JSON tool calls for most models",
      "claim": "The Bitter Lesson of Tool Calling (preprint) reports that exposing tools as typed Python stubs called through code matched or beat native JSON tool calling for 11 of 14 models on BFCL v4, and for 13 of 14 under parallel fan-out.",
      "patterns": [
        "code-mode"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "BFCL v4 accuracy, programmatic versus native JSON tool calling",
          "baseline": "Native JSON tool calling",
          "treatment": "Programmatic tool calling matches or exceeds it in 11 of 14 models; GPT-5.6 family +10.6%",
          "direction": "increase",
          "magnitude": "11 of 14 models; best family +10.6%",
          "n": "14 language models"
        },
        {
          "metric": "Accuracy under parallel fan-out",
          "baseline": "Native JSON tool calling",
          "treatment": "Matches or outperforms in 13 of 14 models",
          "direction": "increase",
          "magnitude": "13 of 14 models",
          "n": "14 language models"
        },
        {
          "metric": "Accuracy under context-rot conditions",
          "baseline": "Native JSON: degrades 2.3% on average",
          "treatment": "Programmatic: holds stable",
          "direction": "mixed",
          "magnitude": "JSON loses 2.3% on average; programmatic stable",
          "n": "14 language models"
        }
      ],
      "agent_profile": {
        "models": [
          "GPT-5.6 family"
        ],
        "models_note": "14 models across current and prior generations; the abstract names only the GPT-5.6 family.",
        "harness": "BFCL v4 (Berkeley Function Calling Leaderboard)",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.06370",
        "title": "The Bitter Lesson of Tool Calling",
        "publisher": "arXiv",
        "authors": [
          "Ishan Patel",
          "Sahil Sen",
          "Elias Lumer",
          "Vamse Kumar Subbiah"
        ],
        "published": "2026-08-06",
        "identifier": "arXiv:2608.06370"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "One benchmark with simulated tools.",
        "Measures call accuracy only; safety, approval and review cost of generated code were not measured."
      ],
      "implications": "Code-style access to your tools, such as typed stubs or an SDK in a sandbox, is a reasonable option next to one call per tool. Weigh it against the harder job of reviewing and approving generated code, which this study did not measure.",
      "related": [
        "ACT-07"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0010",
      "title": "Agent scaffolding, not MCP versus CLI, drove cost",
      "claim": "A controlled comparison (preprint) found that on one git task the agent scaffolding drove cost more than MCP versus CLI: CLI-only scaffoldings were 5.0x to 28x cheaper, and paired MCP-to-CLI cost ratios ranged from 0.43x to 29x.",
      "patterns": [
        "cost",
        "evaluation",
        "measurement"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Cost per run, scaffoldings without MCP support versus those with it (CLI runs only)",
          "baseline": "Five scaffoldings that support MCP",
          "treatment": "Two scaffoldings with no MCP support",
          "direction": "decrease",
          "magnitude": "5.0x to 28x cheaper",
          "n": "Seven scaffoldings; five models; one fixed task of six git operations"
        },
        {
          "metric": "Paired MCP-to-CLI cost ratio",
          "baseline": null,
          "treatment": null,
          "direction": "mixed",
          "magnitude": "Thirteen strictly paired ratios span 0.43x to 29x",
          "n": "13 pairs"
        },
        {
          "metric": "Share of spend that bought no completed work",
          "baseline": "CLI runs: 2.2%",
          "treatment": "MCP runs: 12.9%",
          "direction": "increase",
          "magnitude": "2.2% to 12.9%; failure frequency was the same for both",
          "n": "Original runs and repetitions"
        },
        {
          "metric": "Cost variation of a local 27-billion-parameter model across scaffoldings",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "139x",
          "n": "Seven scaffoldings"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Five language models, including a local 27-billion-parameter model. The abstract does not name them.",
        "harness": "Seven agent scaffoldings (not named in the abstract)",
        "extensions": [
          "MCP server for the git service (in MCP conditions)"
        ],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.08654",
        "title": "The Scaffolding Matters More Than the Interface: A Controlled Comparison of MCP and CLI Tool Use Across Seven Agent Scaffoldings, Five Language Models, and One Software Task",
        "publisher": "arXiv",
        "authors": [
          "Marc Alier Forment",
          "María José Casañ Guerrero",
          "Francisco José García-Peñalvo",
          "Juanan Pereira"
        ],
        "published": "2026-08-09",
        "identifier": "arXiv:2608.08654"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "One task against one kind of service.",
        "Agents often ignored the interface they were assigned, so some comparisons mix interfaces."
      ],
      "implications": "Do not claim that MCP or a CLI is cheaper in general; the client running the agent mattered more here. When you test an interface, check what the agent actually called, because agents often ignored the one they were given.",
      "related": [
        "ACT-07"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0011",
      "title": "Coding agents mostly read instruction files, not docs sites",
      "claim": "An observational study of 557 agentic coding sessions (preprint) found that instruction files and working notes made up 60.5% of agents' documentation interactions, against 10.6% for classical technical documentation and 1.3% for API references.",
      "patterns": [
        "docs-for-agents",
        "agents-md"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Share of documentation interactions by document type",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "Instruction files and working notes 60.5%; classical technical documentation 10.6%; API references 1.3%",
          "n": "557 sessions; 94,813 development events; 3,033 documentation interactions"
        },
        {
          "metric": "What prompts a documentation consultation",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "Self-initiated 70.2%; failure-driven 7.5%",
          "n": "As above"
        },
        {
          "metric": "Behavioural support for \"actionable\" and \"verifiable\" documentation",
          "baseline": null,
          "treatment": "Both properties lack consistent behavioural support",
          "direction": "not-applicable",
          "magnitude": "No consistent support",
          "n": "As above, plus 33,097 agentic pull requests"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Sessions from the public SWE-chat dataset and pull requests from AIDev; agents and models vary and are not named in the abstract.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.20195",
        "title": "From Agent Behaviour to Agent-Friendly Documentation: An Empirical Study of How Coding Agents Discover, Read, and Write Technical Documentation",
        "publisher": "arXiv",
        "authors": [
          "Zhijun Gao",
          "Jing Chen"
        ],
        "published": "2026-08-20",
        "identifier": "arXiv:2608.20195"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Observational: it shows what agents read, not what helps them."
      ],
      "implications": "For coding agents, the document most likely to be read is the repository instruction file, such as AGENTS.md, not your docs site. Do not assume that making docs \"actionable\" or \"verifiable\" changes what agents do; this study found no consistent support for either.",
      "related": [
        "READ-10"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0012",
      "title": "Compact documentation did not help when the source was present",
      "claim": "Across two model families and ten repositories, with a positive control, a preprint found that neither compact natural-language documentation nor retrieved context helped coding agents resolve issues better than the issue alone when the source code was present.",
      "patterns": [
        "docs-for-agents"
      ],
      "finding_type": "negative-result",
      "effect": [
        {
          "metric": "Issue resolution with documentation versus the issue alone (source present)",
          "baseline": "Issue alone",
          "treatment": "Static compact documentation, or retrieved context",
          "direction": "no-change",
          "magnitude": "Neither beat the issue alone; a positive control confirmed the evaluation could detect a real gain",
          "n": "Two model families; ten repositories"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Two model families. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.31587",
        "title": "Compact Documentation for Coding Agents: A Benchmark, an Optimizer, and Why It Does Not Transfer",
        "publisher": "arXiv",
        "authors": [
          "Md Shohel Arman",
          "Igor Molybog"
        ],
        "published": "2026-09-25",
        "identifier": "arXiv:2609.31587"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Applies where the agent can read the source; the paper characterises the boundary where documentation helps, which the abstract does not detail."
      ],
      "implications": "Documentation earns its place where the agent cannot read the source, such as a hosted API. Where the code is already in the workspace, extra summaries may add cost without helping.",
      "related": [],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0013",
      "title": "FAQ blocks and structured data showed no citation effect within a domain",
      "claim": "An observational study of about 2 million AI-engine citations (preprint) found that FAQ blocks, structured data and Core Web Vitals had positive effects on citation in pooled data that reversed or fell to zero once domain fixed effects were applied.",
      "patterns": [
        "discovery",
        "measurement"
      ],
      "finding_type": "null-result",
      "effect": [
        {
          "metric": "Effect of the standard answer-engine checklist (FAQ blocks, structured data, Core Web Vitals) on citation frequency",
          "baseline": "Pooled data: positive effects",
          "treatment": "With domain fixed effects: effects reverse or collapse to zero",
          "direction": "no-change",
          "magnitude": "Reverse or zero within domain (Simpson's paradox)",
          "n": "About 2 million citations; 10,000 pages from nineteen B2B SaaS workspaces; six months"
        },
        {
          "metric": "Prompt-content alignment as a predictor of citation",
          "baseline": null,
          "treatment": null,
          "direction": "increase",
          "magnitude": "beta = +0.37, 95% CI [+0.33, +0.41]; the dominant page-level predictor",
          "n": "As above"
        },
        {
          "metric": "Domain-level AI authority versus the strongest non-alignment page feature",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "About six times larger in mean absolute SHAP value",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [
          "ChatGPT",
          "Claude",
          "Google AI",
          "Gemini"
        ],
        "models_note": "Four commercial answer engines, as products; model versions not stated in the abstract.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": "Six months (dates not given in the abstract)"
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The data come from nineteen B2B SaaS workspaces and the abstract does not say who operates them; the full text was not checked.",
      "source": {
        "url": "https://arxiv.org/abs/2609.35077",
        "title": "What Drives Citations in Production Large Language Models? An Observational Multi-Method Study of Two Million AI Citations Across Ten Thousand Web Pages",
        "publisher": "arXiv",
        "authors": [
          "Ben Moore",
          "Liam Dunne"
        ],
        "published": "2026-09-28",
        "identifier": "arXiv:2609.35077"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Observational; causal claims are limited.",
        "B2B SaaS pages only.",
        "Measures citation, not agent task success."
      ],
      "implications": "Adding FAQ markup or structured data is not, on this evidence, a reliable way to get cited by answer engines. Content that matches what people actually ask carried far more signal.",
      "related": [
        "READ-09"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0014",
      "title": "Allow/ask/never policies blocked less overreach than per-action approval",
      "claim": "In a study of 113 people without software backgrounds (preprint), user-authored allow/ask/never policies blocked 20.1 percentage points less agent overreach than per-action approval, partly because participants chose 'ask' for 114 of 140 rules and then approved most overreach at runtime.",
      "patterns": [
        "approval",
        "confirmation"
      ],
      "finding_type": "negative-result",
      "effect": [
        {
          "metric": "Overreach blocked, user-authored policy versus per-action human approval",
          "baseline": "Per-action human-in-the-loop approval",
          "treatment": "User-authored allow/ask/never policy",
          "direction": "decrease",
          "magnitude": "-20.1 percentage points, 95% CI [-32.1, -8.1]; -14.5 points versus automated per-action review",
          "n": "113 participants; 18-action simulated day with 7 overreach actions"
        },
        {
          "metric": "Runtime prompts per participant",
          "baseline": "Per-action approval: 18.0",
          "treatment": "Policy: 10.9",
          "direction": "decrease",
          "magnitude": "18.0 to 10.9; total intervention time not reliably lower once rule setup is included",
          "n": "As above"
        },
        {
          "metric": "How overreach ran under the policy condition",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "133 of 148 executed overreach actions followed human approval; 15 ran under \"allow\" rules",
          "n": "Exploratory analysis"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "A language model mapped actions to consequence categories; it is not named in the abstract. Participants supervised a simulated day, not a live agent.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.27443",
        "title": "Do User-Authored Permission Policies Improve Protection Against AI Agent Overreach?",
        "publisher": "arXiv",
        "authors": [
          "Ting Yan"
        ],
        "published": "2026-08-27",
        "identifier": "arXiv:2608.27443"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Pre-defined simulation with non-developers.",
        "Some analyses are exploratory, as the abstract states."
      ],
      "implications": "Standing rules do not remove the approval problem if people keep choosing 'ask'. An approval request should show clearly when an action goes beyond what the person originally asked for.",
      "related": [
        "ACT-02",
        "HANDBACK-04"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0015",
      "title": "Approvals that outlive their task raise attack success",
      "claim": "A preprint reports that approvals persisted beyond the context that justified them raised prompt-injection attack success by up to 35.1 percentage points on 508 AgentDojo cases, and by 24.9 points on average in live tests on three production coding agents.",
      "patterns": [
        "approval",
        "prompt-injection"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Attack success rate with residual authority versus a fresh authorisation state",
          "baseline": "Fresh authorisation state",
          "treatment": "Residual authority from earlier approvals",
          "direction": "increase",
          "magnitude": "Up to +35.1 percentage points",
          "n": "508 AgentDojo attack cases across six LLM families"
        },
        {
          "metric": "Attack success in live context-rebinding attacks",
          "baseline": "Without residual-authority replay",
          "treatment": "With replay",
          "direction": "increase",
          "magnitude": "+24.9 percentage points on average",
          "n": "55 Terminal-Bench cases on three production coding agents"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Six LLM families on AgentDojo and three production coding agents. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.33910",
        "title": "When Consent Outlives Context: Residual Authority Replay in Long-Lived Agents",
        "publisher": "arXiv",
        "authors": [
          "Zhihao Zhang",
          "Chao Wang",
          "Rujia Li",
          "Qingze Wang",
          "Xiaoyan Sun",
          "Jun Dai"
        ],
        "published": "2026-09-27",
        "identifier": "arXiv:2609.33910"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Adversarial setting designed by the authors; production-derived authorisation semantics."
      ],
      "implications": "Let an approval expire with the task or context that justified it. A standing \"yes\" is authority an attacker can reuse later.",
      "related": [
        "ACT-02",
        "ACT-05"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0016",
      "title": "Approval records omit the effects a command goes on to trigger",
      "claim": "A preprint reports that coding-agent approval records name the approved command but omit effects its workflow exercises: across 111 approval and trace pairs, unrecorded residual effects fell from 40 with explicit fields to 17 with command semantics and 13 with decision-time metadata.",
      "patterns": [
        "approval",
        "confirmation"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Residual (unrecorded) effects in approval records",
          "baseline": "Explicit fields only: 40",
          "treatment": "With command semantics: 17; with decision-time metadata: 13",
          "direction": "decrease",
          "magnitude": "40 to 13",
          "n": "111 fixed approval-object and trace pairs"
        },
        {
          "metric": "Residual effects when approvals are bound to predicted effects",
          "baseline": "Unbound: 10",
          "treatment": "Bound to source-backed predictions: 3",
          "direction": "decrease",
          "magnitude": "10 to 3; predictions reached 0.926 macro recall and 0.941 macro precision",
          "n": "17 prespecified holdout workflows"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not a model comparison; approval objects and traces from coding-agent frontends.",
        "harness": "Three product frontends (not named in the abstract); a Claude Code PreToolUse integration is described",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.28586",
        "title": "Agent Approval Laundering: Transitive Effects Beyond the Approved Invocation",
        "publisher": "arXiv",
        "authors": [
          "Jinqian Zhang",
          "Haojun Xia",
          "Shujiang Wu",
          "Jingkun Yue",
          "Xia Zhang",
          "Zhangpei Cheng",
          "Bibo Tu"
        ],
        "published": "2026-09-23",
        "identifier": "arXiv:2609.28586"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "The benchmark was built by the authors."
      ],
      "implications": "Approving a command is not approving everything it sets off, such as install hooks or network calls. Show the expected effects when you ask for approval, and record them with the decision.",
      "related": [
        "ACT-01",
        "ACT-02",
        "HANDBACK-04"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0017",
      "title": "Injected skills lowered pass rates and raised token cost on average",
      "claim": "WebDev-Skills-Bench (preprint) found that injecting matched public skills reduced mean Pass@2 by 1.3% to 4.2% across four models and raised token cost by 72% to 394%, with gains in only 17% to 36% of skill-project pairs.",
      "patterns": [
        "skills",
        "cost",
        "context-budget"
      ],
      "finding_type": "negative-result",
      "effect": [
        {
          "metric": "Mean Pass@2 with the target skill injected",
          "baseline": "No skill",
          "treatment": "Target skill injected",
          "direction": "decrease",
          "magnitude": "-1.3% to -4.2%",
          "n": "31 public WebDev skills; 50 projects; 1,000 ordered tasks; four models"
        },
        {
          "metric": "Token cost",
          "baseline": "No skill",
          "treatment": "Target skill injected",
          "direction": "increase",
          "magnitude": "+72% to +394%",
          "n": "As above"
        },
        {
          "metric": "Skill-project pairs that gained",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "17% to 36%",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Four models. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.23067",
        "title": "Signal or Noise? A Benchmark Study of Agent Skills in Web Development",
        "publisher": "arXiv",
        "authors": [
          "Ziyue Yang",
          "Fan Ding"
        ],
        "published": "2026-08-24",
        "identifier": "arXiv:2608.23067"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Web development tasks only.",
        "Skill rankings transferred weakly across models."
      ],
      "implications": "Treat a skill as a hypothesis about one project and model, not a portable asset. Test it against a no-skill baseline and a length-matched control before you ship it; within helpful skills, anti-pattern rules beat example-heavy content.",
      "related": [],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0018",
      "title": "Skill rules that name a command or path change what agents do",
      "claim": "A study of 3,159 skills (preprint) found that adding a checkable rule raised the rate at which four coding agents took the required action by +0.23 on average, with the gain coming mainly from rules naming a command or path the old skill did not mention.",
      "patterns": [
        "skills",
        "description"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Rate at which the agent takes the required action after a rule is added",
          "baseline": "Skill before the revision",
          "treatment": "Skill with the added rule",
          "direction": "increase",
          "magnitude": "+0.23 on average (+0.16 to +0.36)",
          "n": "Four agents in a sandbox"
        },
        {
          "metric": "Single-answer compliance after a rule is added",
          "baseline": "Before",
          "treatment": "After",
          "direction": "increase",
          "magnitude": "+0.41 on average",
          "n": "16 open-weight models"
        },
        {
          "metric": "Final correctness",
          "baseline": "Before",
          "treatment": "After",
          "direction": "increase",
          "magnitude": "+0.10 on average (+0.06 to +0.14)",
          "n": "Three agents assessed by blind judges"
        },
        {
          "metric": "Share of the action gain kept when the skill body loads only on demand",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "About 51% for the four agents; about 38% for the three open models",
          "n": "As above"
        },
        {
          "metric": "Episode tokens when a skill body is loaded",
          "baseline": "Body not loaded",
          "treatment": "Body loaded",
          "direction": "increase",
          "magnitude": "+50% on average",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "21 models for single answers and four agents in a sandbox. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2610.04832",
        "title": "Agent Skill Evolution: How Revisions Affect Coding Agents",
        "publisher": "arXiv",
        "authors": [
          "Jiajie Wang",
          "Yutong Zhao",
          "Tianlin Li",
          "Huashan Chen",
          "Jinfu Chen",
          "Kebin Peng",
          "Sen He"
        ],
        "published": "2026-10-04",
        "identifier": "arXiv:2610.04832"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Limited to revisions that add or remove an automatically checkable rule."
      ],
      "implications": "In a skill, name the exact command or file path the agent should use. Because skills often load only on demand, put the most important rule where the agent sees it without loading the body.",
      "related": [
        "READ-10"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0019",
      "title": "Skill selection precision collapses as the skill pool grows",
      "claim": "A preprint reports that as the pool of available skills grew from 5 to 100, the precision with which agents actually used the right skill fell from 29.6% to 3.3%.",
      "patterns": [
        "skills",
        "selection",
        "discovery"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Actual-use precision of skills",
          "baseline": "Pool of 5 skills: 29.6%",
          "treatment": "Pool of 100 skills: 3.3%",
          "direction": "decrease",
          "magnitude": "29.6% to 3.3%",
          "n": "8,135 normalised trial records"
        },
        {
          "metric": "How skills help, by mode",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "Procedural anchoring 65.7% of skill cases; explicit knowledge injection 4.5%",
          "n": "238 labelled records"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Various benchmarks, harnesses and LLMs. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.14036",
        "title": "Demystifying Agent Skills: Why They Work-Until They Don't",
        "publisher": "arXiv",
        "authors": [
          "Zhiyuan Jiang",
          "Fangrui Huang",
          "Hanwen Xing",
          "Xander Wu",
          "Yipeng Gao",
          "Rui Cao",
          "Mengdi Wang",
          "Shilong Liu",
          "Yijiang Li"
        ],
        "published": "2026-08-14",
        "identifier": "arXiv:2608.14036"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "The abstract reports that downstream success stayed stable despite confusable distractors, so lower selection precision did not always mean failure."
      ],
      "implications": "Choosing a skill is a retrieval problem that gets harder with every skill installed. Keep the set small, and make each description say when the skill applies and when it does not.",
      "related": [
        "FIND-06",
        "READ-04"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0020",
      "title": "Praise and list order move tool selection",
      "claim": "A preregistered preprint with two small OpenAI models found that stacked praise in a tool description raised its pick rate by about 43 percentage points, and that with identical listings the first-listed tool was picked about 72 points more often.",
      "patterns": [
        "description",
        "selection"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Pick rate with stacked praise (four kinds combined)",
          "baseline": "Listing without praise",
          "treatment": "Listing with stacked praise",
          "direction": "increase",
          "magnitude": "About +43 percentage points, matching or beating a verifiable specification",
          "n": "Two OpenAI models; preregistered study"
        },
        {
          "metric": "Pick rate of the first-listed tool with identical listings",
          "baseline": "Listed second",
          "treatment": "Listed first",
          "direction": "increase",
          "magnitude": "About +72 percentage points",
          "n": "As above"
        },
        {
          "metric": "Picking the capable tool on numeric-limit tasks",
          "baseline": "Structured limit fields",
          "treatment": "Structured fields plus provider sales text",
          "direction": "decrease",
          "magnitude": "Sales text reduced or erased the gain from structured fields",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Two small OpenAI models. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2605.23916",
        "title": "Agent-Facing Information Design in LLM Tool Registries: A Preregistered Test of Rhetoric, Position and Structure",
        "publisher": "arXiv",
        "authors": [
          "Haochuan Kevin Wang",
          "Zechen Zhang"
        ],
        "published": "2026-04-12",
        "version": "v2, revised 2026-09-30 (adds the preregistered study)",
        "identifier": "arXiv:2605.23916"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Two small models.",
        "The authors call the results provisional until blind phrase ratings are complete; only stacked praise, not any single kind, replicated on held-out domains."
      ],
      "implications": "Tool descriptions are a ranking surface. Write limits as structured fields and leave out sales language; if you run a registry, randomise the order and hide promotional text from agents.",
      "related": [
        "READ-04",
        "FIND-06"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0021",
      "title": "Agents leave available tools unused: the adoption gap",
      "claim": "On OSWorld-MCP (preprint), a reasoning model given MCP tools called one on only 55 of 309 tasks, 23.9% of the tasks a tool could reach, and the same tools made a non-reasoning model 5.9 points worse.",
      "patterns": [
        "propensity",
        "selection"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Task success with MCP tools added to a GUI agent",
          "baseline": "GUI only",
          "treatment": "GUI plus MCP tools: reasoning model +4.0 points; non-reasoning model -5.9 points",
          "direction": "mixed",
          "magnitude": "+4.0 and -5.9 percentage points",
          "n": "309 tasks; 5 runs each"
        },
        {
          "metric": "Tasks on which the reasoning model called a tool",
          "baseline": null,
          "treatment": "55 of 309 tasks",
          "direction": "not-applicable",
          "magnitude": "23.9% of tool-reachable tasks",
          "n": "309 tasks"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "One reasoning and one non-reasoning model. The abstract does not name them.",
        "harness": "One identical GUI-MCP harness on OSWorld-MCP",
        "extensions": [
          "MCP tools from OSWorld-MCP"
        ],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2608.03327",
        "title": "Screenshots or Tools? Eliciting Tool Use and Managing Multimodal Context in Hybrid GUI-MCP Computer-Use Agents",
        "publisher": "arXiv",
        "authors": [
          "Siqi Fan",
          "Minghao Li",
          "Xiaoqian Ma",
          "Wenhui Tan",
          "Xiusheng Huang",
          "Juntong Wu",
          "Liujie Zhang",
          "Shuo Shang",
          "Weihang Chen"
        ],
        "published": "2026-08-04",
        "version": "v2, revised 2026-08-06",
        "identifier": "arXiv:2608.03327"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "One benchmark.",
        "Training that raised tool adoption did not raise held-out accuracy."
      ],
      "implications": "Shipping a tool does not mean agents will use it. Measure how often agents call it on the tasks it could serve, not only whether it works when called.",
      "related": [
        "FIND-06"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0022",
      "title": "Search-and-execute meta-tools cut catalogue tokens by 99% in production",
      "claim": "PayPal authors report (preprint) that exposing two meta-tools, search and execute, over 2,000+ MCP tools cut tool-token consumption in production from 140.2k tokens (70.1% of context) to 1.3k tokens (0.8%).",
      "patterns": [
        "context-budget",
        "dynamic-tools",
        "cost"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "MCP tool-token consumption per query",
          "baseline": "Full tool schemas: 140.2k tokens (70.1% of context)",
          "treatment": "tool_search and execute_tool meta-tools: 1.3k tokens (0.8% of context)",
          "direction": "decrease",
          "magnitude": "About 99% reduction",
          "n": "2,000+ tools across 200+ MCP servers, in production"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Described as model-agnostic; no models named in the abstract.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "The authors describe their own production system at PayPal; this is a deployment report by the people who built it.",
      "source": {
        "url": "https://arxiv.org/abs/2608.23992",
        "title": "Hybrid Semantic Tool Discovery for Enterprise MCP Gateway: Architecture and Implementation",
        "publisher": "arXiv",
        "authors": [
          "Olympia Saha",
          "Amy Wang",
          "Srinivasan Manoharan"
        ],
        "published": "2026-08-25",
        "identifier": "arXiv:2608.23992"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Token accounting only; the abstract reports no task-success or selection-accuracy comparison."
      ],
      "implications": "For a large tool catalogue, let the agent search for tools instead of loading every schema up front. Check selection accuracy separately, because token savings say nothing about it.",
      "related": [
        "FIND-06"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0023",
      "title": "Hiding tools is not enforcing permissions",
      "claim": "Across 2,160 attempts with four frontier models (preprint), a server with only in-body permission checks exposed forbidden tools in 152 of 720 trials and permission-aware visibility cut that to 0 of 720, yet models named a hidden tool in up to 94% of settings when it was inferable from the prompt.",
      "patterns": [
        "auth-scopes",
        "dynamic-tools",
        "discovery"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Trials in which forbidden tools were exposed",
          "baseline": "In-body checks only: 152 of 720 (21.1%)",
          "treatment": "Permission-aware visibility: 0 of 720",
          "direction": "decrease",
          "magnitude": "21.1% to 0%",
          "n": "2,160 attempts; four frontier models"
        },
        {
          "metric": "Settings in which models referenced a hidden tool by name",
          "baseline": null,
          "treatment": "Up to 94% when the tool was inferable from the prompt",
          "direction": "not-applicable",
          "magnitude": "Up to 94%",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Four frontier LLMs. The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.22573",
        "title": "Zero-Trust Authorization and Discovery for Enterprise MCP",
        "publisher": "arXiv",
        "authors": [
          "Huan Li",
          "Yuwei Wang",
          "Srinivasan Manoharan"
        ],
        "published": "2026-09-18",
        "identifier": "arXiv:2609.22573"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Visibility-only filtering remained bypassable by scripted clients."
      ],
      "implications": "Hiding the tools a caller may not use makes the catalogue easier to choose from, but it is not a security boundary. Enforce permissions when a tool is called.",
      "related": [
        "ACT-05",
        "FIND-06"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0024",
      "title": "Harnesses alter shell calls and the wrong action runs silently",
      "claim": "A preprint reports that in 47,828 production shell calls, Claude Code's Bash tool changed 12.0% of calls carrying code, escape sequences or long text, and that for 80.7% of calls whose backslashes were changed the wrong action ran with no reported error.",
      "patterns": [
        "false-success",
        "evaluation",
        "measurement"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Calls changed in transit by the harness",
          "baseline": null,
          "treatment": "Claude Code's Bash tool: 12.0% of calls carrying code, escape sequences or long text",
          "direction": "not-applicable",
          "magnitude": "12.0%; all 10 measured harnesses change some call",
          "n": "47,828 shell calls in production sessions; 10 harnesses"
        },
        {
          "metric": "Calls with changed backslashes where the wrong action ran with no error",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "80.7%",
          "n": "As above"
        },
        {
          "metric": "Failures blamed on the model by trajectory-based judgement",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "95.1% blamed on the LLM, although the path caused more than half",
          "n": "Production failures"
        },
        {
          "metric": "Token cost per passed task caused by the path",
          "baseline": "Unaltered path",
          "treatment": "Altering path",
          "direction": "increase",
          "magnitude": "2.4 times (up to 12.3 times)",
          "n": "IEC-Bench"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not a model comparison; the abstract does not name the models in the production sessions.",
        "harness": "Claude Code (production sessions); 10 harnesses measured, four in IEC-Bench",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "The authors' repair, IntAct, is deployed in a commercial product.",
      "source": {
        "url": "https://arxiv.org/abs/2610.04375",
        "title": "Do Tool Calls Execute as Intended? Measuring and Repairing Intent-Execution Correspondence in LLM Agents",
        "publisher": "arXiv",
        "authors": [
          "Boyang Yang",
          "Zhenhao Li",
          "Ziyao Yang",
          "Kanghui Jia",
          "Xin Yin",
          "Mingmou Liu",
          "Haoye Tian"
        ],
        "published": "2026-10-03",
        "identifier": "arXiv:2610.04375"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Harness versions are not stated in the abstract."
      ],
      "implications": "When a tool call fails, log what the tool actually received, not only what the model sent. A command that arrives altered can run the wrong action with no error.",
      "related": [
        "RECOVER-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0025",
      "title": "Run-to-run variance dominates, and a verification tool beats a verification prompt",
      "claim": "A preprint reports that about 54% of outcome variance in its coding-agent benchmark came from repeating the same configuration, and that a dedicated verification tool changed verification behaviour substantially where a prompt asking the agent to verify had little effect.",
      "patterns": [
        "evaluation",
        "measurement"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Share of outcome variance from repeating an identical configuration",
          "baseline": null,
          "treatment": null,
          "direction": "not-applicable",
          "magnitude": "About 54%",
          "n": "Four scientific tasks; more than 18,000 released trajectories"
        },
        {
          "metric": "Verification behaviour",
          "baseline": "Prompt asking the agent to verify: little effect",
          "treatment": "Dedicated verification tool: substantial change",
          "direction": "increase",
          "magnitude": "Qualitative in the abstract",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Several backbone models (one of the five configuration factors). The abstract does not name them.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2610.01618",
        "title": "Agents Are Systems, Not Models: Rethinking Agentic Evaluation",
        "publisher": "arXiv",
        "authors": [
          "Luis Wiedmann",
          "Leander Girrbach",
          "Cordelia Schmid",
          "Zeynep Akata"
        ],
        "published": "2026-10-01",
        "identifier": "arXiv:2610.01618"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Four scientific tasks in which a coding agent must operate a published specialist model."
      ],
      "implications": "Run each test several times before you conclude that a change helped. If you want an agent to check its work, give it a tool for that rather than an instruction.",
      "related": [],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0026",
      "title": "Prompt injection split across tool channels evades defences",
      "claim": "Across 12 frontier models and over 15,000 trials (preprint), models that resisted single-channel prompt injection exfiltrated data at up to 100% when the payload was split across two channels, such as a tool description and a tool result, and seven third-party MCP security tools failed to detect it.",
      "patterns": [
        "prompt-injection",
        "description"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Credential exfiltration compliance",
          "baseline": "Single-channel injection: 0% for the named models",
          "treatment": "Two-channel fragmentation: up to 100%",
          "direction": "increase",
          "magnitude": "0% to up to 100%",
          "n": "12 frontier models; three production clients; six payloads; over 15,000 trials"
        },
        {
          "metric": "Detection of fragmented payloads by third-party MCP security tools",
          "baseline": null,
          "treatment": "All seven tools failed to detect them",
          "direction": "not-applicable",
          "magnitude": "0 of 7 tools detected them",
          "n": "Seven tools; three prompt-based defences (model-specific)"
        }
      ],
      "agent_profile": {
        "models": [
          "GPT-4o",
          "Llama 70B",
          "Composer 2",
          "Haiku 4.5"
        ],
        "models_note": "12 frontier models; the abstract names these four as examples of models that resisted single-channel injection.",
        "harness": "Three production clients (not named in the abstract); a VS Code MCP sampling override is described",
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2609.18217",
        "title": "Measuring and Exploiting Implicit Trust in LLM Tool-Calling Pipelines",
        "publisher": "arXiv",
        "authors": [
          "Murali Ediga",
          "Sudipta Chattopadhyay"
        ],
        "published": "2026-09-16",
        "identifier": "arXiv:2609.18217"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "Attack research with payloads designed by the authors."
      ],
      "implications": "Treat every tool description and tool result as untrusted input. Defences that limit what an agent can do, such as allowed destinations and narrow capabilities, hold better than defences that try to spot attacks.",
      "related": [
        "ACT-05"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0027",
      "title": "A capable agent skipped the index and guessed the page",
      "claim": "A preregistered ablation on a 709-page Markdown wiki (preprint) found that a capable tool-using agent never loaded the compact catalogue index, inferring page paths from the question instead, while retrieval-based access kept answer quality non-inferior and cut cost by about a third to over half.",
      "patterns": [
        "docs-for-agents",
        "discovery",
        "context-budget"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Whether the agent loaded the catalogue index",
          "baseline": null,
          "treatment": "Never loaded in the pilot; the agent inferred page paths and read pages directly",
          "direction": "not-applicable",
          "magnitude": "Never",
          "n": "Pilot"
        },
        {
          "metric": "Cost of answering, retrieval arm versus index baseline",
          "baseline": "Index baseline",
          "treatment": "Retrieval arm",
          "direction": "decrease",
          "magnitude": "About a third (self-routing agent) to well over half (catalogue preload); all confidence intervals exclude zero",
          "n": "Four corpus arms crossed with three access conditions; blind grading against verified gold answers"
        },
        {
          "metric": "Answer quality",
          "baseline": "Index baseline",
          "treatment": "Retrieval arm",
          "direction": "no-change",
          "magnitude": "Non-inferior within the preregistered margin",
          "n": "As above"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "A protocol-constrained agent, a free self-routing agent and a catalogue-preload regime. The abstract does not name the models.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "preprint",
      "conflicts_of_interest": "None declared in the abstract. The full text was not checked for a competing-interest statement.",
      "source": {
        "url": "https://arxiv.org/abs/2607.04576",
        "title": "Progressive Disclosure for LLM-Maintained Wiki Knowledge Bases: a Preregistered Ablation",
        "publisher": "arXiv",
        "authors": [
          "Theodore O. Cochran"
        ],
        "published": "2026-07-06",
        "identifier": "arXiv:2607.04576"
      },
      "retrieved": "2026-10-08",
      "verification": "abstract-only",
      "verification_note": "Every number in this record was checked against the live arXiv abstract page on 2026-10-08. The full text was not re-checked.",
      "limitations": [
        "Preprint, not peer reviewed.",
        "One wiki maintained by an LLM.",
        "Single author.",
        "Does not test public web standards such as llms.txt."
      ],
      "implications": "Do not assume an agent will read an index or catalogue page before acting; a capable agent may guess the path. Stable, guessable addresses and targeted retrieval did more here than the index.",
      "related": [
        "FIND-03",
        "FIND-04"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0028",
      "title": "A JSON input mode for a CLI raised agent cost 4x to 11x",
      "claim": "Microsoft reports that on a synthetic CLI, run five times per model in GitHub Copilot Chat, every model was correct 5/5 with individual arguments, while a --json payload mode cost 4x to 11x more per task and dropped Claude Haiku 4.5 to 2/5 and MAI-Code-1-Flash to 3/5.",
      "patterns": [
        "cost",
        "measurement"
      ],
      "finding_type": "negative-result",
      "effect": [
        {
          "metric": "Correct deployments out of five runs",
          "baseline": "Individual arguments: 5/5 for every model",
          "treatment": "--json payload: 5/5 for Claude Sonnet 5, Claude Sonnet 4.6 and GPT-5.3-Codex; 3/5 for MAI-Code-1-Flash; 2/5 for Claude Haiku 4.5",
          "direction": "decrease",
          "magnitude": "Down to 2/5 for the smallest model",
          "n": "Five runs per model per input mode; five models"
        },
        {
          "metric": "Cost per task (GitHub Models pricing)",
          "baseline": "Arguments: GPT-5.3-Codex $0.05; MAI-Code-1-Flash $0.01; Claude Sonnet 4.6 $0.05; Claude Haiku 4.5 $0.03; Claude Sonnet 5 $0.08",
          "treatment": "JSON: $0.54 (11x); $0.08 (10x); $0.47 (9x); $0.23 (8x); $0.32 (4x)",
          "direction": "increase",
          "magnitude": "4x to 11x",
          "n": "As above"
        },
        {
          "metric": "Arguments-to-JSON cost gap by shell, Claude Sonnet 4.6",
          "baseline": "Bash on macOS: 1.5x",
          "treatment": "PowerShell on Windows: 9x",
          "direction": "increase",
          "magnitude": "1.5x on Bash, 9x on PowerShell",
          "n": "One model, two shells"
        }
      ],
      "agent_profile": {
        "models": [
          "Claude Haiku 4.5",
          "Claude Sonnet 4.6",
          "Claude Sonnet 5",
          "GPT-5.3-Codex",
          "MAI-Code-1-Flash"
        ],
        "harness": "GitHub Copilot Chat",
        "extensions": [],
        "os": "Windows (PowerShell) for the main experiment; macOS (Bash) for the Sonnet 4.6 shell comparison",
        "run_dates": null
      },
      "evidence_class": "vendor-measurement",
      "conflicts_of_interest": "Microsoft authors evaluating with Microsoft tools (GitHub Copilot Chat, VS Code) and, where noted, Microsoft products. Microsoft reports the results itself. The CLI (podctl) is synthetic and was built for the test; MAI-Code-1-Flash is a Microsoft model.",
      "source": {
        "url": "https://developer.microsoft.com/blog/dont-rewrite-your-cli-for-agents/",
        "title": "Don't rewrite your CLI for agents",
        "publisher": "Microsoft for Developers",
        "authors": [
          "Waldek Mastykarz"
        ],
        "published": "2026-07-07"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it.",
      "limitations": [
        "Vendor-run evaluation, not peer reviewed; no raw data or harness was found published with the post.",
        "Five runs per condition; Microsoft scopes each result to the agent profile it measured.",
        "One synthetic CLI and one deployment scenario with 30+ values."
      ],
      "implications": "Keep named arguments when you make a CLI agent-friendly; a JSON payload adds failure modes such as shell quoting. If you add --json, keep the arguments too and measure both on the shells your users run.",
      "related": [
        "ACT-07",
        "READ-05"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0029",
      "title": "A warning that names the failing plan redirected agents; a tip did not",
      "claim": "Microsoft reports that a documentation tip pointing to the right tool got 1 of 5 agent runs to use it, while a warning naming the agent's failing approach ('Manually updating package.json alone will result in build failures') got 5 of 5.",
      "patterns": [
        "docs-for-agents",
        "propensity"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Runs in which the agent used the recommended CLI for an SPFx upgrade",
          "baseline": "More direct tip naming the tool: 1 of 5",
          "treatment": "Warning naming the failing approach: 5 of 5",
          "direction": "increase",
          "magnitude": "1/5 to 5/5",
          "n": "Five runs per condition"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "The post does not state the profile. Microsoft's SPFx case study of the same documentation work used GitHub Copilot Chat in VS Code on Windows with Claude Sonnet 4.6.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "vendor-measurement",
      "conflicts_of_interest": "Microsoft authors evaluating with Microsoft tools (GitHub Copilot Chat, VS Code) and, where noted, Microsoft products. Microsoft reports the results itself. The documentation measured is Microsoft's SharePoint Framework documentation.",
      "source": {
        "url": "https://developer.microsoft.com/blog/your-agent-already-has-a-plan/",
        "title": "Your agent already has a plan",
        "publisher": "Microsoft for Developers",
        "authors": [
          "Garry Trinder"
        ],
        "published": "2026-06-26"
      },
      "corroborating_sources": [
        {
          "url": "https://devblogs.microsoft.com/microsoft365dev/behind-spfx-dev-skills-testing-what-agents-know-and-fixing-what-they-miss/",
          "title": "Behind SPFx Dev Skills: testing what agents know and fixing what they miss",
          "publisher": "Microsoft 365 Developer Blog",
          "authors": [],
          "published": "2026-09-02"
        },
        {
          "url": "https://github.com/SharePoint/sp-dev-docs/pull/10921",
          "title": "SharePoint/sp-dev-docs#10921: Optimize SPFx release pages for agentic upgrade workflows",
          "publisher": "GitHub",
          "authors": [],
          "published": "2026-07-14"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The page displays 25 June 2026; its article:published_time metadata is 2026-06-26T00:00:05Z. Pull request #10921, which shipped the warning across release pages, was confirmed merged on 2026-07-14.",
      "limitations": [
        "Vendor-run evaluation, not peer reviewed; no raw data or harness was found published with the post.",
        "Five runs per condition; Microsoft scopes each result to the agent profile it measured.",
        "One upgrade scenario; the authors advise using this tactic sparingly."
      ],
      "implications": "If agents keep taking a wrong path despite your docs, name that path on the page and say plainly that it fails, then give the right alternative. Keep it for cases where the default really fails.",
      "related": [],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0030",
      "title": "Adding the context7 MCP server gave no lift; its tools went unused",
      "claim": "Microsoft reports that adding the context7 MCP server to an anti-hallucination skill gave no meaningful lift on an SPFx upgrade: its tools did not load in 3 of 5 runs and were not called in the other 2, while telling the agent to use CLI for Microsoft 365 raised configuration correctness from 30/80 to 75/80.",
      "patterns": [
        "propensity",
        "docs-for-agents",
        "skills"
      ],
      "finding_type": "null-result",
      "effect": [
        {
          "metric": "Configuration correctness (points out of 80)",
          "baseline": "Bare baseline: 30/80",
          "treatment": "Anti-hallucination skill: 37/80; skill plus context7: 41/80; told to use CLI for Microsoft 365: 75/80",
          "direction": "increase",
          "magnitude": "context7 added no meaningful lift over the skill; the CLI instruction reached 75/80",
          "n": "Five runs per configuration"
        },
        {
          "metric": "context7 tool use",
          "baseline": null,
          "treatment": "Tools not loaded into context in 3 of 5 runs; available but never invoked in 2 of 5",
          "direction": "not-applicable",
          "magnitude": "0 of 5 runs used it",
          "n": "Five runs"
        },
        {
          "metric": "Dependency currency (points out of 50)",
          "baseline": "Bare baseline: 34/50",
          "treatment": "Skill: 38/50; skill plus context7: 39/50; CLI instruction: 49/50",
          "direction": "increase",
          "magnitude": "34/50 to 49/50 with the CLI instruction",
          "n": "Five runs per configuration"
        }
      ],
      "agent_profile": {
        "models": [
          "Claude Sonnet 4.6"
        ],
        "harness": "GitHub Copilot Chat in Visual Studio Code",
        "extensions": [
          "SPFx anti-hallucination skill",
          "context7 MCP server"
        ],
        "os": "Windows",
        "run_dates": null
      },
      "evidence_class": "vendor-measurement",
      "conflicts_of_interest": "Microsoft authors evaluating with Microsoft tools (GitHub Copilot Chat, VS Code) and, where noted, Microsoft products. Microsoft reports the results itself. The task is an upgrade of Microsoft's SharePoint Framework; context7 is a third-party product.",
      "source": {
        "url": "https://devblogs.microsoft.com/microsoft365dev/behind-spfx-dev-skills-testing-what-agents-know-and-fixing-what-they-miss/",
        "title": "Behind SPFx Dev Skills: testing what agents know and fixing what they miss",
        "publisher": "Microsoft 365 Developer Blog",
        "authors": [],
        "published": "2026-09-02"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The post page names no individual author; published 2026-09-02 per its metadata.",
      "limitations": [
        "Vendor-run evaluation, not peer reviewed; no raw data or harness was found published with the post.",
        "Five runs per condition; Microsoft scopes each result to the agent profile it measured.",
        "One upgrade task (SPFx 1.21.1 to 1.22.2)."
      ],
      "implications": "Installing another documentation source does not mean the agent will use it. Check whether a tool loaded and was called before you credit it with a change.",
      "related": [
        "FIND-06"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0031",
      "title": "A model with cheaper tokens cost 3.7x more per run",
      "claim": "Microsoft reports that on SharePoint Framework upgrade tasks Claude Sonnet 5 cost $2.01 per run against $0.55 for Claude Sonnet 4.6, 3.7x more despite 33% lower per-token prices, while on architecture tasks it was 12% cheaper.",
      "patterns": [
        "cost",
        "measurement",
        "evaluation"
      ],
      "finding_type": "measurement",
      "effect": [
        {
          "metric": "Average cost per run, code upgrade tasks",
          "baseline": "Claude Sonnet 4.6: $0.55",
          "treatment": "Claude Sonnet 5: $2.01",
          "direction": "increase",
          "magnitude": "3.7x",
          "n": "3 SPFx scenarios; 5 runs per model per scenario"
        },
        {
          "metric": "Average cost per run, architecture tasks",
          "baseline": "Claude Sonnet 4.6: $0.54",
          "treatment": "Claude Sonnet 5: $0.47",
          "direction": "decrease",
          "magnitude": "12% cheaper",
          "n": "12 architecture scenarios; 5 runs per model per scenario"
        },
        {
          "metric": "Median token use, Sonnet 5 versus Sonnet 4.6",
          "baseline": "Claude Sonnet 4.6",
          "treatment": "Claude Sonnet 5",
          "direction": "increase",
          "magnitude": "12x on architecture tasks; 10x on code upgrades",
          "n": "150 runs in total"
        },
        {
          "metric": "Quality and completion",
          "baseline": "Claude Sonnet 4.6: idiomatic quality 90% (architecture); upgrade completion 60%",
          "treatment": "Claude Sonnet 5: idiomatic quality 78%; upgrade completion 100%",
          "direction": "mixed",
          "magnitude": "Quality down 12 points on architecture; completion up 40 points on upgrades",
          "n": "150 runs in total"
        }
      ],
      "agent_profile": {
        "models": [
          "Claude Sonnet 4.6",
          "Claude Sonnet 5"
        ],
        "harness": "GitHub Copilot Chat in VS Code",
        "extensions": [],
        "os": "Windows",
        "run_dates": null
      },
      "evidence_class": "vendor-measurement",
      "conflicts_of_interest": "Microsoft authors evaluating with Microsoft tools (GitHub Copilot Chat, VS Code) and, where noted, Microsoft products. Microsoft reports the results itself. Costs are priced at GitHub Copilot's published rates; quality is scored by an LLM judge on Microsoft's own evaluation platform.",
      "source": {
        "url": "https://developer.microsoft.com/blog/not-all-model-upgrades-are-upgrades/",
        "title": "Not all model upgrades are upgrades",
        "publisher": "Microsoft for Developers",
        "authors": [
          "Waldek Mastykarz"
        ],
        "published": "2026-07-06"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it.",
      "limitations": [
        "Vendor-run evaluation, not peer reviewed; no raw data or harness was found published with the post.",
        "Five runs per condition; Microsoft scopes each result to the agent profile it measured.",
        "Quality scored by an LLM judge on Microsoft's own, unpublished evaluation platform."
      ],
      "implications": "Compare models on cost per completed task, not on price per token. Run your own tasks, because the answer flipped between task types.",
      "related": [],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0032",
      "title": "Vendor claim: agents never invoked a docs skill in 56% of eval cases",
      "claim": "Vercel states that in its Next.js 16 evals the docs skill was never invoked in 56% of cases, so the skill matched the 53% pass rate of no docs, while an 8KB docs index in AGENTS.md reached 100%.",
      "patterns": [
        "propensity",
        "skills",
        "agents-md",
        "docs-for-agents"
      ],
      "finding_type": "vendor-claim",
      "effect": [
        {
          "metric": "Eval cases in which the skill was never invoked",
          "baseline": null,
          "treatment": "56%",
          "direction": "not-applicable",
          "magnitude": "56%",
          "n": "Not stated"
        },
        {
          "metric": "Pass rate on the hardened Next.js 16 eval suite",
          "baseline": "No docs: 53%",
          "treatment": "Skill, default: 53%; skill with explicit AGENTS.md instructions: 79%; docs index in AGENTS.md: 100%",
          "direction": "increase",
          "magnitude": "+0, +26 and +47 percentage points",
          "n": "Not stated"
        },
        {
          "metric": "Skill trigger rate with explicit instructions",
          "baseline": "Default: skill not invoked in 56% of cases",
          "treatment": "Explicit instruction in AGENTS.md: 95%+",
          "direction": "increase",
          "magnitude": "To 95%+",
          "n": "Not stated"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "The post does not name the models or the agent used.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "vendor-claim",
      "conflicts_of_interest": "Vercel maintains Next.js and the @next/codemod tool that installs the AGENTS.md docs index the post recommends.",
      "source": {
        "url": "https://vercel.com/blog/agents-md-outperforms-skills-in-our-agent-evals",
        "title": "AGENTS.md outperforms skills in our agent evals",
        "publisher": "Vercel",
        "authors": [
          "Jude Gao"
        ],
        "published": "2026-01-27"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it.",
      "limitations": [
        "A vendor claim: no run counts, model names or raw data are published.",
        "The vendor's own eval suite, focused on Next.js 16 APIs not in model training data."
      ],
      "implications": "Do not assume an agent will choose to load a skill. For knowledge every task needs, passive context such as an AGENTS.md entry was more reliable in this vendor's tests; check it on your own tasks.",
      "related": [
        "READ-10"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0033",
      "title": "Vendor claim: agents were 48% of Wrangler CLI use",
      "claim": "Cloudflare states that agents accounted for 48% of Wrangler CLI use in the week before 28 September 2026, up from a quarter in March 2026 and single-digit percentages a year earlier.",
      "patterns": [
        "measurement",
        "propensity"
      ],
      "finding_type": "vendor-claim",
      "effect": [
        {
          "metric": "Agent share of Wrangler use",
          "baseline": "A year earlier: single-digit percentages; March 2026: a quarter",
          "treatment": "Week before launch: 48%",
          "direction": "increase",
          "magnitude": "To 48%",
          "n": "Not stated"
        },
        {
          "metric": "Command breadth, agents versus humans",
          "baseline": null,
          "treatment": "Agents use almost twice as many distinct commands per day and are almost four times as likely to use six or more commands",
          "direction": "not-applicable",
          "magnitude": "About 2x and 4x",
          "n": "Not stated"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not stated. The measure covers agents the CLI detects; Cloudflare does not publish the method.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "vendor-claim",
      "conflicts_of_interest": "Cloudflare reports usage of its own product in the launch post for a new agent-focused CLI.",
      "source": {
        "url": "https://blog.cloudflare.com/cloudflare-cf-cli-launch/",
        "title": "Introducing cf: the agentic CLI for the entire Cloudflare API",
        "publisher": "Cloudflare",
        "authors": [
          "Matt \"TK\" Taylor",
          "Samuel Macleod"
        ],
        "published": "2026-09-28"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it.",
      "limitations": [
        "A vendor claim: the unit (requests, sessions or users) and the detection method are not published."
      ],
      "implications": "Agent use can be a large share of a developer tool's use, so it is worth measuring as its own segment. Ask how a vendor defines and detects \"agent use\" before comparing figures.",
      "related": [
        "HANDBACK-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0034",
      "title": "Vendor claim: users approved about 93% of permission prompts",
      "claim": "Anthropic states that its telemetry showed users approved roughly 93% of Claude Code permission prompts, and that an operating-system sandbox reduced permission prompts by 84%.",
      "patterns": [
        "approval",
        "confirmation"
      ],
      "finding_type": "vendor-claim",
      "effect": [
        {
          "metric": "Permission prompts approved by users",
          "baseline": null,
          "treatment": "Roughly 93%",
          "direction": "not-applicable",
          "magnitude": "About 93%",
          "n": "Not stated (telemetry)"
        },
        {
          "metric": "Permission prompts with an OS-level sandbox",
          "baseline": "Without the sandbox",
          "treatment": "With the sandbox",
          "direction": "decrease",
          "magnitude": "84% fewer",
          "n": "Not stated"
        },
        {
          "metric": "Auto mode classifier",
          "baseline": null,
          "treatment": "Catches roughly 83% of overeager behaviours; about 0.4% of benign commands blocked; about 17% of overeager actions get through",
          "direction": "not-applicable",
          "magnitude": "About 83% caught; about 0.4% false blocks",
          "n": "Not stated"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not stated for these figures.",
        "harness": "Claude Code",
        "extensions": [],
        "os": "macOS (Seatbelt) and Linux (bubblewrap) for the sandbox",
        "run_dates": null
      },
      "evidence_class": "vendor-claim",
      "conflicts_of_interest": "Anthropic reports telemetry from its own product.",
      "source": {
        "url": "https://www.anthropic.com/engineering/how-we-contain-claude",
        "title": "How we contain Claude across products",
        "publisher": "Anthropic",
        "authors": [],
        "published": "2026-05-25"
      },
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it.",
      "limitations": [
        "A vendor claim from telemetry with no published method; not independently replicated."
      ],
      "implications": "When nearly every approval request is accepted, people stop reading them. Ask less often, for example by sandboxing low-risk actions, so the requests that remain get attention.",
      "related": [
        "HANDBACK-04",
        "ACT-02"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0035",
      "title": "Vendor claim: 70% of new Supabase databases are created by agents or AI tools",
      "claim": "Supabase states that it adds more than 4 million databases a month and that 70% of new databases are created by agents or AI-driven tools.",
      "patterns": [
        "measurement"
      ],
      "finding_type": "vendor-claim",
      "effect": [
        {
          "metric": "Share of new databases created by agents or AI-driven tools",
          "baseline": null,
          "treatment": "70%",
          "direction": "not-applicable",
          "magnitude": "70%",
          "n": "More than 4 million databases and 1 million users added per month"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not applicable: a platform usage figure.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "vendor-claim",
      "conflicts_of_interest": "Stated by Supabase in a press release announcing new funding and an acquisition (Turso).",
      "source": {
        "url": "https://www.prnewswire.com/news-releases/supabase-announces-150m-in-new-funding-and-turso-acquisition-302896752.html",
        "title": "Supabase Announces $150M in New Funding and Turso Acquisition",
        "publisher": "PR Newswire (Supabase)",
        "authors": [],
        "published": "2026-10-02"
      },
      "corroborating_sources": [
        {
          "url": "https://supabase.com/blog/supabase-is-acquiring-turso",
          "title": "Supabase is acquiring Turso",
          "publisher": "Supabase",
          "authors": [
            "Paul Copplestone"
          ],
          "published": "2026-10-02"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The 70% figure appears in the press release; the Supabase blog post says agents are spinning up millions of databases and that Supabase launches over one million databases per week.",
      "limitations": [
        "A vendor claim in a funding announcement; \"created by agents or AI-driven tools\" is not defined."
      ],
      "implications": "On some developer platforms, agents rather than people now create most new resources. Make resource creation work without a person at the keyboard, and make ownership clear afterwards.",
      "related": [
        "HANDBACK-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0036",
      "title": "A CLI exits 0 when a destructive command is refused",
      "claim": "Cloudflare's cf CLI documents that in a non-interactive session a destructive command without --force prints 'Aborted.' and exits with status 0, and a public issue reproduces this for workflows delete.",
      "patterns": [
        "exit-codes",
        "non-interactive",
        "false-success",
        "confirmation"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Exit status of a destructive command refused in non-interactive mode",
          "baseline": null,
          "treatment": "'Aborted.' on standard error, no deletion, exit status 0",
          "direction": "not-applicable",
          "magnitude": "Exit 0 on a refused operation",
          "n": "One public reproduction (cf 1.0.0-beta.5); documented as current behaviour"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: CLI behaviour in a non-interactive session (CI=true).",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "independent-measurement",
      "conflicts_of_interest": "None apparent. The issue was filed by a user, not by Cloudflare.",
      "source": {
        "url": "https://github.com/cloudflare/cf/issues/94",
        "title": "cloudflare/cf#94: workflows delete exits 0 when noninteractive confirmation aborts the operation",
        "publisher": "GitHub",
        "authors": [
          "matthewbjones"
        ],
        "published": "2026-09-29"
      },
      "corroborating_sources": [
        {
          "url": "https://developers.cloudflare.com/cf/agents/",
          "title": "Use cf with coding agents",
          "publisher": "Cloudflare Docs",
          "authors": [],
          "published": "2026-09-29"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The documentation page (last updated 29 September 2026) states the behaviour and warns that \"A successful exit does not mean the resource was deleted.\" The issue was open on the retrieval date.",
      "limitations": [
        "Beta software (1.0.0-beta); the behaviour may change before general availability.",
        "The documentation states the behaviour plainly and warns about it under \"Check for aborted deletes\" (\"A successful exit does not mean the resource was deleted\"), so it is documented and known, not hidden; the public issue asks for a non-zero exit instead."
      ],
      "implications": "A refused or cancelled operation should exit non-zero, because the exit status is the main success signal an agent or script checks. If it must exit 0, the agent has to parse standard error, which is fragile.",
      "related": [
        "RECOVER-01",
        "ACT-01"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0037",
      "title": "Destructive and bulk commands that report success but do nothing",
      "claim": "Users of Cloudflare's cf beta report commands that exit 0 without doing the work: R2 deletes of keys containing '/' (a cleanup script reported 2,220 objects deleted that were all still present) and a secrets bulk update that deployed an empty change to 100%.",
      "patterns": [
        "exit-codes",
        "false-success"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Objects actually deleted when a script reported success",
          "baseline": "Reported: 2,220 objects deleted, zero errors",
          "treatment": "Re-listing: all 2,220 objects still in the bucket",
          "direction": "not-applicable",
          "magnitude": "0 of 2,220 deleted; exit status 0",
          "n": "One user report with root cause (slashes percent-encoded in the key path)"
        },
        {
          "metric": "Effect of a secrets bulk update with an unparsable or unwrapped body",
          "baseline": null,
          "treatment": "A new Worker version and a 100% deployment with zero binding changes and no error",
          "direction": "not-applicable",
          "magnitude": "Silent no-op deployed to 100%",
          "n": "One user report"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: CLI behaviour reported by users.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "independent-measurement",
      "conflicts_of_interest": "None apparent. Both issues were filed by users, not by Cloudflare.",
      "source": {
        "url": "https://github.com/cloudflare/cf/issues/209",
        "title": "cloudflare/cf#209: r2 objects delete and bulk-delete exit 0 without deleting keys that contain '/'",
        "publisher": "GitHub",
        "authors": [
          "eduMNG"
        ],
        "published": "2026-10-05"
      },
      "corroborating_sources": [
        {
          "url": "https://github.com/cloudflare/cf/issues/201",
          "title": "cloudflare/cf#201: workers secrets bulk: --file and unwrapped bodies silently create an empty version + 100% deployment with no error",
          "publisher": "GitHub",
          "authors": [
            "cameronm123"
          ],
          "published": "2026-10-04"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. Both issues were open on the retrieval date.",
      "limitations": [
        "Beta software; individual user reports, not a survey of the CLI."
      ],
      "implications": "Before reporting success, check that a write took effect: inspect the response, and read back when the API treats a missing target as success. Exit 0 should mean the work was done.",
      "related": [
        "RECOVER-03",
        "READ-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0038",
      "title": "Dry-run output printed secrets until a fix redacted them",
      "claim": "An issue on Cloudflare's cf beta reported that --dry-run printed secret values in plain text, and Cloudflare merged a fix redacting sensitive values in dry-run output on 5 October 2026.",
      "patterns": [
        "dry-run",
        "secrets"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Secret values in --dry-run output",
          "baseline": "cf 1.0.0-beta.5: printed in plain text",
          "treatment": "After pull request #167: redacted",
          "direction": "decrease",
          "magnitude": "Plain text to redacted",
          "n": "One reproduction"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: CLI behaviour.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "independent-measurement",
      "conflicts_of_interest": "None apparent. The issue was filed by a user; the fix was made by a Cloudflare maintainer.",
      "source": {
        "url": "https://github.com/cloudflare/cf/issues/103",
        "title": "cloudflare/cf#103: `--dry-run` prints secret values in plain text",
        "publisher": "GitHub",
        "authors": [
          "joeblew999"
        ],
        "published": "2026-09-30"
      },
      "corroborating_sources": [
        {
          "url": "https://github.com/cloudflare/cf/pull/167",
          "title": "cloudflare/cf#167: fix: redact sensitive values in dry-run output",
          "publisher": "GitHub",
          "authors": [
            "penalosa"
          ],
          "published": "2026-10-05"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The issue was closed and pull request #167 was merged on 2026-10-05.",
      "limitations": [
        "Fixed; recorded as an example of a failure mode, not current behaviour."
      ],
      "implications": "Dry-run output ends up in logs, pull requests and agent transcripts. Redact secrets in it by default.",
      "related": [
        "ACT-01"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0039",
      "title": "Launch post and documentation disagree on agent output format",
      "claim": "Cloudflare's cf launch post says JSON output is 'condensed for agents', but the cf documentation says JSON output is indented whether or not output is a terminal, and a public issue reports byte-identical output with an agent detected.",
      "patterns": [
        "drift",
        "context-budget"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Output size with and without a detected agent (CLAUDECODE=1)",
          "baseline": "No agent detected: 4,293 bytes",
          "treatment": "Agent detected: 4,293 bytes, byte-identical and still pretty-printed",
          "direction": "no-change",
          "magnitude": "No difference",
          "n": "One reproduction (cf 1.0.0-beta.5, macOS)"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: CLI output with the agent-detection environment variable set.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "independent-measurement",
      "conflicts_of_interest": "None apparent. The issue was filed by a user, not by Cloudflare.",
      "source": {
        "url": "https://github.com/cloudflare/cf/issues/105",
        "title": "cloudflare/cf#105: Output for scripts and agents: empty stdout with exit 0, `--version` not machine-readable, agent detection changes nothing",
        "publisher": "GitHub",
        "authors": [
          "joeblew999"
        ],
        "published": "2026-09-30"
      },
      "corroborating_sources": [
        {
          "url": "https://blog.cloudflare.com/cloudflare-cf-cli-launch/",
          "title": "Introducing cf: the agentic CLI for the entire Cloudflare API",
          "publisher": "Cloudflare",
          "authors": [
            "Matt \"TK\" Taylor",
            "Samuel Macleod"
          ],
          "published": "2026-09-28"
        },
        {
          "url": "https://developers.cloudflare.com/cf/agents/",
          "title": "Use cf with coding agents",
          "publisher": "Cloudflare Docs",
          "authors": [],
          "published": "2026-09-29"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. The launch post sentence (\"pretty printed for humans and condensed for agents\") and the documentation sentence (\"JSON output is indented, whether or not standard output is a terminal\") were both read on the retrieval date. The issue was open.",
      "limitations": [
        "Beta software; the claim may yet ship or be withdrawn."
      ],
      "implications": "Check that what launch posts and docs say about agent behaviour matches what the tool does, and fix whichever is wrong. Agents and the people who set them up rely on those statements.",
      "related": [
        "BOUND-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    },
    {
      "id": "EV-0040",
      "title": "Agent skills recommended a package that does not exist",
      "claim": "Merged pull requests in Vercel's agent plugin repository corrected skill instructions that recommended an npm package that is not published, and plugin guidance that advertised deployment cards the production MCP server does not expose.",
      "patterns": [
        "drift",
        "skills"
      ],
      "finding_type": "observation",
      "effect": [
        {
          "metric": "Facts in agent instructions that did not match production",
          "baseline": null,
          "treatment": "A recommended package (@neondatabase/vercel-postgres-compat) returns E404 on npm; AI SDK version facts were wrong; plugin instructions advertised interactive deployment cards production MCP does not expose and described the connection as read-only",
          "direction": "not-applicable",
          "magnitude": "Several drifted facts corrected across two pull requests",
          "n": "Two merged pull requests (#306, #307), following a run of drift pull requests dated 2026-10-05"
        }
      ],
      "agent_profile": {
        "models": [],
        "models_note": "Not an agent run: corrections to agent-facing instruction files.",
        "harness": null,
        "extensions": [],
        "os": null,
        "run_dates": null
      },
      "evidence_class": "vendor-measurement",
      "conflicts_of_interest": "Vercel's own pull requests about its own plugin; recorded as an example of drift, not a criticism.",
      "source": {
        "url": "https://github.com/vercel/vercel-plugin/pull/306",
        "title": "vercel/vercel-plugin#306: Fix the leftovers the 2026-10-05 drift PRs corrected elsewhere",
        "publisher": "GitHub",
        "authors": [
          "molebox"
        ],
        "published": "2026-10-05"
      },
      "corroborating_sources": [
        {
          "url": "https://github.com/vercel/vercel-plugin/pull/307",
          "title": "vercel/vercel-plugin#307: [plugin] align mcp guidance and preserve submission safeguards",
          "publisher": "GitHub",
          "authors": [
            "joshuasoup"
          ],
          "published": "2026-10-06"
        }
      ],
      "retrieved": "2026-10-08",
      "verification": "verified-live",
      "verification_note": "The full primary source was fetched on 2026-10-08 and every number in this record was found in it. Both pull requests were merged (2026-10-05 and 2026-10-06). `npm view @neondatabase/vercel-postgres-compat` returned E404 on the retrieval date.",
      "limitations": [
        "One vendor's repository; shows that drift happens, not how often."
      ],
      "implications": "Skills, plugins and llms.txt files go out of date as products change. Check mechanically that every package, tool and command they name exists in production.",
      "related": [
        "FIND-04",
        "BOUND-03"
      ],
      "supersedes": [],
      "contradicts": [],
      "added": "2026-10-08"
    }
  ]
}
