{
  "title": "I Tried to Give an AI Agent Real Authority. The Model Wasn’t the Hard Part.",
  "canonical_url": "https://calvinkennedy.com/writing/ai-agent-gold-rush-control-layer",
  "thesis": "Agents can act before organisations can verify reality. Bound actions, enforce authority at commit time, verify externally, and expand permission only when evidence stays fresh.",
  "current_evidence_ceiling": "A2",
  "production_effects": "blocked",
  "human_value": [
    "understand problem, control, and payoff in forty-five seconds",
    "experience an unknown-outcome payment failure",
    "compare capability with authority",
    "manipulate reliability and expected-value mathematics",
    "choose a defensible autonomy ceiling"
  ],
  "agent_value": [
    "machine-readable A/B/C argument map",
    "three concrete implementation patterns",
    "legal action set and commit-time authority contract",
    "independent evidence requirements",
    "evaluation and release gates",
    "explicit stop and reconciliation conditions"
  ],
  "required_architecture": [
    "frontier planner",
    "JEV-style bounded selector",
    "deterministic policy",
    "authoritative effect boundary",
    "independent verifier",
    "evidence ledger",
    "runtime contraction"
  ],
  "default_contract": {
    "domain": "operations",
    "goal": "Prepare a bounded recommendation without creating an external effect.",
    "authority": "A2",
    "effect": "none",
    "maximumLossAud": 500,
    "evidenceTtlMinutes": 30,
    "independentVerifier": true,
    "retryRule": "Unknown outcomes must reconcile authoritative state before retry.",
    "stopConditions": [
      "authority expired",
      "policy changed",
      "evidence stale",
      "verifier unavailable",
      "unexpected external state"
    ],
    "requiredEvidence": [
      "identity",
      "scope",
      "current policy",
      "authoritative state",
      "complete trajectory",
      "independent outcome check"
    ]
  },
  "implementation_request": "Produce the operating domain, legal actions, authority matrix, evidence model, failure modes, evaluations, monitors, gates, and smallest effect-disabled shadow experiment. Mark unsupported assumptions.",
  "sources": [
    {
      "id": "S01",
      "title": "Wikipedia: Signs of AI writing",
      "publisher": "Wikipedia / WikiProject AI Cleanup",
      "url": "https://en.wikipedia.org/wiki/Wikipedia:Signs_of_AI_writing",
      "use": "Editorial audit: generic significance claims, vague attribution, superficial analysis, formulaic structure, excessive formatting.",
      "caveat": "The page describes possible patterns, not reliable proof of AI authorship."
    },
    {
      "id": "S02",
      "title": "Interrogating Design Homogenization in Web Vibe Coding",
      "publisher": "Shin et al., arXiv 2603.13036",
      "url": "https://arxiv.org/abs/2603.13036",
      "use": "Design audit: frictionless generation can reproduce dominant conventions; productive friction can preserve creator intent.",
      "caveat": "A 2026 preprint and sociotechnical analysis, not a universal visual-quality test."
    },
    {
      "id": "S03",
      "title": "Progressive Disclosure",
      "publisher": "Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/progressive-disclosure/",
      "use": "HCI structure: reveal advanced detail when needed while keeping core actions visible.",
      "caveat": "Progressive disclosure can hide important information when hierarchy is designed badly."
    },
    {
      "id": "S04",
      "title": "10 Usability Heuristics for User Interface Design",
      "publisher": "Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/ten-usability-heuristics/",
      "use": "Visible state, user control, error prevention, recognition over recall, and recovery support.",
      "caveat": "Heuristics guide review; they do not replace testing with representative users."
    },
    {
      "id": "S05",
      "title": "Animation from Interactions",
      "publisher": "W3C WCAG 2.2 Understanding Document",
      "url": "https://www.w3.org/WAI/WCAG22/Understanding/animation-from-interactions.html",
      "use": "Motion system: reader-triggered animation must be disable-able unless essential.",
      "caveat": "Passing one motion criterion does not establish complete accessibility."
    },
    {
      "id": "S06",
      "title": "TEVV-Athlon Framework for Evaluating AI Systems",
      "publisher": "NIST",
      "url": "https://www.nist.gov/artificial-intelligence/ai-research/tevv-athlon-framework-evaluating-ai-systems",
      "use": "Evaluation framing: objectives, context, measurement concepts, events, tools, and lifecycle evidence.",
      "caveat": "The cited 2026 document is an initial public draft."
    },
    {
      "id": "S07",
      "title": "Demystifying evals for AI agents",
      "publisher": "Anthropic Engineering",
      "url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents",
      "use": "Evaluation vocabulary: tasks, trials, graders, transcripts, trajectories, outcomes, harnesses, and repeated trials.",
      "caveat": "Vendor engineering guidance; domain-specific assurance still requires independent design."
    },
    {
      "id": "S08",
      "title": "Making retries safe with idempotent APIs",
      "publisher": "AWS Builders' Library",
      "url": "https://aws.amazon.com/builders-library/making-retries-safe-with-idempotent-APIs/",
      "use": "Stable caller intent, idempotency identity, retries, and unknown-outcome handling.",
      "caveat": "Patterns must be adapted to each external system's actual guarantees."
    },
    {
      "id": "S09",
      "title": "Addressing Cascading Failures",
      "publisher": "Google Site Reliability Engineering",
      "url": "https://sre.google/sre-book/addressing-cascading-failures/",
      "use": "Retry budgets, load shedding, admission control, graceful degradation, and overload testing.",
      "caveat": "Service reliability patterns do not by themselves validate agent decisions."
    },
    {
      "id": "S10",
      "title": "Remote ATtestation procedureS Architecture",
      "publisher": "IETF RFC 9334",
      "url": "https://datatracker.ietf.org/doc/rfc9334/",
      "use": "Separate evidence producer, verifier, and authority-deciding relying party.",
      "caveat": "Remote attestation concepts do not prove external business outcomes without appropriate evidence."
    },
    {
      "id": "S11",
      "title": "Model Context Protocol: Authorization",
      "publisher": "Model Context Protocol",
      "url": "https://modelcontextprotocol.io/specification/2025-11-25/basic/authorization",
      "use": "Protocol authentication and authorization boundaries for tool connectivity.",
      "caveat": "Protocol authorization does not substitute for local business policy and effect controls."
    },
    {
      "id": "S12",
      "title": "Bounded Agentic Assurance V10 execution record",
      "publisher": "Calvin Kennedy research artifact",
      "url": "evidence/CANONICAL_EXECUTION_RECORD_V10.json",
      "use": "Article-specific synthetic metrics, failure simulations, and authority ceiling.",
      "caveat": "Synthetic and bounded reference evidence; no production autonomy claim."
    },
    {
      "id": "S13",
      "title": "The Agent Trust Gap: What Our Research Reveals About Agentic AI Security",
      "publisher": "Cisco Security, March 2026",
      "url": "https://blogs.cisco.com/security/the-agent-trust-gap-what-our-research-reveals-about-agentic-ai-security",
      "use": "Current adoption gap: 85% experimenting, piloting, or deploying; 5% broad production; security and access control remain core barriers.",
      "caveat": "Cisco surveyed its own customer base; the results are not a census of all organisations."
    },
    {
      "id": "S14",
      "title": "The pulse of Agentic AI in 2026",
      "publisher": "Dynatrace LinkedIn post and survey report",
      "url": "https://www.linkedin.com/posts/dynatrace_the-pulse-of-agentic-ai-in-2026-activity-7420564449478053889-W9u1",
      "use": "Current manual-verification signal: the published graphic states that 69% manually verify agentic-AI decisions.",
      "caveat": "A vendor survey of 919 leaders; methodology and wording should be inspected in the full report before generalising."
    },
    {
      "id": "S15",
      "title": "Agentic AI in Industry: Adoption Level and Deployment Barriers",
      "publisher": "Apostolou, Bosch, and Holmström Olsson, arXiv 2605.14675",
      "url": "https://arxiv.org/abs/2605.14675",
      "use": "Industrial evidence for a capability-deployment verification gap across interviews with 16 practitioners from 12 companies.",
      "caveat": "A small qualitative sample and preprint; useful for mechanisms, not population estimates."
    },
    {
      "id": "S16",
      "title": "Tasmania’s Justice Department to review AI use in parole board decisions",
      "publisher": "ABC News, 19 September 2026",
      "url": "https://www.abc.net.au/news/2026-09-19/parole-board-ai-use-review-after-neill-fraser-case/107172064",
      "use": "A current Australian high-stakes example where fictitious case law, likely AI-generated, entered decision materials.",
      "caveat": "This is one reported incident under review; it does not establish prevalence across the justice system."
    },
    {
      "id": "S17",
      "title": "JPMorgan rolls out Claude changes: spending limits and extra security",
      "publisher": "Business Insider, September 2026",
      "url": "https://www.businessinsider.com/jpmorgan-claude-spending-limit-security-engineers-2026-9",
      "use": "Illustrates cost caps and isolated execution environments appearing in serious enterprise deployment practice.",
      "caveat": "Paywalled reporting based on company sources; implementation details are incomplete."
    },
    {
      "id": "S18",
      "title": "Huawei forecasts billions of agents will dominate AI traffic by 2035",
      "publisher": "Reuters, 16 September 2026",
      "url": "https://www.reuters.com/legal/litigation/chinas-huawei-forecasts-billions-agents-will-dominate-ai-traffic-by-2035-2026-09-16/",
      "use": "Illustrates the scale of current industry expectations for agent-generated traffic and infrastructure demand.",
      "caveat": "A corporate forecast, not an observed deployment count or neutral prediction."
    },
    {
      "id": "S19",
      "title": "State of UX 2026: Design Deeper to Differentiate",
      "publisher": "Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/state-of-ux-2026/",
      "use": "Design direction: AI trust requires transparency, control, consistency, and support when systems fail.",
      "caveat": "Expert synthesis, not a controlled validation of this article’s interface."
    },
    {
      "id": "S20",
      "title": "There are 3 telltale signs that you used AI to make your app",
      "publisher": "Business Insider, July 2026",
      "url": "https://www.businessinsider.com/ai-coded-app-user-interface-experience-design-2026-7",
      "use": "Secondary industry critique of homogeneous visuals, polished surfaces masking weak function, and neglected edge cases.",
      "caveat": "Journalistic synthesis and expert commentary, not a formal UI-quality instrument."
    },
    {
      "id": "S21",
      "title": "Vibe Coding: Practice, Performance, Productivity, and Risk",
      "publisher": "Michels et al., arXiv 2608.20446",
      "url": "https://arxiv.org/abs/2608.20446",
      "use": "Research synthesis showing that productivity claims vary substantially with task, measurement method, codebase maturity, and time horizon.",
      "caveat": "A recent preprint review; some included studies and claims may change after peer review."
    },
    {
      "id": "S22",
      "title": "How many words do we read per minute? A review and meta-analysis of reading rate",
      "publisher": "Marc Brysbaert, Journal of Memory and Language, 2019",
      "url": "https://doi.org/10.1016/j.jml.2019.104047",
      "use": "Reading-time calibration; adult silent nonfiction reading commonly falls within a broad range, so this page uses a conservative 200-wpm technical baseline.",
      "caveat": "Population averages do not predict one reader, one device, or comprehension of specialised material."
    },
    {
      "id": "S23",
      "title": "Understanding Success Criterion 1.4.10: Reflow",
      "publisher": "W3C WCAG 2.2 Understanding Document",
      "url": "https://www.w3.org/WAI/WCAG22/Understanding/reflow.html",
      "use": "Responsive fallback: text must reflow without two-dimensional scrolling at narrow equivalent widths.",
      "caveat": "The Understanding document is informative; a single criterion does not establish complete accessibility."
    },
    {
      "id": "S24",
      "title": "Understanding Success Criterion 2.5.8: Target Size (Minimum)",
      "publisher": "W3C WCAG 2.2 Understanding Document",
      "url": "https://www.w3.org/WAI/WCAG22/Understanding/target-size-minimum.html",
      "use": "Minimum target-size and spacing checks for interactive controls across mobile layouts.",
      "caveat": "Minimum conformance is not the same as comfortable use; this article generally targets larger controls."
    },
    {
      "id": "S25",
      "title": "Understanding Success Criterion 2.3.3: Animation from Interactions",
      "publisher": "W3C WCAG 2.2 Understanding Document",
      "url": "https://www.w3.org/WAI/WCAG22/Understanding/animation-from-interactions.html",
      "use": "Motion controls and reduced-motion fallbacks for reader-triggered animation.",
      "caveat": "Disabling motion does not by itself guarantee cognitive or vestibular accessibility."
    },
    {
      "id": "S26",
      "title": "Progressive Disclosure",
      "publisher": "Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/progressive-disclosure/",
      "use": "Three reading lenses: show the essential claim first, narrative context on request, and engineering evidence in Audit.",
      "caveat": "Poorly chosen hierarchy can conceal important information; the split still requires user testing."
    },
    {
      "id": "S27",
      "title": "Multimedia Learning, third edition: signaling and contiguity principles",
      "publisher": "Richard E. Mayer, Cambridge University Press",
      "url": "https://www.cambridge.org/core/books/multimedia-learning/",
      "use": "Keep explanatory words, visual state, and causal timing close enough to be mentally integrated.",
      "caveat": "Learning principles do not yield a universal scroll formula; V6’s choreography coefficients are local design heuristics."
    },
    {
      "id": "S28",
      "title": "Narrative Visualization: Telling Stories with Data",
      "publisher": "Segel and Heer, IEEE Transactions on Visualization and Computer Graphics",
      "url": "https://doi.org/10.1109/TVCG.2010.179",
      "use": "Balance authored narrative sequence with reader-controlled exploration in scrollytelling and simulations.",
      "caveat": "A design-space analysis, not evidence that any one narrative interface improves comprehension."
    },
    {
      "id": "S29",
      "title": "Concise, SCANNABLE, and Objective: How to Write for the Web",
      "publisher": "Morkes and Nielsen / Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/concise-scannable-and-objective-how-to-write-for-the-web/",
      "use": "Scan-route design: concise language, scannable structure, and objective wording are treated as separate usability levers.",
      "caveat": "A small late-1990s study; its exact effect sizes should not be treated as universal contemporary constants."
    },
    {
      "id": "S30",
      "title": "Rational Analyses of Information Foraging on the Web",
      "publisher": "Peter Pirolli, Cognitive Science, 2005",
      "url": "https://doi.org/10.1207/s15516709cog0000_20",
      "use": "Information-scent model: readers continually compare expected information value against the cost of continuing.",
      "caveat": "A cognitive modelling framework, not a direct comprehension test of this article."
    },
    {
      "id": "S31",
      "title": "VISHIEN-MAAT: Scrollytelling visualization design for explaining Siamese Neural Network concepts",
      "publisher": "Chotisarn et al., 2023",
      "url": "https://arxiv.org/abs/2304.03288",
      "use": "Scrollytelling design: coordinate authored sequence, reader pace, and visual explanation for a complex AI concept.",
      "caveat": "A specific study and concept; it does not prove every scrollytelling interface improves understanding."
    },
    {
      "id": "S32",
      "title": "Predicting Text Readability from Scrolling Interactions",
      "publisher": "Gooding et al., 2021",
      "url": "https://aclanthology.org/2021.emnlp-main.324/",
      "use": "Supports treating scroll behaviour and reading difficulty as related signals rather than assuming a fixed universal reading path.",
      "caveat": "Interaction patterns predict readability statistically; they do not reveal whether one reader understood a specific claim."
    },
    {
      "id": "S33",
      "title": "Information Foraging: A Theory of How People Navigate on the Web",
      "publisher": "Nielsen Norman Group",
      "url": "https://www.nngroup.com/articles/information-foraging/",
      "use": "Practical information-scent guidance for labels, routes, and deciding whether continuing appears worth the effort.",
      "caveat": "A practitioner synthesis; representative user testing remains necessary."
    },
    {
      "id": "S34",
      "title": "Getting the Most from Eye-Tracking: User-Interaction Based Reading Region Estimation",
      "publisher": "Kong et al., 2023",
      "url": "https://arxiv.org/abs/2306.07455",
      "use": "Persona modelling reference: distinguish skip, skim, and detailed reading at the content-region level.",
      "caveat": "The article uses heuristic simulations, not eye-tracking or validated per-user reading inference."
    }
  ],
  "release": "article-v6",
  "argument_map": {
    "A_problem": "The model can be locally reasonable while its view of external reality is stale or incomplete.",
    "B_control": "Typed legal actions, deterministic commit-time authority, idempotent effects, reconciliation, and independent verification.",
    "C_payoff": "More verified business value per unit of human attention, compute, and risk."
  }
}
