{
  "schema": "yarn-pack/2",
  "id": "framework-agentic",
  "version": "0.1.0",
  "name": "Agentic AI governance (framework self-assessment)",
  "engagement": "AISG — framework self-assessment: workshop prep, workshop, or diagnostic strand",
  "intro": "A guided conversation against Agentic AI governance, not a form. Answer in your own words and name the document or record that shows it if you can. About 22 minutes. Assesses the definitional test and the ten-control set the practice synthesised for agentic systems from OWASP, IMDA and the AISI. Not a certifiable standard: no ISO agentic standard exists.",
  "tone_default": "professional",
  "prefill_fields": {
    "department": [
      "Executive",
      "Finance",
      "Operations",
      "Customer / Sales",
      "Technology / IT",
      "Data & Analytics",
      "People & Culture",
      "Risk & Compliance",
      "Marketing",
      "Product"
    ]
  },
  "scales": [
    {
      "id": "maturity5",
      "name": "Maturity (distilled model)",
      "levels": [
        {
          "value": 1,
          "label": "Does not exist",
          "gloss": "No capability. Absent, or purely ad hoc / accidental."
        },
        {
          "value": 2,
          "label": "Partially exists",
          "gloss": "Emerging and inconsistent. Pockets of activity, not joined up."
        },
        {
          "value": 3,
          "label": "Fully exists",
          "gloss": "Defined, documented and operating across the organisation."
        },
        {
          "value": 4,
          "label": "Fully exists & optimised",
          "gloss": "Measured, refined and improving against targets."
        },
        {
          "value": 5,
          "label": "Fully exists & adaptive",
          "gloss": "Continuously self-adjusting; a source of advantage."
        }
      ],
      "signals": {
        "1": [
          "no ",
          "not ",
          "none",
          "never",
          "don't",
          "do not",
          "nothing",
          "absent",
          "unaware",
          "haven't",
          "ad hoc",
          "ad-hoc",
          "nonexistent",
          "no idea",
          "not really"
        ],
        "2": [
          "some ",
          "starting",
          "beginning",
          "emerging",
          "pilot",
          "trial",
          "informal",
          "inconsistent",
          "pockets",
          "a bit",
          "occasionally",
          "early",
          "experiment",
          "trying",
          "patchy"
        ],
        "3": [
          "documented",
          "defined",
          "standard",
          "standardised",
          "established",
          "policy",
          "framework",
          "process",
          "consistent",
          "across the",
          "in place",
          "formal",
          "governed",
          "rolled out"
        ],
        "4": [
          "measured",
          "metrics",
          "optimis",
          "improving",
          "kpi",
          "monitored",
          "reviewed",
          "refined",
          "benchmarked",
          "targets",
          "tracked",
          "mature",
          "regularly review"
        ],
        "5": [
          "continuous",
          "adaptive",
          "self-",
          "automated end",
          "best in class",
          "best-in-class",
          "competitive advantage",
          "industry leading",
          "always",
          "real-time monitoring",
          "feedback loop"
        ]
      }
    }
  ],
  "categories": [
    {
      "id": "define",
      "name": "Definition and register",
      "order": 1,
      "target_default": 3
    },
    {
      "id": "identity",
      "name": "Identity and scope",
      "order": 2,
      "target_default": 3
    },
    {
      "id": "bound",
      "name": "Boundaries and approvals",
      "order": 3,
      "target_default": 3
    },
    {
      "id": "observe",
      "name": "Observability and stopping",
      "order": 4,
      "target_default": 3
    }
  ],
  "audiences": [
    {
      "id": "lead",
      "name": "Governance / risk lead",
      "desc": "Owns the policy, the register or the risk framework",
      "deep_dive_sections": []
    },
    {
      "id": "owner",
      "name": "System or use-case owner",
      "desc": "Runs an AI system or use case day to day",
      "deep_dive_sections": []
    },
    {
      "id": "exec",
      "name": "Executive / sponsor",
      "desc": "Accountable for the outcome, not the mechanics",
      "deep_dive_sections": []
    }
  ],
  "sections": [
    {
      "id": "core",
      "title": "Agentic elements",
      "blurb": "Everyone answers these. The named accountable owner of each agent, whoever runs identity and access (the ISMS team), the engineering lead who built the agents, and the executive who will answer ‘who authorised this?’.",
      "optional": false,
      "questions": [
        {
          "id": "agentic-def",
          "type": "scored_text",
          "category": "define",
          "name": "Definitional discipline",
          "text": "Walk me through what your most autonomous system can actually do on its own. If it drafted something wrong right now, would a person catch it before it went out, or would it already be sent?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "No distinction is drawn; ‘agentic’ is used for automation and assistants alike, and nobody can say what any system can do without asking anyone.",
            "3": "Each AI system is classified against the four tests (plans, holds memory, calls tools, delegated authority) and its ‘without asking anyone’ answer is recorded.",
            "5": "Classification is re-run whenever a system gains tools, memory or authority, so a system that crosses into agentic is picked up before it goes live."
          },
          "help": "Evidence that would show it: Classification of each AI system against the four agentic tests; Recorded answer per system to ‘what can it do without asking anyone?’; Use case register with an agentic flag per row.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-register",
          "type": "scored_text",
          "category": "define",
          "name": "Agent register",
          "text": "How many agents are running in your organisation right now, and who owns each one? If I picked one at random, who would answer for what it did last week?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "No agent inventory exists; agents are found by asking around, and no one is named as the owner of any of them.",
            "3": "A register lists every agent with a named accountable human owner per row, and it is the source the NSW register or APRA AI inventory is answered from.",
            "5": "No agent reaches production without a register row and an owner; the register is reconciled against what is actually running and unregistered agents are flagged."
          },
          "help": "Evidence that would show it: Agent register with an accountable human owner per row; Reconciliation of the register against agents running in production; NSW AI use case register entries or APRA AI inventory drawn from it.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-identity",
          "type": "scored_text",
          "category": "identity",
          "name": "Distinct agent identity",
          "text": "When one of your agents does something, what name does it show up as in your logs? Is that its own identity, or does it borrow a person’s login or a shared account?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Agents run on a shared service account or a staff member’s login; when something goes wrong, ‘who authorised this?’ has no clean answer.",
            "3": "Every agent has its own identity, distinct from any human and any other agent, and each action is attributable to that agent and the authority it acted under.",
            "5": "Agent identities are verifiable to a peer (for instance a signed Agent Card) and delegation chains are recorded, so a fake or drifted peer is rejected automatically."
          },
          "help": "Evidence that would show it: Identity directory entries, one per agent, none shared with a person; Credential inventory showing no agent running on a human user’s login; Delegation record: which agent acted under whose authorisation.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-privilege",
          "type": "scored_text",
          "category": "identity",
          "name": "Least privilege and short-lived credentials",
          "text": "If one of your agents was compromised this afternoon, what could the attacker reach with its credentials, and for how long before they stopped working?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Agents hold standing, broad credentials that never expire; a compromised agent exposes everything it can reach.",
            "3": "Each agent’s permissions are scoped to its task, its credentials are short-lived and expiring, and the scope is documented and reviewed.",
            "5": "Scope is issued per task at run time from a safe default and withdrawn on completion; over-scoped grants are detected and revoked without waiting for a review."
          },
          "help": "Evidence that would show it: Permission scope per agent, mapped to its task; Credential lifetime and expiry settings for agent credentials; Access review records for agent permissions.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-tools",
          "type": "scored_text",
          "category": "identity",
          "name": "Tool allowlists",
          "text": "Can you list the tools each agent is allowed to call? If someone stood up a new MCP server tomorrow, could an agent in production start using it without anyone signing off?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Agents discover and call whatever tools they can reach; nobody can list which tools an agent used last month or who approved them.",
            "3": "An explicit allowlist per agent exists and is enforced; runtime discovery of arbitrary tools is blocked in production, and the list is reviewed.",
            "5": "Any change to an agent’s tool set triggers reassessment before it takes effect, so the capability boundary assessed at go-live is kept current."
          },
          "help": "Evidence that would show it: Tool allowlist per agent, with reviewer and review date; Gateway or configuration proving runtime tool discovery is off in production; Change record for additions to an allowlist.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-sandbox",
          "type": "scored_text",
          "category": "identity",
          "name": "Sandboxed execution",
          "text": "Where does the agent’s code actually run? If it generated a command to reach a system you never gave it, what stops that from working?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Agents execute on hosts with access to the wider environment; generated code or commands can run with whatever the host can reach.",
            "3": "Each agent runs in an isolated execution environment whose reach is limited to what was deliberately granted, and the isolation has been tested.",
            "5": "Sandbox escape attempts are detected and contained automatically, and isolation is re-tested whenever the agent’s tools or runtime change."
          },
          "help": "Evidence that would show it: Execution environment design showing the isolation boundary per agent; Test results for sandbox isolation or escape attempts; Record of what each sandbox was deliberately granted.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-limits",
          "type": "scored_text",
          "category": "bound",
          "name": "Action limits and blast-radius caps",
          "text": "What is the most an agent could do in an hour if it got stuck in a loop: how many actions, how much money, how many messages? Where is that limit set, and has it ever tripped?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "No limits on how much, how fast or how irreversibly an agent can act; a runaway loop is discovered by its consequences.",
            "3": "Rate, value, volume and irreversibility thresholds are set per agent and enforced in the runtime; an agent that hits one stops and alerts.",
            "5": "Thresholds are tuned from monitored behaviour and tightened on anomaly; a cap tripped in one agent is checked across the agents connected to it."
          },
          "help": "Evidence that would show it: Configured thresholds per agent: rate, value, volume, irreversibility; Alert and stop records from a threshold being hit; Runtime ceiling and worst plausible cost decided before go-live.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-approvals",
          "type": "scored_text",
          "category": "bound",
          "name": "Approval checkpoints at action boundaries",
          "text": "When an agent needs a person to say yes, what does that person actually see? Is it something the agent wrote, or something pulled from a source the agent cannot touch?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Approvals are added after build if at all; the agent takes irreversible actions unprompted, or the human approves from a screen the agent wrote.",
            "3": "Irreversible actions have a defined approval checkpoint set before build; the approver sees information rendered independently of what the agent reports.",
            "5": "Approval design is verified against ASI09 on every change, and automation bias on the approval step is monitored rather than assumed away."
          },
          "help": "Evidence that would show it: Action-boundary definition per agent, dated before build; Approval screen design showing the independent source the human sees; Approval records for irreversible actions.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-memory",
          "type": "scored_text",
          "category": "bound",
          "name": "Memory hygiene",
          "text": "What does your agent remember from one run to the next, where does that come from, and could someone plant something in it that changed how it behaved next week?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Long-term memory accumulates without scoping, expiry or any record of where an entry came from; nobody can say what an agent currently believes.",
            "3": "Memory is scoped per agent and task, entries expire, and each entry’s provenance is recorded; memory contents can be inspected and purged.",
            "5": "Entries from untrusted sources are quarantined or rejected automatically, and memory is re-validated when a poisoning signal is detected."
          },
          "help": "Evidence that would show it: Memory scoping and expiry configuration per agent; Provenance field on entries in long-term memory; Procedure and record for inspecting or purging agent memory.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-traces",
          "type": "scored_text",
          "category": "observe",
          "name": "Full action traces",
          "text": "Pick an agent and a day last week. Can you show me every tool it called, in order, and who it was acting for? How long would that take to pull?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "Only application logs exist; an agent’s tool calls and delegations cannot be reconstructed after the event.",
            "3": "Every tool call and delegation is traced with the agent identity and the authority it acted under, and any past run can be replayed step by step.",
            "5": "Traces are tamper-evident and drive detection: divergence from policy or expected behaviour raises an alert without a person reading the logs."
          },
          "help": "Evidence that would show it: Trace of a chosen past run: each tool call and delegation, replayable; Log schema showing agent identity and authorising principal per action; Retention setting for agent action traces.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        },
        {
          "id": "agentic-kill",
          "type": "scored_text",
          "category": "observe",
          "name": "Emergency termination",
          "text": "If an agent started doing something wrong at 2 am on a Sunday, who could stop it, how would they be reached, and when did you last prove the stop actually works?",
          "scale": "maturity5",
          "scored": true,
          "star": false,
          "rubric": {
            "1": "No way to stop an agent short of taking the platform down; nobody is named to pull the switch and it has never been tried.",
            "3": "A named, reachable owner can terminate any agent; the mechanism has been tested and the test is dated.",
            "5": "Termination triggers automatically on pre-defined alert thresholds (a tripped cap, a policy divergence) and is re-tested after every material change."
          },
          "help": "Evidence that would show it: Kill switch procedure with named owner and contact route; Dated record of the most recent termination test; Alert threshold configuration that can trigger termination.",
          "adaptive": {
            "allow_probe": true,
            "allow_skip": false,
            "max_probes": 1
          },
          "ai_drafted": false
        }
      ]
    }
  ],
  "grids": {},
  "outputs": [
    "Level per element and per category, gated",
    "Contested-element view (spread of 2 or more)",
    "Coverage of evidence: confirmed, stated, inferred",
    "Where to start, foundations first",
    "Printable client report"
  ],
  "report_defaults": [
    "rpt-maturity-standard"
  ],
  "playbook": {
    "sequence": [
      "define",
      "identity",
      "bound",
      "observe"
    ],
    "sequence_note": "Bound autonomy before you build, not after: specified upward from a safe default it is a config change, retrofitted downward it is a rebuild. The first three controls are non-negotiable.",
    "actions": {
      "define": {
        "to_3": [
          "Run the practice test on every AI system: what can it do without asking anyone?",
          "Stand up the agent register with a named accountable human owner per row"
        ],
        "to_5": [
          "Reconcile the register against production and treat an unregistered agent as a compliance failure"
        ]
      },
      "identity": {
        "to_3": [
          "Give every agent its own identity; reuse the ISMS identity, least-privilege and credential controls rather than rebuilding",
          "Write and enforce a tool allowlist per agent; turn off runtime tool discovery in production"
        ],
        "to_5": [
          "Adopt the protocol conventions (signed Agent Cards, SAFE-MCP) so identity and tool trust are verifiable to peers"
        ]
      },
      "bound": {
        "to_3": [
          "Define action boundaries and approval checkpoints before build, on every irreversible action",
          "Set rate, value, volume and irreversibility caps per agent; scope and expire memory, with provenance"
        ],
        "to_5": [
          "Verify the approver sees something independent of the agent (ASI09) and monitor the approval step for automation bias"
        ]
      },
      "observe": {
        "to_3": [
          "Trace every tool call and delegation so last Tuesday can be replayed with who authorised it",
          "Name a reachable kill-switch owner and test the switch"
        ],
        "to_5": [
          "Use pre-defined alert thresholds and automated monitoring to stop and alert without a person reading logs"
        ]
      }
    }
  }
}