{
  "scorecard": "Open Agentic Platform openness scorecard",
  "version": "1.0",
  "source": "https://openagenticplatform.com/openness-scorecard",
  "license": "CC BY 4.0, attribution Alex Merced, openagenticplatform.com",
  "instructions": "Score each test 0 (absent), 1 (partial), or 2 (demonstrated) and record the evidence. Total is out of 12.",
  "stack": null,
  "assessedOn": null,
  "scale": {
    "0": "absent",
    "1": "partial",
    "2": "demonstrated"
  },
  "tests": [
    {
      "id": "replaceable",
      "title": "Replaceable",
      "question": "Can one component be swapped without rebuilding the system?",
      "rubric": {
        "0": "Changing the model, engine, or harness means rebuilding most of the system.",
        "1": "Some components swap cleanly; others need prompts, definitions, or integrations rebuilt.",
        "2": "A major component has been swapped, and the work was measured in days, not months."
      },
      "evidencePrompt": "Which component did you last swap, and how long did it take?",
      "score": null,
      "evidence": null
    },
    {
      "id": "inspectable",
      "title": "Inspectable",
      "question": "Can a builder understand what runs and why?",
      "rubric": {
        "0": "Behavior is visible only through final outputs.",
        "1": "A console shows traces, but they cannot be exported or leave out the exact context sent.",
        "2": "The exact context, tool calls, and results for each step are exported to storage you own."
      },
      "evidencePrompt": "Where do traces live, what do they contain, and how long are they kept?",
      "score": null,
      "evidence": null
    },
    {
      "id": "portable",
      "title": "Portable",
      "question": "Can identity, skills, context, and work move?",
      "rubric": {
        "0": "Agent definitions, skills, and memory exist only inside one product.",
        "1": "Some artifacts export, but in a proprietary format or with pieces missing.",
        "2": "Identity, skills, and work are files in open formats and have been loaded into a second tool."
      },
      "evidencePrompt": "What have you exported and reloaded somewhere else?",
      "score": null,
      "evidence": null
    },
    {
      "id": "bounded",
      "title": "Bounded",
      "question": "Are authority and approval requirements explicit?",
      "rubric": {
        "0": "Limits are requests written into the prompt.",
        "1": "Role permissions exist, but there are no per-action scopes or approvals.",
        "2": "Tool scopes, delegated user identity, and approvals are enforced outside the model and tested adversarially."
      },
      "evidencePrompt": "What stops the agent from taking an action its user could not take?",
      "score": null,
      "evidence": null
    },
    {
      "id": "grounded",
      "title": "Grounded",
      "question": "Do agents share durable data and semantic meaning?",
      "rubric": {
        "0": "Agents answer from model memory or guess meaning from column names.",
        "1": "Agents reach real data, but without shared metric definitions.",
        "2": "Agents query governed metrics over versioned tables and cite the definition and data version."
      },
      "evidencePrompt": "Which semantic definitions and tables do agents use?",
      "score": null,
      "evidence": null
    },
    {
      "id": "auditable",
      "title": "Auditable",
      "question": "Can people reconstruct decisions and outcomes?",
      "rubric": {
        "0": "There is no durable record of agent runs.",
        "1": "Logs are partial or kept only briefly.",
        "2": "Runs, approvals, and data versions are recorded long enough to reconstruct any decision."
      },
      "evidencePrompt": "Could you reconstruct an agent decision from three months ago?",
      "score": null,
      "evidence": null
    }
  ],
  "total": null,
  "max": 12,
  "bands": [
    {
      "from": 0,
      "to": 4,
      "label": "Closed in practice",
      "advice": "Most exits would mean a rebuild. Start with the lowest-scoring test that blocks the next change you already know is coming."
    },
    {
      "from": 5,
      "to": 8,
      "label": "Partly open",
      "advice": "Some seams are real and some are assumed. Fix the weakest test first; a single 0 usually costs more than several 1s."
    },
    {
      "from": 9,
      "to": 11,
      "label": "Open with gaps",
      "advice": "The architecture holds up. Turn each remaining 1 into a 2 by exercising it: swap, export, or replay something for real."
    },
    {
      "from": 12,
      "to": 12,
      "label": "Open by evidence",
      "advice": "Every test is demonstrated. Keep the evidence current and repeat the score when a major component changes."
    }
  ]
}
