{"rubric":"1.0.0","agentTested":false,"dimensions":[{"id":"D1","name":"Legibility","max":15,"blurb":"can an agent learn this tool from its own surface?"},{"id":"D2","name":"Structured I/O","max":20,"blurb":"is the output machine-parseable?"},{"id":"D3","name":"Non-blocking","max":20,"blurb":"does it ever hang when no human is present?"},{"id":"D4","name":"Context economy","max":15,"blurb":"how many tokens does it cost to use?"},{"id":"D6","name":"Safety rails","max":10,"blurb":"can an agent preview before it destroys?"},{"id":"D7","name":"Error recovery","max":20,"blurb":"does a mistake produce an honest, actionable signal?"},{"id":"D9","name":"Cold credentials","max":20,"blurb":"can an agent get authenticated without a human?"}],"grades":[[90,"A"],[80,"B"],[70,"C"],[60,"D"],[0,"F"]],"fairness":["N/A never penalizes. A non-applicable check is excluded from both numerator and denominator.","Subcommand help is unioned with top-level help because many CLIs document JSON, confirmation, and field flags only where they apply.","Filters, REPLs, and other tools designed to read stdin are not penalized for waiting on a non-delivering pipe.","Read-only tools are not penalized for missing dry-run or confirmation flags.","Every verdict requires at least three isolated trials with the same prompt and links to the raw trajectory."],"scoreTypes":[{"name":"Facts","method":"Deterministic capture and probes","licenses":"Claims about observed behavior—exit status, streams, hangs, flags, bytes, and schemas—and a readiness hypothesis.","forbids":"It never licenses “an agent passed” and is never used to rank the leaderboard."},{"name":"Verdict","method":"Three or more identical agent trials","licenses":"Claims that an agent completed or failed the sampled task, with median cost and cross-trial variance.","forbids":"It does not claim universal success across every task, version, model, or credential state."}],"principleMapping":{"count":72,"groups":[{"principles":"1–3","count":3,"dimension":"Frame","mapping":"Human DX and agent DX serve different operators; support both paths.","coverage":"advisory"},{"principles":"4–12","count":9,"dimension":"D2","mapping":"Structured request payloads, JSON output, and streamable non-TTY defaults.","coverage":"cold"},{"principles":"13–18","count":6,"dimension":"D1","mapping":"Runtime schema, self-description, types, scopes, and canonical capability discovery.","coverage":"cold"},{"principles":"19–21","count":3,"dimension":"D4","mapping":"Field masks and response shaping conserve agent context.","coverage":"cold"},{"principles":"22–23","count":2,"dimension":"D2","mapping":"NDJSON pagination and incremental processing.","coverage":"cold / auth"},{"principles":"24–25","count":2,"dimension":"D4","mapping":"Shipped guidance teaches context conservation.","coverage":"static"},{"principles":"26–38","count":13,"dimension":"D5","mapping":"Treat agent input as adversarial: traversal, controls, query fragments, and double encoding.","coverage":"cold / advisory"},{"principles":"39–40","count":2,"dimension":"D8","mapping":"Ship structured skill files for each surface and workflow.","coverage":"static"},{"principles":"41","count":1,"dimension":"D1","mapping":"Agent guidance must be reachable from the tool itself.","coverage":"static"},{"principles":"42–50","count":9,"dimension":"D8","mapping":"Encode invariants and expose schema-derived MCP and native agent surfaces.","coverage":"cold / static"},{"principles":"51–55","count":5,"dimension":"D3 / D9","mapping":"Credential injection and headless authentication must not require a browser.","coverage":"cold / static"},{"principles":"56–61","count":6,"dimension":"D6","mapping":"Dry-run, confirmation, sanitization, and response prompt-injection defenses.","coverage":"cold / auth"},{"principles":"62","count":1,"dimension":"D2","mapping":"Retrofit machine-readable JSON output first.","coverage":"cold"},{"principles":"63","count":1,"dimension":"D5","mapping":"Then validate agent-supplied input defensively.","coverage":"cold"},{"principles":"64","count":1,"dimension":"D1","mapping":"Then expose a queryable schema.","coverage":"cold"},{"principles":"65","count":1,"dimension":"D4","mapping":"Then add field selection and context controls.","coverage":"cold"},{"principles":"66","count":1,"dimension":"D6","mapping":"Then add dry-run for mutation.","coverage":"cold"},{"principles":"67–68","count":2,"dimension":"D8","mapping":"Then ship skills and an MCP surface from the same source.","coverage":"cold / static"},{"principles":"69–70","count":2,"dimension":"D5","mapping":"Fuzz agent-typical mistakes, double encoding, and control characters.","coverage":"cold"},{"principles":"71–72","count":2,"dimension":"D6","mapping":"Verify dry-run validation and defend against prompt injection in retrieved data.","coverage":"cold / auth"}]},"versionPolicy":[{"level":"Patch","rule":"Clarifies wording or fixes an implementation bug without changing intended scoring. Existing reports stay pinned."},{"level":"Minor","rule":"Adds or reweights a check. A new rubric version is published first and affected corpus reports are rerun."},{"level":"Major","rule":"Changes a dimension or what a score licenses. Old permalinks remain readable with their original rubric version."}]}