{
  "name": "AgentScore API",
  "version": "v1",
  "description": "The API benchmark, scored from an agent's point of view. Scores are derived from real runs and probes, not from reviews.",
  "documentation": "https://agentsco.re/methodology",
  "openapi": "https://agentsco.re/openapi.json",
  "llms_txt": "https://agentsco.re/llms.txt",
  "endpoints": [
    {
      "method": "GET",
      "path": "/api/v1/tools",
      "description": "List the scored tools.",
      "query": {
        "category": "Filter by category. See /api/v1/categories.",
        "limit": "1 to 100, default 25.",
        "cursor": "Identifier of the last item on the previous page.",
        "fields": "Comma-separated list. Shrinks the response."
      }
    },
    {
      "method": "GET",
      "path": "/api/v1/tools/{id}",
      "description": "Detail of a tool and its surfaces."
    },
    {
      "method": "GET",
      "path": "/api/v1/categories",
      "description": "Categories and whether they are comparable. A score is only comparable within one category."
    },
    {
      "method": "GET",
      "path": "/api/v1/rubric",
      "description": "Criteria, weights and methods of the current rubric."
    },
    {
      "method": "GET",
      "path": "/tools/{id}.md",
      "description": "Markdown twin of the entry. Also served on /tools/{id} with Accept: text/markdown."
    }
  ],
  "score_semantics": {
    "scale": "0 to 100, always oriented higher-is-better, costs included.",
    "statuses": {
      "not_yet_scored": "Surface catalogued, benchmark not run. `overall` is null: this is an ABSENCE OF MEASUREMENT, never a score of zero. Do not rank this surface last.",
      "partial": "Partial benchmark: coverage is below the publication threshold, so no overall score is computed. The criteria present in `criteria` are valid and usable individually.",
      "scored": "Complete benchmark. `overall` is a weighted average of the applicable criteria, comparable to other surfaces in the same category and the same rubric version."
    },
    "comparability": "Two scores are only comparable within the same category AND the same rubric version. `interaction_cost` and `task_atomicity` are normalised against the best in category.",
    "null_is_not_zero": "A missing criterion means NOT MEASURED or NOT APPLICABLE. It is excluded from the computation and the weights are renormalised — it is never counted as zero."
  },
  "rubric": {
    "version": "0.2.0",
    "status": "draft",
    "criteria_count": 11,
    "min_weight_coverage": 0.7
  },
  "conventions": {
    "errors": "RFC 9457 (application/problem+json). Every 4xx names the offending field and lists the allowed values.",
    "rate_limits": "X-RateLimit-Limit / -Remaining / -Reset headers on every response. A 429 always carries a usable Retry-After.",
    "caching": "ETag on 200 responses. Send If-None-Match back to get a 304.",
    "auth": "No key required today. An Authorization: Bearer <key> header is already accepted and will be required beyond the free tier — wire it in now, it will not break."
  }
}