{
  "specVersion": "1.0",
  "host": {
    "displayName": "Evals by Jetty",
    "identifier": "did:web:evaljetty.com",
    "documentationUrl": "https://evaljetty.com/llms.txt",
    "logoUrl": "https://evaljetty.com/assets/pelly-circle.png"
  },
  "entries": [
    {
      "identifier": "urn:air:evaljetty.com:api:data",
      "displayName": "Evals by Jetty data API",
      "type": "application/vnd.oai.openapi+json",
      "url": "https://evaljetty.com/openapi.json",
      "description": "Read-only JSON/CSV results and run lists for every eval on evaljetty.com, plus a Markdown twin of every page.",
      "tags": [
        "benchmarks",
        "evals",
        "datasets",
        "llm-evaluation"
      ],
      "representativeQueries": [
        "which LLM won the OpenRA Red Alert round robin",
        "do coding agents grade their own SVG drawings accurately",
        "which model plans a Micropolis city best",
        "download the pelican benchmark results as JSON"
      ]
    },
    {
      "identifier": "urn:air:evaljetty.com:skill:evaljetty-benchmarks",
      "displayName": "Evals by Jetty benchmarks skill",
      "type": "text/markdown",
      "url": "https://evaljetty.com/.well-known/agent-skills/evaljetty-benchmarks/SKILL.md",
      "description": "Agent skill: where the data lives and how to answer and cite questions about these benchmarks.",
      "tags": [
        "agent-skill",
        "benchmarks"
      ],
      "representativeQueries": [
        "how do I read the evaljetty benchmark results",
        "cite the evaljetty pelican leaderboard"
      ]
    },
    {
      "identifier": "urn:air:evaljetty.com:mcp:jetty",
      "displayName": "Jetty MCP server",
      "type": "application/mcp-server-card+json",
      "url": "https://evaljetty.com/.well-known/mcp/server-card.json",
      "description": "Jetty's hosted MCP server (Streamable HTTP, bearer Jetty API key): deploy, run, schedule and inspect runbooks like these evals.",
      "version": "1.1.0",
      "capabilities": [
        "list-collections",
        "get-collection",
        "list-tasks",
        "get-task",
        "create-task",
        "update-task",
        "get-trial-status",
        "activate-trial",
        "run-workflow",
        "list-trajectories",
        "get-trajectory",
        "get-stats",
        "add-label",
        "list-step-templates",
        "get-step-template",
        "check-secrets",
        "set-environment-vars",
        "list-routines",
        "get-routine",
        "create-routine",
        "update-routine",
        "delete-routine",
        "pause-routine",
        "resume-routine",
        "run-routine-now",
        "list-routine-runs"
      ],
      "tags": [
        "mcp",
        "runbooks",
        "evaluation"
      ],
      "representativeQueries": [
        "run my own agent eval on Jetty",
        "reproduce the pelican benchmark with my model",
        "schedule a Jetty runbook to run nightly"
      ]
    },
    {
      "identifier": "urn:air:evaljetty.com:api:jetty",
      "displayName": "Jetty REST API",
      "type": "application/vnd.oai.openapi+json",
      "url": "https://flows-api.jetty.io/openapi.json",
      "description": "The Jetty API these evals were run on: deploy runbooks as tasks, start runs, poll results. Bearer API key.",
      "tags": [
        "api",
        "runbooks"
      ],
      "representativeQueries": [
        "start a Jetty run from my own code",
        "poll a Jetty run and download its outputs"
      ]
    },
    {
      "identifier": "urn:air:evaljetty.com:doc:llms-txt",
      "displayName": "Evals by Jetty agent guide",
      "type": "text/plain",
      "url": "https://evaljetty.com/llms.txt",
      "description": "llms.txt index of every eval page, Markdown twin and data file, with the key results.",
      "tags": [
        "llms.txt",
        "documentation"
      ],
      "representativeQueries": [
        "what agent benchmarks does evaljetty publish",
        "evaljetty key results"
      ]
    }
  ]
}
