{
  "name": "Iris",
  "id": "iris-eval",
  "description": "Stop shipping agents on vibes. Score every agent output for quality, safety, and cost.",
  "homepage": "https://iris-eval.com",
  "repository": "https://github.com/iris-eval/mcp-server",
  "npm": "@iris-eval/mcp-server",
  "version": "0.20.0",
  "license": "MIT",
  "transport": [
    "stdio",
    "http"
  ],
  "tools": [
    {
      "name": "log_trace",
      "description": "Store one agent execution (input, output, tool calls, spans, cost, tokens) and get the trace_id later calls key on."
    },
    {
      "name": "evaluate_output",
      "description": "Score an output with deterministic rules: a ship verdict with its basis, per-rule evidence, what was unjudged."
    },
    {
      "name": "get_traces",
      "description": "Query stored traces with filters, pagination and sorting; optionally with the dashboard summary."
    },
    {
      "name": "compare_runs",
      "description": "Did this change make the agent worse? Compares two runs: worse, better, equivalent, or too little evidence to tell."
    },
    {
      "name": "compare_traces",
      "description": "How reliably does the agent answer the same question? Per-case pass rates, flaky cases, and an interval that respects repeats."
    },
    {
      "name": "evaluate_runs",
      "description": "Re-score every trace in a run under the current rules into a new run, so a rules change is compared, not overwritten."
    },
    {
      "name": "list_rules",
      "description": "The rule inventory: every built-in rule with what it needs, its effective criticality and published accuracy, plus every deployed custom rule."
    },
    {
      "name": "deploy_rule",
      "description": "Deploy a custom rule that fires on every future evaluate_output call of its bundle — persisted, active immediately, audited."
    },
    {
      "name": "delete_rule",
      "description": "Remove a deployed custom rule — or, with enabled, disable or re-enable it without removing it — effective on the next evaluate_output call."
    },
    {
      "name": "delete_trace",
      "description": "Remove one stored trace by id; its spans go with it, and every evaluation linked to it keeps its verdict and loses its text."
    },
    {
      "name": "evaluate_with_llm_judge",
      "description": "Score an output with an LLM judge on your own key: a 0..1 score, a rationale, the spend."
    },
    {
      "name": "verify_citations",
      "description": "Is each citation in an output supported by its source? Extracts, fetches (opt-in, SSRF-guarded) and judges on your key."
    }
  ],
  "resources": [
    {
      "uri": "iris://capabilities",
      "name": "capabilities",
      "description": "What this server can judge: the rule roster with what each rule needs and its published accuracy, the judge state with the steps that enable it, the citation verifier posture, the dashboard address, the limits, and the tools, resources and prompts registered."
    },
    {
      "uri": "iris://proof",
      "name": "proof",
      "description": "The published accuracy of every measured built-in rule (the same numbers as https://iris-eval.com/proof): precision and recall on the proof corpus with 95% intervals, the confusion counts, positive predictive value at four prevalences, and the corpus version and labelling the numbers come from."
    },
    {
      "uri": "iris://dashboard/summary",
      "name": "dashboard-summary",
      "description": "Dashboard summary with key metrics and trends for the last hour"
    },
    {
      "uri": "iris://audit",
      "name": "audit",
      "description": "The newest 100 audit entries, newest first: every rule deploy, delete, toggle and update, and every trace deletion, with its time and what it touched"
    },
    {
      "uri": "iris://traces/{trace_id}",
      "name": "trace-detail",
      "description": "One stored trace with its spans and every evaluation linked to it",
      "template": true
    },
    {
      "uri": "iris://evaluations/{id}",
      "name": "evaluation-detail",
      "description": "One stored evaluation, in the same shape evaluate_output returned it: verdict, coverage, provenance and every rule result with its evidence",
      "template": true
    }
  ],
  "prompts": [
    {
      "name": "evaluate-my-agent",
      "description": "A walk through logging an agent run, evaluating it and reading the verdict with Iris, in plain words."
    }
  ],
  "install": {
    "claude_code": {
      "command": "claude mcp add iris-eval -- npx -y @iris-eval/mcp-server@0.20.0 --dashboard"
    },
    "claude_desktop": {
      "mcpServers": {
        "iris-eval": {
          "command": "npx",
          "args": [
            "-y",
            "@iris-eval/mcp-server@0.20.0",
            "--dashboard"
          ]
        }
      }
    },
    "cursor": {
      "mcpServers": {
        "iris-eval": {
          "command": "npx",
          "args": [
            "-y",
            "@iris-eval/mcp-server@0.20.0",
            "--dashboard"
          ]
        }
      }
    },
    "docker": {
      "command": "docker run -p 3000:3000 -p 6920:6920 -v iris-data:/data -e IRIS_API_KEY=<your key> ghcr.io/iris-eval/mcp-server:v0.20.0"
    }
  },
  "links": {
    "capabilities": "https://iris-eval.com/capabilities",
    "proof": "https://iris-eval.com/proof",
    "llms": "https://iris-eval.com/llms.txt",
    "security": "https://iris-eval.com/.well-known/security.txt"
  },
  "discovery": "Add the config block, restart your client, and every session lists Iris's tools on connect. Iris never intercepts: it runs when your agent calls one of its tools, when a host hook or `iris-eval ingest` hands it a trace, or when you POST one to its HTTP API.",
  "dataResidency": "Nothing leaves your machine unless you turn one of these on: an OpenTelemetry endpoint (IRIS_OTEL_ENDPOINT), which exports traces to the collector you name; the LLM judge with your own key, which sends the text it judges to that provider, and whose citation check fetches the pages an output cites; or a webhook, which posts ids, the verdict and rule names, never the text, to the address you set."
}
