{
  "$schema": "https://modelcontextprotocol.io/schema/server-card.json",
  "name": "ContextRail AI MCP Server",
  "version": "1.2.0",
  "description": "Official Model Context Protocol (MCP) server for ContextRail AI: semantic context routing, cross-session agent memory, and sub-50ms RAG token compression optimized for Anthropic Claude (Claude Sonnet and Haiku).",
  "vendor": {
    "name": "ContextRail AI",
    "url": "https://contextrail.cloud/",
    "contactEmail": "admin@contextrail.cloud"
  },
  "websiteUrl": "https://contextrail.cloud/",
  "documentationUrl": "https://contextrail.cloud/#api",
  "support": {
    "url": "https://contextrail.cloud/#faq",
    "email": "admin@contextrail.cloud"
  },
  "license": "proprietary",
  "homepage": "https://contextrail.cloud/",
  "endpoint": "https://api.contextrail.cloud/mcp",
  "transport": "http",
  "transportDocs": "https://modelcontextprotocol.io/specification",
  "securitySchemes": [
    {
      "type": "http",
      "scheme": "bearer",
      "description": "ContextRail AI API key issued at https://contextrail.cloud/#pricing"
    }
  ],
  "security": [
    {
      "bearer": []
    }
  ],
  "capabilities": {
    "tools": [
      {
        "name": "route_context",
        "description": "Dynamically prunes irrelevant context and routes high-salience tokens into Claude's context window, maximizing attention focus and prompt cache alignment.",
        "inputSchema": {
          "type": "object",
          "properties": {
            "query": {
              "type": "string",
              "description": "User turn to route context for."
            },
            "candidate_documents": {
              "type": "array",
              "items": {
                "type": "string"
              },
              "description": "RAG candidate chunks."
            },
            "max_tokens": {
              "type": "integer",
              "default": 4096,
              "description": "Token budget for the routed window."
            }
          },
          "required": [
            "query"
          ]
        }
      },
      {
        "name": "compress_rag_context",
        "description": "Executes low-latency semantic token compression on multi-document RAG retrieval chunks, reducing token payloads up to 78% while preserving 99.4% needle recall.",
        "inputSchema": {
          "type": "object",
          "properties": {
            "documents": {
              "type": "array",
              "items": {
                "type": "string"
              },
              "description": "RAG chunks to compress."
            },
            "target_compression": {
              "type": "number",
              "default": 0.78,
              "description": "Target compression ratio between 0 and 1."
            }
          },
          "required": [
            "documents"
          ]
        }
      },
      {
        "name": "retrieve_agent_memory",
        "description": "Queries ContextRail persistent episodic and semantic graph memory across agent sessions for autonomous Claude workflows.",
        "inputSchema": {
          "type": "object",
          "properties": {
            "query": {
              "type": "string",
              "description": "Semantic query over episodic memory."
            },
            "top_k": {
              "type": "integer",
              "default": 8
            }
          },
          "required": [
            "query"
          ]
        }
      },
      {
        "name": "store_agent_memory",
        "description": "Persists structured agent states, decisions, tool call outcomes, and key entities into long-term vector-graph hybrid storage.",
        "inputSchema": {
          "type": "object",
          "properties": {
            "session_id": {
              "type": "string",
              "description": "Agent session identifier."
            },
            "state": {
              "type": "object",
              "description": "Serializable agent state, decisions and tool outcomes."
            }
          },
          "required": [
            "session_id"
          ]
        }
      },
      {
        "name": "get_context_metrics",
        "description": "Retrieves real-time telemetry on token savings, Claude Prompt Cache hit rates, context saturation warnings, and millisecond routing latency.",
        "inputSchema": {
          "type": "object",
          "properties": {
            "timeframe": {
              "type": "string",
              "enum": [
                "1h",
                "24h",
                "7d",
                "30d"
              ],
              "default": "24h"
            }
          },
          "required": [
            "timeframe"
          ]
        }
      }
    ],
    "resources": [
      {
        "uri": "contextrail://active-session/routing-graph",
        "name": "Active Routing Graph",
        "description": "Live DAG representation of injected context branches, token compression ratios, and cache boundary indicators."
      },
      {
        "uri": "contextrail://metrics/token-efficiency",
        "name": "Token Efficiency Stream",
        "description": "Real-time metrics stream tracking cumulative tokens saved, TTFT speedup, and Claude API cost reduction."
      }
    ],
    "prompts": [
      {
        "name": "route_context_for_prompt",
        "description": "Builds a minimal high-salience context window for a user turn and returns cache_control breakpoints.",
        "arguments": [
          {
            "name": "query",
            "description": "User turn to route context for.",
            "required": true
          },
          {
            "name": "candidate_documents",
            "description": "RAG candidate chunks to rank and prune.",
            "required": true
          }
        ]
      }
    ]
  }
}
