{
  "serverInfo": {
    "name": "robotsgate",
    "title": "RobotsGate",
    "version": "0.2.9"
  },
  "description": "Generate and validate robots.txt rules for AI crawlers, check which AI bots a live site allows, and look up a sourced AI crawler registry. Free: 20 checks and 60 generate/validate calls per minute; Pro $9/mo. RobotsGate is not affiliated with Meta or any other crawler or agent operator.",
  "homepage": "https://robotsgate.mike-tusa.workers.dev/",
  "contact": {
    "email": "digitalpromohub.support@gmail.com"
  },
  "transport": {
    "type": "streamable-http",
    "url": "https://robotsgate.mike-tusa.workers.dev/mcp"
  },
  "protocolVersions": [
    "2026-07-28",
    "2025-11-25",
    "2025-06-18",
    "2025-03-26",
    "2024-11-05"
  ],
  "capabilities": {
    "tools": {
      "listChanged": false
    }
  },
  "authentication": {
    "required": false,
    "schemes": [
      "bearer"
    ]
  },
  "instructions": "RobotsGate generates and validates robots.txt rules for AI crawlers and checks which AI crawlers a live site's robots.txt allows. Tools: list_crawlers (the sourced AI crawler registry), generate_robots (build a robots.txt from category/agent choices), validate_robots (RFC 9309 validation of robots.txt text, plus which AI crawlers it blocks for a path) and check_site (fetch a site's /robots.txt and analyse it). robots.txt is voluntary and is not access control; it only reaches crawlers that read and follow it. Free limits per client IP: 20 check_site calls and 60 generate_robots/validate_robots calls per 60 seconds, shared with the JSON API. RobotsGate Pro ($9/mo) raises them to 300 and 600: send \"Authorization: Bearer <RBTG license key>\" or \"X-License-Key: <key>\". A missing or invalid key simply uses the free limits. RobotsGate is not affiliated with Meta or any other crawler or agent operator.",
  "tools": [
    {
      "name": "list_crawlers",
      "title": "List the AI crawler registry",
      "description": "Return RobotsGate's AI crawler registry: each crawler's robots.txt token, operator, category (training, dataset, search, user_fetch or other), whether the operator says it respects robots.txt, and a link to the operator's documentation, plus unverified entries and notes. Read-only, no network access. Same data as GET /api/crawlers.",
      "inputSchema": {
        "type": "object",
        "properties": {
          "category": {
            "type": "string",
            "enum": [
              "training",
              "dataset",
              "search",
              "user_fetch",
              "other"
            ],
            "description": "Optional: only crawlers in this category."
          }
        },
        "additionalProperties": false
      },
      "outputSchema": {
        "type": "object",
        "properties": {
          "schema_version": {
            "type": "integer"
          },
          "generated": {
            "type": "string"
          },
          "categories": {
            "type": "object"
          },
          "crawlers": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "unverified": {
            "type": "array",
            "items": {
              "type": "object"
            }
          }
        },
        "required": [
          "categories",
          "crawlers"
        ]
      },
      "annotations": {
        "title": "List AI crawlers",
        "readOnlyHint": true,
        "destructiveHint": false,
        "idempotentHint": true,
        "openWorldHint": false
      }
    },
    {
      "name": "generate_robots",
      "title": "Generate a robots.txt for AI crawlers",
      "description": "Build a robots.txt from per-category choices (allow, block or omit for training, dataset, search, user_fetch or other), optional per-token overrides, a default policy for other bots, paths, sitemaps and custom rules. Returns the file text, a per-crawler summary and notes. Read-only: it only returns text and changes nothing. Same engine as POST /api/generate.",
      "inputSchema": {
        "type": "object",
        "properties": {
          "categories": {
            "type": "object",
            "description": "Action per crawler category. Omitted categories get no group of their own.",
            "properties": {
              "training": {
                "type": "string",
                "enum": [
                  "allow",
                  "block",
                  "omit"
                ]
              },
              "dataset": {
                "type": "string",
                "enum": [
                  "allow",
                  "block",
                  "omit"
                ]
              },
              "search": {
                "type": "string",
                "enum": [
                  "allow",
                  "block",
                  "omit"
                ]
              },
              "user_fetch": {
                "type": "string",
                "enum": [
                  "allow",
                  "block",
                  "omit"
                ]
              },
              "other": {
                "type": "string",
                "enum": [
                  "allow",
                  "block",
                  "omit"
                ]
              }
            },
            "additionalProperties": false
          },
          "agents": {
            "type": "object",
            "description": "Per-token overrides, e.g. {\"GPTBot\": \"block\"}. Tokens: letters, digits, \"_\", \".\" and \"-\", up to 64 characters (not \"*\"). At most 100.",
            "maxProperties": 100,
            "propertyNames": {
              "pattern": "^[A-Za-z0-9_.-]{1,64}$"
            },
            "additionalProperties": {
              "type": "string",
              "enum": [
                "allow",
                "block",
                "omit"
              ]
            }
          },
          "default": {
            "type": "string",
            "enum": [
              "allow",
              "block"
            ],
            "description": "Policy for \"User-agent: *\" (default allow)."
          },
          "disallow_paths": {
            "type": "array",
            "maxItems": 100,
            "items": {
              "type": "string",
              "minLength": 1,
              "maxLength": 512
            },
            "description": "Paths to disallow for \"*\" and for allowed AI groups. Each path starts with \"/\" or \"*\" and has no whitespace or \"#\". At most 100."
          },
          "allow_paths": {
            "type": "array",
            "maxItems": 100,
            "items": {
              "type": "string",
              "minLength": 1,
              "maxLength": 512
            },
            "description": "Paths to allow explicitly. Each path starts with \"/\" or \"*\" and has no whitespace or \"#\". At most 100."
          },
          "sitemaps": {
            "type": "array",
            "maxItems": 20,
            "items": {
              "type": "string",
              "maxLength": 2048
            },
            "description": "Absolute http(s) sitemap URLs. At most 20."
          },
          "sitemap": {
            "type": "string",
            "maxLength": 2048,
            "description": "A single sitemap URL (alternative to sitemaps)."
          },
          "custom_rules": {
            "type": "string",
            "maxLength": 20000,
            "description": "Extra robots.txt lines appended after validation (max 20000 bytes)."
          },
          "date": {
            "type": "string",
            "maxLength": 10,
            "pattern": "^\\d{4}-\\d{2}-\\d{2}$",
            "description": "Date for the header comment (YYYY-MM-DD). Default: today (UTC)."
          }
        },
        "additionalProperties": false
      },
      "outputSchema": {
        "type": "object",
        "properties": {
          "robots_txt": {
            "type": "string"
          },
          "summary": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "notes": {
            "type": "array",
            "items": {
              "type": "string"
            }
          },
          "personal_agents_note": {
            "type": "string"
          }
        },
        "required": [
          "robots_txt",
          "summary",
          "notes"
        ]
      },
      "annotations": {
        "title": "Generate robots.txt",
        "readOnlyHint": true,
        "destructiveHint": false,
        "idempotentHint": true,
        "openWorldHint": false
      }
    },
    {
      "name": "validate_robots",
      "title": "Validate robots.txt text",
      "description": "Validate robots.txt text against RFC 9309 and report errors, warnings and, for one URL path, which registry AI crawlers are allowed or blocked and by which rule. Read-only, no network access. Same engine as POST /api/validate. Up to 50,000 characters over MCP (the whole JSON-RPC request must also fit in 64 KB); for larger files use POST /api/validate or check_site.",
      "inputSchema": {
        "type": "object",
        "properties": {
          "robots_txt": {
            "type": "string",
            "maxLength": 50000,
            "description": "The robots.txt file content."
          },
          "path": {
            "type": "string",
            "maxLength": 2048,
            "description": "URL path to test, starting with \"/\" (default \"/\")."
          }
        },
        "required": [
          "robots_txt"
        ],
        "additionalProperties": false
      },
      "outputSchema": {
        "type": "object",
        "properties": {
          "valid": {
            "type": "boolean"
          },
          "errors": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "warnings": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "info": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "diagnostics": {
            "type": "object"
          },
          "stats": {
            "type": "object"
          },
          "path": {
            "type": "string"
          },
          "ai_crawlers": {
            "type": "array",
            "items": {
              "type": "object"
            }
          },
          "personal_agents": {
            "type": "object"
          }
        },
        "required": [
          "valid",
          "errors",
          "warnings",
          "ai_crawlers"
        ]
      },
      "annotations": {
        "title": "Validate robots.txt",
        "readOnlyHint": true,
        "destructiveHint": false,
        "idempotentHint": true,
        "openWorldHint": false
      }
    },
    {
      "name": "check_site",
      "title": "Check which AI crawlers a live site allows",
      "description": "Fetch a site's /robots.txt (http/https on the default port only; private, loopback and internal addresses are refused; up to 512,000 bytes; results cached for 5 minutes) and report, for one path, which registry AI crawlers are allowed or blocked, plus validation errors and warnings. Read-only: one GET of /robots.txt with a RobotsGate user-agent. Same engine as GET /api/check.",
      "inputSchema": {
        "type": "object",
        "properties": {
          "url": {
            "type": "string",
            "minLength": 1,
            "maxLength": 2048,
            "description": "Site URL or bare hostname, e.g. example.com. Only the scheme and host are used."
          },
          "path": {
            "type": "string",
            "maxLength": 2048,
            "description": "URL path to test, starting with \"/\" (default \"/\")."
          }
        },
        "required": [
          "url"
        ],
        "additionalProperties": false
      },
      "outputSchema": {
        "type": "object",
        "properties": {
          "url": {
            "type": "string"
          },
          "final_url": {
            "type": "string"
          },
          "http_status": {
            "type": "integer"
          },
          "redirects": {
            "type": "integer"
          },
          "fetched_at": {
            "type": "string"
          },
          "cache": {
            "type": "object"
          },
          "robots_found": {
            "type": "boolean"
          },
          "interpretation": {
            "type": "string",
            "enum": [
              "parsed",
              "unavailable",
              "unreachable"
            ]
          },
          "path": {
            "type": "string"
          },
          "ai_crawlers": {
            "anyOf": [
              {
                "type": "array",
                "items": {
                  "type": "object"
                }
              },
              {
                "type": "null"
              }
            ],
            "description": "Per-crawler results, or null when robots.txt was unreachable"
          },
          "personal_agents": {
            "anyOf": [
              {
                "type": "object"
              },
              {
                "type": "null"
              }
            ],
            "description": "Object, or null when robots.txt was unreachable"
          }
        },
        "required": [
          "url",
          "final_url",
          "http_status",
          "robots_found",
          "interpretation"
        ]
      },
      "annotations": {
        "title": "Check a live site",
        "readOnlyHint": true,
        "destructiveHint": false,
        "idempotentHint": true,
        "openWorldHint": true
      }
    }
  ],
  "resources": [],
  "prompts": []
}