{
  "name": "robots-check",
  "summary": "Fetches robots.txt and answers whether a user agent may crawl a URL, showing the group, the winning rule and every rule that competed with it.",
  "rationale": "It needs the network, and the part an agent gets wrong on its own is precedence: RFC 9309 resolves conflicts by longest match, not first match. One fetch answers up to 50 paths.",
  "family": "feeds",
  "price_usd": "0.002",
  "free": false,
  "timeout_ms": 8000,
  "max_timeout_seconds": 29,
  "input_schema": {
    "$schema": "https://json-schema.org/draft/2020-12/schema",
    "type": "object",
    "properties": {
      "url": {
        "type": "string",
        "maxLength": 2048,
        "format": "uri",
        "description": "Absolute http(s) URL to check, up to 2048 characters; robots.txt is fetched from the root of its origin and the verdict is for its path and query."
      },
      "user_agent": {
        "default": "*",
        "description": "Crawler product token or full User-Agent string, printable ASCII up to 200 characters; the text before the first slash or space picks the group, and the default is *.",
        "type": "string",
        "minLength": 1,
        "maxLength": 200,
        "pattern": "^[\\x20-\\x7e]+$"
      },
      "include_sitemaps": {
        "default": true,
        "description": "Whether to return the Sitemap URLs listed in robots.txt; false returns null for sitemaps and its counters.",
        "type": "boolean"
      },
      "additional_paths": {
        "description": "Up to 50 more root-relative paths or absolute URLs on the same origin as url, each answered from the same robots.txt fetch in additional_results.",
        "maxItems": 50,
        "type": "array",
        "items": {
          "type": "string",
          "minLength": 1,
          "maxLength": 2048
        }
      }
    },
    "required": [
      "url"
    ],
    "additionalProperties": false
  },
  "input_example": {
    "url": "https://grist.tools/v1/whois",
    "user_agent": "GPTBot",
    "include_sitemaps": true,
    "additional_paths": [
      "/docs/whois",
      "/v1/whois/schema"
    ]
  },
  "output_example": {
    "url": "https://grist.tools/v1/whois",
    "source_url": "https://grist.tools/robots.txt",
    "robots_url": "https://grist.tools/robots.txt",
    "robots_status": "found",
    "robots_http_status": 200,
    "user_agent": "GPTBot",
    "user_agent_token": "GPTBot",
    "allowed": false,
    "decision_reason": "rule",
    "matched_rule": {
      "type": "disallow",
      "path": "/v1/",
      "line": 3,
      "octet_length": 4
    },
    "matched_group": {
      "user_agents": [
        "*"
      ],
      "is_wildcard": true,
      "line": 1
    },
    "competing_rules": [
      {
        "type": "disallow",
        "path": "/v1/",
        "line": 3,
        "octet_length": 4
      },
      {
        "type": "allow",
        "path": "/",
        "line": 2,
        "octet_length": 1
      }
    ],
    "competing_rules_total": 2,
    "competing_rules_truncated": false,
    "crawl_delay": null,
    "sitemaps": [
      "https://grist.tools/sitemap.xml"
    ],
    "sitemaps_total": 1,
    "sitemaps_truncated": false,
    "groups_found": 1,
    "rules_parsed": 3,
    "unparsed_lines": 0,
    "robots_size_bytes": 141,
    "robots_truncated": false,
    "additional_results": [
      {
        "path": "/docs/whois",
        "allowed": true,
        "decision_reason": "rule",
        "matched_rule": {
          "type": "allow",
          "path": "/",
          "line": 2,
          "octet_length": 1
        }
      },
      {
        "path": "/v1/whois/schema",
        "allowed": true,
        "decision_reason": "rule",
        "matched_rule": {
          "type": "allow",
          "path": "/v1/*/schema",
          "line": 4,
          "octet_length": 12
        }
      }
    ],
    "fetched_at": "2026-09-01T12:00:00.000Z",
    "checked_at": "2026-09-01T12:00:00.000Z"
  },
  "errors": [
    "invalid_input",
    "blocked_target",
    "upstream_timeout",
    "internal"
  ]
}