{
  "$comment": "The single source of truth for which crawlers this product knows about. robots.txt and the package's generator are built from this file. Edit here, run tools/gen-robots.py, never edit robots.txt by hand. inGeneratedRobots marks the agents the openaeo-audit package names in a customer's robots.txt; the rest are listed here for this site's own file and for the watcher to track.",
  "last_reviewed": "2026-08-06",
  "review_interval_days": 180,
  "private_paths": [
    "/admin/",
    "/wp-admin/",
    "/login/",
    "/cart/",
    "/checkout/",
    "/account/",
    "/api/",
    "/staging/",
    "/tmp/",
    "/internal/"
  ],
  "crawlers": [
    {
      "agent": "Claude-SearchBot",
      "key": "claude-searchbot",
      "operator": "Anthropic",
      "purpose": "search",
      "description": "Anthropic's search crawler for Claude with web browsing.",
      "docs": "https://support.anthropic.com/en/articles/8896518",
      "tier": "tier1",
      "surface": "Claude",
      "rangeFeed": "https://claude.com/crawling/bots.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 4
    },
    {
      "agent": "Claude-User",
      "key": "claude-user",
      "operator": "Anthropic",
      "purpose": "user_fetch",
      "description": "Fetches a URL when a user explicitly asks Claude about it.",
      "docs": "https://support.anthropic.com/en/articles/8896518",
      "tier": "listed",
      "surface": null,
      "rangeFeed": "https://claude.com/crawling/bots.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 5
    },
    {
      "agent": "ClaudeBot",
      "key": "claudebot",
      "operator": "Anthropic",
      "purpose": "training",
      "description": "Anthropic's training crawler.",
      "docs": "https://support.anthropic.com/en/articles/8896518",
      "tier": "tier1",
      "surface": "Claude",
      "rangeFeed": "https://claude.com/crawling/bots.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 3
    },
    {
      "agent": "Applebot",
      "key": "applebot",
      "operator": "Apple",
      "purpose": "search",
      "description": "Apple's search crawler. Powers Spotlight, Siri Suggestions, Safari smart search.",
      "docs": "https://support.apple.com/en-us/119829",
      "tier": "tier1",
      "surface": "Siri",
      "rangeFeed": "https://search.developer.apple.com/applebot.json",
      "rdnsSuffixes": [
        ".applebot.apple.com",
        ".apple.com"
      ],
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 9
    },
    {
      "agent": "Applebot-Extended",
      "key": "applebot-extended",
      "operator": "Apple",
      "purpose": "training",
      "description": "Controls whether content is used for Apple Intelligence training.",
      "docs": "https://support.apple.com/en-us/119829",
      "tier": "policy-token",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "note": "robots.txt token only; no crawler identifies as it",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "Bytespider",
      "key": "bytespider",
      "operator": "ByteDance",
      "purpose": "training",
      "description": "ByteDance crawler. Feeds Doubao and other ByteDance AI products. Material in non-Western mobile-first markets.",
      "docs": "https://www.bytedance.com/",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "cohere-ai",
      "key": "cohere-ai",
      "operator": "Cohere",
      "purpose": "training",
      "description": "Cohere's training crawler.",
      "docs": "https://docs.cohere.com/",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "CCBot",
      "key": "ccbot",
      "operator": "Common Crawl",
      "purpose": "other",
      "description": "Training. Common Crawl. Feeds many open LLM training datasets used by Anthropic, Mistral, Meta research and others.",
      "docs": "https://commoncrawl.org/ccbot",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "Google-Extended",
      "key": "google-extended",
      "operator": "Google",
      "purpose": "training",
      "description": "Controls whether content is used for training Bard/Gemini and Vertex AI.",
      "docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
      "tier": "policy-token",
      "surface": "Gemini",
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "note": "robots.txt token only; no crawler identifies as it",
      "inGeneratedRobots": true,
      "generatorOrder": 7
    },
    {
      "agent": "Googlebot",
      "key": "googlebot",
      "operator": "Google",
      "purpose": "search",
      "description": "Standard Google search crawler. Powers Google AI Overviews and Gemini search responses.",
      "docs": "https://developers.google.com/search/docs/crawling-indexing/google-special-case-crawlers",
      "tier": "tier1",
      "surface": "AI Overviews",
      "rangeFeed": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
      "rdnsSuffixes": [
        ".googlebot.com",
        ".google.com",
        ".googleusercontent.com"
      ],
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "FacebookExternalHit",
      "key": "facebookexternalhit",
      "operator": "Meta",
      "purpose": "other",
      "description": "Link_preview. Generates link previews for Facebook and Instagram posts.",
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "Meta-ExternalAgent",
      "key": "meta-externalagent",
      "operator": "Meta",
      "purpose": "training",
      "description": "Meta's general-purpose crawler for AI training and tooling.",
      "docs": "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    },
    {
      "agent": "Bingbot",
      "key": "bingbot",
      "operator": "Microsoft",
      "purpose": "search",
      "description": "Microsoft Bing's search crawler. Powers Copilot (formerly Bing Chat) responses.",
      "docs": "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
      "tier": "tier1",
      "surface": "Copilot",
      "rangeFeed": "https://www.bing.com/toolbox/bingbot.json",
      "rdnsSuffixes": [
        ".search.msn.com"
      ],
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 8
    },
    {
      "agent": "ChatGPT-User",
      "key": "chatgpt-user",
      "operator": "OpenAI",
      "purpose": "user_fetch",
      "description": "Fetches a URL when a user explicitly asks ChatGPT about it.",
      "docs": "https://platform.openai.com/docs/bots",
      "tier": "listed",
      "surface": null,
      "rangeFeed": "https://openai.com/chatgpt-user.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 2
    },
    {
      "agent": "GPTBot",
      "key": "gptbot",
      "operator": "OpenAI",
      "purpose": "training",
      "description": "OpenAI's training crawler. Allowing makes content eligible for future model training.",
      "docs": "https://platform.openai.com/docs/bots",
      "tier": "tier1",
      "surface": "ChatGPT",
      "rangeFeed": "https://openai.com/gptbot.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 0
    },
    {
      "agent": "OAI-SearchBot",
      "key": "oai-searchbot",
      "operator": "OpenAI",
      "purpose": "search",
      "description": "Powers ChatGPT search results (web browsing). Allow for AI-search visibility.",
      "docs": "https://platform.openai.com/docs/bots",
      "tier": "tier1",
      "surface": "ChatGPT",
      "rangeFeed": "https://openai.com/searchbot.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 1
    },
    {
      "agent": "PerplexityBot",
      "key": "perplexitybot",
      "operator": "Perplexity",
      "purpose": "search",
      "description": "Perplexity's web crawler for search results.",
      "docs": "https://docs.perplexity.ai/guides/bots",
      "tier": "tier1",
      "surface": "Perplexity",
      "rangeFeed": "https://www.perplexity.ai/perplexitybot.json",
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": true,
      "generatorOrder": 6
    },
    {
      "agent": "YouBot",
      "key": "youbot",
      "operator": "You.com",
      "purpose": "other",
      "description": "Search. You.com AI search crawler.",
      "docs": "https://about.you.com/youbot/",
      "tier": "listed",
      "surface": null,
      "rangeFeed": null,
      "rdnsSuffixes": null,
      "robotsPolicy": "allow",
      "inGeneratedRobots": false,
      "generatorOrder": null
    }
  ]
}
