{
  "project": "Verified Bots Directory",
  "generated_at": "2026-08-23T13:31:38.100Z",
  "count": 30,
  "bots": [
    {
      "id": "ai2bot",
      "name": "AI2Bot",
      "operator": {
        "name": "Allen Institute for AI",
        "url": "https://allenai.org"
      },
      "description": "The Allen Institute for AI's crawler, used to build open training datasets such as Dolma. AI2 publishes no IP ranges, ASN, or reverse-DNS pattern.",
      "docs": [
        "https://allenai.org/crawler"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "AI2Bot"
        ],
        "instances": [
          "Mozilla/5.0 (compatible) AI2Bot (+https://www.allenai.org/crawler)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "AI2Bot"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "amazonbot",
      "name": "Amazonbot",
      "operator": {
        "name": "Amazon",
        "url": "https://www.amazon.com"
      },
      "description": "Amazon's web crawler used to improve Amazon's products and services, including training AI models. Amazon publishes a human-readable IP list but no machine-parseable feed in a format this project's schema supports.",
      "docs": [
        "https://developer.amazon.com/en/amazonbot"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Amazonbot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; Amazonbot/0.1) Chrome/W.X.Y.Z Safari/537.36"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Amazonbot"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "bytespider",
      "name": "Bytespider",
      "operator": {
        "name": "ByteDance",
        "url": "https://www.bytedance.com"
      },
      "description": "ByteDance's web crawler, used to gather training data for its AI models. ByteDance publishes no IP ranges, ASN, or reverse-DNS pattern for it.",
      "docs": [
        "https://zhanzhang.toutiao.com/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Bytespider"
        ],
        "instances": [
          "Mozilla/5.0 (iPhone; CPU iPhone OS 11_0 like Mac OS X) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/54.0.3478.1649 Mobile Safari/537.36; Bytespider"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Bytespider"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "ccbot",
      "name": "CCBot",
      "operator": {
        "name": "Common Crawl Foundation",
        "url": "https://commoncrawl.org"
      },
      "description": "Common Crawl's crawler that builds the freely available Common Crawl web archive, widely reused as AI training data by third parties.",
      "docs": [
        "https://commoncrawl.org/ccbot"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "CCBot/\\d"
        ],
        "instances": [
          "CCBot/2.0 (https://commoncrawl.org/faq/)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "CCBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://index.commoncrawl.org/ccbot.json",
          "format": "prefixes"
        },
        {
          "type": "rdns",
          "masks": [
            "*.crawl.commoncrawl.org"
          ]
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "18.97.14.80/29",
        "18.97.14.88/30",
        "18.97.9.168/29",
        "2600:1f28:365:8000::/56",
        "3.41.188.32/29",
        "98.85.178.216/32"
      ]
    },
    {
      "id": "claude-searchbot",
      "name": "Claude-SearchBot",
      "operator": {
        "name": "Anthropic",
        "url": "https://www.anthropic.com"
      },
      "description": "Anthropic's crawler that indexes pages to improve search-style answer quality in Claude products, distinct from ClaudeBot's training-data crawl.",
      "docs": [
        "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Claude-SearchBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 (compatible; Claude-SearchBot/1.0; +Claude-SearchBot@anthropic.com)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Claude-SearchBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://claude.com/crawling/bots.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "136.107.176.208/32",
        "16.58.26.69/32",
        "18.225.238.228/32",
        "18.227.226.255/32",
        "20.102.46.224/28",
        "20.64.57.208/28",
        "216.73.216.0/22",
        "34.11.34.31/32",
        "34.150.241.79/32",
        "34.162.191.81/32",
        "34.162.230.222/32",
        "34.162.244.71/32",
        "34.182.140.95/32",
        "34.182.161.143/32",
        "34.182.218.27/32",
        "34.182.220.85/32",
        "34.182.222.37/32",
        "34.182.225.167/32",
        "34.182.226.151/32",
        "34.182.226.221/32",
        "34.186.108.163/32",
        "34.85.172.162/32",
        "35.221.29.174/32",
        "35.245.175.129/32",
        "35.245.89.239/32",
        "40.124.101.48/28"
      ]
    },
    {
      "id": "claudebot",
      "name": "ClaudeBot",
      "operator": {
        "name": "Anthropic",
        "url": "https://www.anthropic.com"
      },
      "description": "Anthropic's web crawler collecting publicly available web data for training Claude models. Anthropic publishes its crawler source IPs as a machine-readable feed.",
      "docs": [
        "https://support.anthropic.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "ClaudeBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; ClaudeBot/1.0; +claudebot@anthropic.com)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "ClaudeBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://claude.com/crawling/bots.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "136.107.176.208/32",
        "16.58.26.69/32",
        "18.225.238.228/32",
        "18.227.226.255/32",
        "20.102.46.224/28",
        "20.64.57.208/28",
        "216.73.216.0/22",
        "34.11.34.31/32",
        "34.150.241.79/32",
        "34.162.191.81/32",
        "34.162.230.222/32",
        "34.162.244.71/32",
        "34.182.140.95/32",
        "34.182.161.143/32",
        "34.182.218.27/32",
        "34.182.220.85/32",
        "34.182.222.37/32",
        "34.182.225.167/32",
        "34.182.226.151/32",
        "34.182.226.221/32",
        "34.186.108.163/32",
        "34.85.172.162/32",
        "35.221.29.174/32",
        "35.245.175.129/32",
        "35.245.89.239/32",
        "40.124.101.48/28"
      ]
    },
    {
      "id": "cloudflare-ai-search",
      "name": "Cloudflare AI Search",
      "operator": {
        "name": "Cloudflare",
        "url": "https://www.cloudflare.com"
      },
      "description": "The crawler behind Cloudflare AI Search, which indexes website content so it can be searched. Cloudflare documents that it only crawls a website the customer owns — the domain must exist in the same Cloudflare account and be selected as an AI Search data source.",
      "docs": [
        "https://developers.cloudflare.com/fundamentals/reference/cloudflare-site-crawling/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Cloudflare-AI-Search"
        ],
        "instances": [
          "Cloudflare-AI-Search"
        ]
      },
      "behavior": {
        "respects_robots_txt": false
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-30T16:13:28.285Z",
      "resolved_cidrs": []
    },
    {
      "id": "cloudflare-browser-run-crawler",
      "name": "Cloudflare Browser Run Crawler",
      "operator": {
        "name": "Cloudflare",
        "url": "https://www.cloudflare.com"
      },
      "description": "The crawler behind the /crawl endpoint of Cloudflare's Browser Run (Browser Rendering) developer product, which crawls third-party websites on behalf of Cloudflare customers building applications on the platform. Its user agent is not configurable, and every request is signed with Web Bot Auth HTTP message signatures that site owners can verify against Cloudflare's published key directory.",
      "docs": [
        "https://developers.cloudflare.com/fundamentals/reference/cloudflare-site-crawling/",
        "https://developers.cloudflare.com/browser-rendering/reference/automatic-request-headers/",
        "https://developers.cloudflare.com/browser-rendering/reference/robots-txt/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "CloudflareBrowserRenderingCrawler/[\\d.]+"
        ],
        "instances": [
          "CloudflareBrowserRenderingCrawler/1.0"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "CloudflareBrowserRenderingCrawler"
      },
      "verification": [
        {
          "type": "web_bot_auth",
          "signature_agent_url": "https://web-bot-auth.cloudflare-browser-rendering-085.workers.dev/.well-known/http-message-signatures-directory"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-30T16:13:28.285Z",
      "resolved_cidrs": []
    },
    {
      "id": "diffbot",
      "name": "Diffbot",
      "operator": {
        "name": "Diffbot",
        "url": "https://www.diffbot.com"
      },
      "description": "Diffbot's general-purpose web crawler, used to build its Knowledge Graph and power its structured-extraction APIs. Diffbot documents robots.txt behavior but publishes no IP list, ASN, or reverse-DNS pattern.",
      "docs": [
        "https://docs.diffbot.com/docs/does-crawl-respect-robotstxt"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Diffbot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 (Windows; U; Windows NT 5.1; en-US; rv:1.9.1.2) Gecko/20090729 Firefox/3.5.2 (.NET CLR 3.5.30729; Diffbot/0.1; +http://www.diffbot.com)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Diffbot"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "firecrawl-agent",
      "name": "FirecrawlAgent",
      "operator": {
        "name": "Firecrawl",
        "url": "https://firecrawl.dev"
      },
      "description": "The fetcher behind Firecrawl's scraping and crawling API, which retrieves web pages on behalf of the developers and AI agents calling that API. Firecrawl's crawl endpoint honours robots.txt by default; ignoring it is an enterprise option that must be enabled by the operator. Firecrawl documents that it does not use a fixed set of outbound IP addresses.",
      "docs": [
        "https://docs.firecrawl.dev/advanced-scraping-guide"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "FirecrawlAgent"
        ],
        "instances": [
          "Mozilla/5.0 (compatible; FirecrawlAgent; +https://firecrawl.dev/)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "FirecrawlAgent"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-08-23T13:20:46.873Z",
      "resolved_cidrs": []
    },
    {
      "id": "google-cloudvertexbot",
      "name": "Google-CloudVertexBot",
      "operator": {
        "name": "Google",
        "url": "https://cloud.google.com"
      },
      "description": "Google Cloud crawler that fetches sites on the site owners' request when building Vertex AI Agents. Crawling preferences addressed to its user agent have no effect on Google Search or other Google products.",
      "docs": [
        "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Google-CloudVertexBot"
        ],
        "instances": [
          "Google-CloudVertexBot"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Google-CloudVertexBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://developers.google.com/static/crawling/ipranges/common-crawlers.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-09T03:01:47.690Z",
      "resolved_cidrs": [
        "192.178.4.0/27",
        "192.178.4.128/27",
        "192.178.4.160/27",
        "192.178.4.192/27",
        "192.178.4.224/27",
        "192.178.4.32/27",
        "192.178.4.64/27",
        "192.178.4.96/27",
        "192.178.5.0/27",
        "192.178.6.0/27",
        "192.178.6.128/27",
        "192.178.6.160/27",
        "192.178.6.192/27",
        "192.178.6.224/27",
        "192.178.6.32/27",
        "192.178.6.64/27",
        "192.178.6.96/27",
        "192.178.7.0/27",
        "192.178.7.128/27",
        "192.178.7.160/27",
        "192.178.7.192/27",
        "192.178.7.224/27",
        "192.178.7.32/27",
        "192.178.7.64/27",
        "192.178.7.96/27",
        "2001:4860:4801:10::/64",
        "2001:4860:4801:11::/64",
        "2001:4860:4801:12::/64",
        "2001:4860:4801:13::/64",
        "2001:4860:4801:14::/64",
        "2001:4860:4801:15::/64",
        "2001:4860:4801:16::/64",
        "2001:4860:4801:17::/64",
        "2001:4860:4801:18::/64",
        "2001:4860:4801:19::/64",
        "2001:4860:4801:1a::/64",
        "2001:4860:4801:1b::/64",
        "2001:4860:4801:1c::/64",
        "2001:4860:4801:1d::/64",
        "2001:4860:4801:1e::/64",
        "2001:4860:4801:1f::/64",
        "2001:4860:4801:20::/64",
        "2001:4860:4801:21::/64",
        "2001:4860:4801:22::/64",
        "2001:4860:4801:23::/64",
        "2001:4860:4801:24::/64",
        "2001:4860:4801:25::/64",
        "2001:4860:4801:26::/64",
        "2001:4860:4801:27::/64",
        "2001:4860:4801:28::/64",
        "2001:4860:4801:29::/64",
        "2001:4860:4801:2::/64",
        "2001:4860:4801:2a::/64",
        "2001:4860:4801:2b::/64",
        "2001:4860:4801:2c::/64",
        "2001:4860:4801:2d::/64",
        "2001:4860:4801:2e::/64",
        "2001:4860:4801:2f::/64",
        "2001:4860:4801:30::/64",
        "2001:4860:4801:31::/64",
        "2001:4860:4801:32::/64",
        "2001:4860:4801:33::/64",
        "2001:4860:4801:34::/64",
        "2001:4860:4801:35::/64",
        "2001:4860:4801:36::/64",
        "2001:4860:4801:37::/64",
        "2001:4860:4801:38::/64",
        "2001:4860:4801:39::/64",
        "2001:4860:4801:3a::/64",
        "2001:4860:4801:3b::/64",
        "2001:4860:4801:3c::/64",
        "2001:4860:4801:3d::/64",
        "2001:4860:4801:3e::/64",
        "2001:4860:4801:3f::/64",
        "2001:4860:4801:40::/64",
        "2001:4860:4801:41::/64",
        "2001:4860:4801:42::/64",
        "2001:4860:4801:44::/64",
        "2001:4860:4801:45::/64",
        "2001:4860:4801:46::/64",
        "2001:4860:4801:47::/64",
        "2001:4860:4801:48::/64",
        "2001:4860:4801:49::/64",
        "2001:4860:4801:4a::/64",
        "2001:4860:4801:4b::/64",
        "2001:4860:4801:4c::/64",
        "2001:4860:4801:4d::/64",
        "2001:4860:4801:4e::/64",
        "2001:4860:4801:50::/64",
        "2001:4860:4801:51::/64",
        "2001:4860:4801:52::/64",
        "2001:4860:4801:53::/64",
        "2001:4860:4801:54::/64",
        "2001:4860:4801:55::/64",
        "2001:4860:4801:56::/64",
        "2001:4860:4801:57::/64",
        "2001:4860:4801:58::/64",
        "2001:4860:4801:59::/64",
        "2001:4860:4801:60::/64",
        "2001:4860:4801:61::/64",
        "2001:4860:4801:62::/64",
        "2001:4860:4801:63::/64",
        "2001:4860:4801:64::/64",
        "2001:4860:4801:65::/64",
        "2001:4860:4801:66::/64",
        "2001:4860:4801:67::/64",
        "2001:4860:4801:68::/64",
        "2001:4860:4801:69::/64",
        "2001:4860:4801:6a::/64",
        "2001:4860:4801:6b::/64",
        "2001:4860:4801:6c::/64",
        "2001:4860:4801:6d::/64",
        "2001:4860:4801:6e::/64",
        "2001:4860:4801:6f::/64",
        "2001:4860:4801:70::/64",
        "2001:4860:4801:71::/64",
        "2001:4860:4801:72::/64",
        "2001:4860:4801:73::/64",
        "2001:4860:4801:74::/64",
        "2001:4860:4801:75::/64",
        "2001:4860:4801:76::/64",
        "2001:4860:4801:77::/64",
        "2001:4860:4801:78::/64",
        "2001:4860:4801:79::/64",
        "2001:4860:4801:7a::/64",
        "2001:4860:4801:7b::/64",
        "2001:4860:4801:7c::/64",
        "2001:4860:4801:7d::/64",
        "2001:4860:4801:7e::/64",
        "2001:4860:4801:7f::/64",
        "2001:4860:4801:80::/64",
        "2001:4860:4801:81::/64",
        "2001:4860:4801:82::/64",
        "2001:4860:4801:83::/64",
        "2001:4860:4801:84::/64",
        "2001:4860:4801:85::/64",
        "2001:4860:4801:86::/64",
        "2001:4860:4801:87::/64",
        "2001:4860:4801:88::/64",
        "2001:4860:4801:90::/64",
        "2001:4860:4801:91::/64",
        "2001:4860:4801:92::/64",
        "2001:4860:4801:93::/64",
        "2001:4860:4801:94::/64",
        "2001:4860:4801:95::/64",
        "2001:4860:4801:96::/64",
        "2001:4860:4801:97::/64",
        "2001:4860:4801:a0::/64",
        "2001:4860:4801:a1::/64",
        "2001:4860:4801:a2::/64",
        "2001:4860:4801:a3::/64",
        "2001:4860:4801:a4::/64",
        "2001:4860:4801:a5::/64",
        "2001:4860:4801:a6::/64",
        "2001:4860:4801:a7::/64",
        "2001:4860:4801:a8::/64",
        "2001:4860:4801:a9::/64",
        "2001:4860:4801:aa::/64",
        "2001:4860:4801:ab::/64",
        "2001:4860:4801:ac::/64",
        "2001:4860:4801:ad::/64",
        "2001:4860:4801:ae::/64",
        "2001:4860:4801:b0::/64",
        "2001:4860:4801:b1::/64",
        "2001:4860:4801:b2::/64",
        "2001:4860:4801:b3::/64",
        "2001:4860:4801:b4::/64",
        "2001:4860:4801:b5::/64",
        "2001:4860:4801:b6::/64",
        "2001:4860:4801:c::/64",
        "2001:4860:4801:f::/64",
        "34.100.182.96/28",
        "34.101.50.144/28",
        "34.118.254.0/28",
        "34.118.66.0/28",
        "34.126.178.96/28",
        "34.146.150.144/28",
        "34.147.110.144/28",
        "34.151.74.144/28",
        "34.152.50.64/28",
        "34.154.114.144/28",
        "34.155.98.32/28",
        "34.165.18.176/28",
        "34.175.160.64/28",
        "34.176.130.16/28",
        "34.22.85.0/27",
        "34.64.82.64/28",
        "34.65.242.112/28",
        "34.80.50.80/28",
        "34.88.194.0/28",
        "34.89.10.80/28",
        "34.89.198.80/28",
        "34.96.162.48/28",
        "35.247.243.240/28",
        "66.249.64.0/27",
        "66.249.64.128/27",
        "66.249.64.160/27",
        "66.249.64.192/27",
        "66.249.64.224/27",
        "66.249.64.32/27",
        "66.249.64.64/27",
        "66.249.64.96/27",
        "66.249.65.0/27",
        "66.249.65.128/27",
        "66.249.65.160/27",
        "66.249.65.192/27",
        "66.249.65.224/27",
        "66.249.65.32/27",
        "66.249.65.64/27",
        "66.249.65.96/27",
        "66.249.66.0/27",
        "66.249.66.128/27",
        "66.249.66.160/27",
        "66.249.66.192/27",
        "66.249.66.224/27",
        "66.249.66.32/27",
        "66.249.66.64/27",
        "66.249.66.96/27",
        "66.249.67.0/27",
        "66.249.67.32/27",
        "66.249.67.64/27",
        "66.249.68.0/27",
        "66.249.68.128/27",
        "66.249.68.160/27",
        "66.249.68.192/27",
        "66.249.68.32/27",
        "66.249.68.64/27",
        "66.249.68.96/27",
        "66.249.69.0/27",
        "66.249.69.128/27",
        "66.249.69.160/27",
        "66.249.69.192/27",
        "66.249.69.224/27",
        "66.249.69.32/27",
        "66.249.69.64/27",
        "66.249.69.96/27",
        "66.249.70.0/27",
        "66.249.70.128/27",
        "66.249.70.160/27",
        "66.249.70.192/27",
        "66.249.70.224/27",
        "66.249.70.32/27",
        "66.249.70.64/27",
        "66.249.70.96/27",
        "66.249.71.0/27",
        "66.249.71.128/27",
        "66.249.71.160/27",
        "66.249.71.192/27",
        "66.249.71.224/27",
        "66.249.71.32/27",
        "66.249.71.64/27",
        "66.249.71.96/27",
        "66.249.72.0/27",
        "66.249.72.128/27",
        "66.249.72.160/27",
        "66.249.72.192/27",
        "66.249.72.224/27",
        "66.249.72.32/27",
        "66.249.72.64/27",
        "66.249.72.96/27",
        "66.249.73.0/27",
        "66.249.73.128/27",
        "66.249.73.160/27",
        "66.249.73.192/27",
        "66.249.73.224/27",
        "66.249.73.32/27",
        "66.249.73.64/27",
        "66.249.73.96/27",
        "66.249.74.0/27",
        "66.249.74.128/27",
        "66.249.74.160/27",
        "66.249.74.192/27",
        "66.249.74.224/27",
        "66.249.74.32/27",
        "66.249.74.64/27",
        "66.249.74.96/27",
        "66.249.75.0/27",
        "66.249.75.128/27",
        "66.249.75.160/27",
        "66.249.75.192/27",
        "66.249.75.224/27",
        "66.249.75.32/27",
        "66.249.75.64/27",
        "66.249.75.96/27",
        "66.249.76.0/27",
        "66.249.76.128/27",
        "66.249.76.160/27",
        "66.249.76.192/27",
        "66.249.76.224/27",
        "66.249.76.32/27",
        "66.249.76.64/27",
        "66.249.76.96/27",
        "66.249.77.0/27",
        "66.249.77.128/27",
        "66.249.77.160/27",
        "66.249.77.192/27",
        "66.249.77.224/27",
        "66.249.77.32/27",
        "66.249.77.64/27",
        "66.249.77.96/27",
        "66.249.78.0/27",
        "66.249.78.128/27",
        "66.249.78.160/27",
        "66.249.78.192/27",
        "66.249.78.224/27",
        "66.249.78.32/27",
        "66.249.78.64/27",
        "66.249.78.96/27",
        "66.249.79.0/27",
        "66.249.79.128/27",
        "66.249.79.160/27",
        "66.249.79.192/27",
        "66.249.79.224/27",
        "66.249.79.32/27",
        "66.249.79.64/27"
      ]
    },
    {
      "id": "gptbot",
      "name": "GPTBot",
      "operator": {
        "name": "OpenAI",
        "url": "https://openai.com"
      },
      "description": "OpenAI's web crawler that gathers publicly available data used to train OpenAI's models.",
      "docs": [
        "https://developers.openai.com/api/docs/bots"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "GPTBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; GPTBot/1.2; +https://openai.com/gptbot"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "GPTBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://openai.com/gptbot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "132.196.86.0/24",
        "172.182.202.0/25",
        "172.182.204.0/24",
        "172.182.207.0/25",
        "172.182.214.0/24",
        "172.182.215.0/24",
        "20.125.66.80/28",
        "20.171.206.0/24",
        "20.171.207.0/24",
        "4.227.36.0/25",
        "52.230.152.0/24",
        "74.7.175.128/25",
        "74.7.227.0/25",
        "74.7.227.128/25",
        "74.7.228.0/25",
        "74.7.230.0/25",
        "74.7.241.0/25",
        "74.7.241.128/25",
        "74.7.242.0/25",
        "74.7.243.128/25",
        "74.7.244.0/25"
      ]
    },
    {
      "id": "imagesift",
      "name": "ImagesiftBot",
      "operator": {
        "name": "Hive",
        "url": "https://imagesift.com"
      },
      "description": "ImageSift's crawler, operated by Hive. It scrapes publicly available images across the web to support Hive's ImageSift reverse-image-search and web intelligence products. Standard robots.txt directives are respected.",
      "docs": [
        "https://imagesift.com/about"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "ImagesiftBot"
        ],
        "instances": [
          "Mozilla/5.0 (compatible; ImagesiftBot; +imagesift.com)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "ImagesiftBot"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-16T13:18:54.422Z",
      "resolved_cidrs": []
    },
    {
      "id": "kimi-searchbot",
      "name": "Kimi-SearchBot",
      "operator": {
        "name": "Moonshot AI",
        "url": "https://www.moonshot.ai"
      },
      "description": "Moonshot AI's search crawler for Kimi, which analyses pages for relevance and builds the index behind Kimi's search features. Blocking it removes a site from Kimi search results. Listed separately from KimiBot because Moonshot documents it as an independently configurable crawler with its own robots.txt token and its own published address list.",
      "docs": [
        "https://www.kimi.ai/policies/kimi-crawlers"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "Kimi-SearchBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; Kimi-SearchBot/1.0; +https://www.kimi.com/policies/kimi-crawlers"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "Kimi-SearchBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://www.kimi.ai/policies/kimi-searchbot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-08-23T04:44:25.122Z",
      "resolved_cidrs": [
        "47.79.225.241/32",
        "47.79.230.254/32",
        "47.79.253.42/32",
        "47.79.253.91/32"
      ]
    },
    {
      "id": "kimibot",
      "name": "KimiBot",
      "operator": {
        "name": "Moonshot AI",
        "url": "https://www.moonshot.ai"
      },
      "description": "Moonshot AI's crawler for the Kimi assistant, gathering content that may be used to train Kimi's foundation models. Moonshot documents robots.txt control per crawler and publishes a machine-readable list of the addresses its crawlers use, while noting the ranges are dynamic and may change.",
      "docs": [
        "https://www.kimi.ai/policies/kimi-crawlers"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "KimiBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; KimiBot/1.0; +https://www.kimi.com/policies/kimi-crawlers"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "KimiBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://www.kimi.ai/policies/kimibot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-08-23T04:44:25.122Z",
      "resolved_cidrs": [
        "47.79.225.241/32",
        "47.79.230.254/32",
        "47.79.253.42/32",
        "47.79.253.91/32"
      ]
    },
    {
      "id": "leipzig-lcc",
      "name": "Leipzig Corpora Collection Crawler",
      "operator": {
        "name": "Leipzig University Natural Language Processing Group",
        "url": "https://www.uni-leipzig.de/"
      },
      "description": "The LCC crawler is operated by Leipzig University to collect web text for the Leipzig Corpora Collection, a set of linguistic corpora used in natural language processing research.",
      "docs": [
        "https://corpora.wortschatz-leipzig.de/crawler_faq.html"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "^LCC "
        ],
        "instances": [
          "LCC (+http://corpora.informatik.uni-leipzig.de/crawler_faq.html)"
        ]
      },
      "behavior": {
        "respects_robots_txt": false
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-16T01:14:20.999Z",
      "resolved_cidrs": []
    },
    {
      "id": "linkupbot",
      "name": "LinkupBot",
      "operator": {
        "name": "Linkup",
        "url": "https://www.linkup.so"
      },
      "description": "Linkup's web crawler. It discovers and retrieves publicly accessible pages to build and maintain the Linkup search index, which powers search and web grounding for AI applications. Linkup publishes the CIDR ranges its crawler originates from and asks site owners to check both the source IP and the user-agent token, since user agents can be spoofed.",
      "docs": [
        "https://www.linkup.so/bot"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "LinkupBot/\\d"
        ],
        "instances": [
          "LinkupBot/1.0 (LinkupBot for web indexing; https://www.linkup.so/bot; bot@linkup.so)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "LinkupBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://www.linkup.so/linkupbot-ips.txt",
          "format": "text_lines"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-08-23T13:20:46.873Z",
      "resolved_cidrs": [
        "35.198.113.100/32"
      ]
    },
    {
      "id": "macocu",
      "name": "MaCoCu",
      "operator": {
        "name": "MaCoCu project (Jožef Stefan Institute)",
        "url": "https://www.clarin.si/info/macocu-massive-collection-and-curation-of-monolingual-and-bilingual-data/"
      },
      "description": "Crawler for the CEF-funded MaCoCu project, which collects, curates and enriches monolingual and parallel web text to build language corpora for under-resourced languages. The project publishes no IP ranges, so requests cannot be verified.",
      "docs": [
        "https://www.clarin.si/info/macocu-massive-collection-and-curation-of-monolingual-and-bilingual-data/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "MaCoCu"
        ],
        "instances": [
          "Mozilla/5.0 (compatible; MaCoCu; +https://www.clarin.si/info/macocu-massive-collection-and-curation-of-monolingual-and-bilingual-data/)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "MaCoCu"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-20T20:56:47.446Z",
      "resolved_cidrs": []
    },
    {
      "id": "meta-externalagent",
      "name": "Meta External Agent",
      "operator": {
        "name": "Meta",
        "url": "https://www.meta.com"
      },
      "description": "Meta's crawler for AI training data and content indexing across Meta products. Verified by ASN lookup (AS32934); Meta publishes no IP feed.",
      "docs": [
        "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "meta-externalagent/\\d"
        ],
        "instances": [
          "meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "meta-externalagent"
      },
      "verification": [
        {
          "type": "asn",
          "numbers": [
            32934
          ]
        }
      ],
      "tier": 2,
      "tier_label": "Verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "meta-webindexer",
      "name": "Meta Web Indexer",
      "operator": {
        "name": "Meta",
        "url": "https://www.meta.com"
      },
      "description": "Meta's crawler that navigates the web to improve the quality of Meta AI search results, analysing page content for relevance and accuracy in Meta AI responses. Verified by ASN lookup (AS32934); Meta publishes no IP feed.",
      "docs": [
        "https://developers.facebook.com/docs/sharing/webmasters/web-crawlers/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "meta-webindexer/\\d"
        ],
        "instances": [
          "meta-webindexer/1.1"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "meta-webindexer"
      },
      "verification": [
        {
          "type": "asn",
          "numbers": [
            32934
          ]
        }
      ],
      "tier": 2,
      "tier_label": "Verifiable",
      "first_seen": "2026-07-26T13:26:40.725Z",
      "resolved_cidrs": []
    },
    {
      "id": "mistralai-index",
      "name": "MistralAI-Index",
      "operator": {
        "name": "Mistral AI",
        "url": "https://mistral.ai"
      },
      "description": "Mistral AI's indexing crawler. It crawls the web automatically to build the index behind Mistral search, which answers user questions in Vibe. Mistral documents that content collected by this crawler is not used for generative AI training of any kind, and publishes the crawler's addresses as a JSON feed.",
      "docs": [
        "https://docs.mistral.ai/robots"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "MistralAI-Index/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Index/1.0; +https://docs.mistral.ai/robots)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "MistralAI-Index"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://mistral.ai/mistralai-index-ips.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-08-23T13:20:46.873Z",
      "resolved_cidrs": [
        "135.225.57.92/32",
        "20.240.192.194/32"
      ]
    },
    {
      "id": "mistralai-training",
      "name": "MistralAI-Training",
      "operator": {
        "name": "Mistral AI",
        "url": "https://mistral.ai"
      },
      "description": "Mistral AI's training crawler. It collects web content to help build the datasets used to train Mistral's generative AI models, and is documented as separate from both the search-index crawler and the user-triggered fetcher. Mistral states webmasters can disallow this user agent in robots.txt.",
      "docs": [
        "https://docs.mistral.ai/robots"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "MistralAI-Training/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; MistralAI-Training/1.0; +https://docs.mistral.ai/robots)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "MistralAI-Training"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-08-23T13:20:46.873Z",
      "resolved_cidrs": []
    },
    {
      "id": "nict-crawler",
      "name": "ICC-Crawler",
      "operator": {
        "name": "National Institute of Information and Communications Technology (NICT)",
        "url": "https://www.nict.go.jp/en/"
      },
      "description": "ICC-Crawler is a web crawler operated by Japan's NICT that collects web pages across the internet to build datasets for information and language processing research.",
      "docs": [
        "https://ucri.nict.go.jp/en/icccrawler.html"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "ICC-Crawler"
        ],
        "instances": [
          "ICC-Crawler/2.0 (Mozilla-compatible; ; http://ucri.nict.go.jp/en/icccrawler.html)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true
      },
      "verification": [
        {
          "type": "static_cidrs",
          "cidrs": [
            "202.180.34.186/32",
            "61.86.246.72/32"
          ]
        }
      ],
      "tier": 2,
      "tier_label": "Verifiable",
      "first_seen": "2026-07-16T01:14:20.999Z",
      "resolved_cidrs": [
        "202.180.34.186/32",
        "61.86.246.72/32"
      ]
    },
    {
      "id": "oai-searchbot",
      "name": "OAI-SearchBot",
      "operator": {
        "name": "OpenAI",
        "url": "https://openai.com"
      },
      "description": "OpenAI's crawler that indexes pages to power search results inside ChatGPT. It is distinct from GPTBot and is not used to gather training data.",
      "docs": [
        "https://developers.openai.com/api/docs/bots"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "OAI-SearchBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36; compatible; OAI-SearchBot/1.4; +https://openai.com/searchbot"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "OAI-SearchBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://openai.com/searchbot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "104.210.140.128/28",
        "135.234.64.0/24",
        "172.182.193.224/28",
        "172.182.193.80/28",
        "172.182.194.144/28",
        "172.182.194.32/28",
        "172.182.195.48/28",
        "172.182.209.208/28",
        "172.182.211.192/28",
        "172.182.213.192/28",
        "172.182.224.0/28",
        "172.203.190.128/28",
        "20.14.99.96/28",
        "20.168.18.32/28",
        "20.169.6.224/28",
        "20.169.7.48/28",
        "20.169.77.0/25",
        "20.171.123.64/28",
        "20.171.53.224/28",
        "20.25.151.224/28",
        "20.42.10.176/28",
        "4.227.36.0/25",
        "40.67.175.0/25",
        "40.90.214.16/28",
        "51.8.102.0/24",
        "74.7.175.128/25",
        "74.7.228.0/25",
        "74.7.228.128/25",
        "74.7.229.0/25",
        "74.7.229.128/25",
        "74.7.230.0/25",
        "74.7.241.128/25",
        "74.7.242.128/25",
        "74.7.243.0/25",
        "74.7.244.0/25"
      ]
    },
    {
      "id": "omgili",
      "name": "omgili",
      "operator": {
        "name": "Webz.io",
        "url": "https://webz.io"
      },
      "description": "Webz.io's legacy crawler user agent (formerly \"Omgilibot\"), used to collect web, forum, and news content for its data feeds. Webz.io's current public documentation describes successor crawlers (\"webzio\" / \"webzio-extended\") and no longer documents this UA string or any IP verification method for it.",
      "docs": [
        "https://webz.io/blog/company/from-omgilibot-to-the-webzbot-duo-a-powerful-leap-for-ethical-and-comprehensive-data-collection/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "omgili"
        ],
        "instances": [
          "omgili/0.5 +http://omgili.com"
        ]
      },
      "behavior": {
        "respects_robots_txt": true
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": []
    },
    {
      "id": "perplexitybot",
      "name": "PerplexityBot",
      "operator": {
        "name": "Perplexity",
        "url": "https://www.perplexity.ai"
      },
      "description": "Perplexity's crawler that indexes pages to surface and link websites in Perplexity search results. Perplexity states it is not used to train models.",
      "docs": [
        "https://docs.perplexity.ai/docs/resources/perplexity-crawlers"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "PerplexityBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; PerplexityBot/1.0; +https://perplexity.ai/perplexitybot)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "PerplexityBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://www.perplexity.com/perplexitybot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "107.20.236.150/32",
        "18.210.92.235/32",
        "18.97.1.228/30",
        "18.97.9.96/29",
        "3.211.124.183/32",
        "3.222.232.239/32",
        "3.224.62.45/32",
        "3.231.139.107/32"
      ]
    },
    {
      "id": "sbintuitions-bot",
      "name": "SBIntuitionsBot",
      "operator": {
        "name": "SB Intuitions Corp.",
        "url": "https://www.sbintuitions.co.jp"
      },
      "description": "Crawler operated by SB Intuitions (a SoftBank AI subsidiary) that collects web pages for AI development and information analysis, including training of its Sarashina language models. SB Intuitions publishes no IP ranges, so requests cannot be verified.",
      "docs": [
        "https://www.sbintuitions.co.jp/bot/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "SBIntuitionsBot/[\\d.]+"
        ],
        "instances": [
          "Mozilla/5.0 (compatible; SBIntuitionsBot/0.1; +https://www.sbintuitions.co.jp/bot/)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "SBIntuitionsBot"
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-20T20:56:47.446Z",
      "resolved_cidrs": []
    },
    {
      "id": "shapbot",
      "name": "ShapBot",
      "operator": {
        "name": "Parallel Web Systems",
        "url": "https://parallel.ai"
      },
      "description": "Parallel's indexing crawler, which collects publicly available web content to build and maintain the search index behind Parallel's web APIs. The operator documents the robots.txt token ShapBot and publishes the crawler's addresses as a JSON feed.",
      "docs": [
        "https://parallel.ai/parallel-web-systems-bots"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "ShapBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko); compatible; ShapBot/0.1.0"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "ShapBot"
      },
      "verification": [
        {
          "type": "cidr_feed",
          "url": "https://docs.parallel.ai/resources/shapbot.json",
          "format": "prefixes"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-08-23T13:20:46.873Z",
      "resolved_cidrs": [
        "23.251.146.115/32",
        "34.122.173.216/32",
        "34.31.203.120/32",
        "34.44.142.114/32",
        "34.44.196.215/32",
        "34.45.55.180/32",
        "34.46.138.166/32",
        "34.68.39.29/32",
        "34.70.129.79/32",
        "34.9.172.22/32"
      ]
    },
    {
      "id": "webzio-extended",
      "name": "Webzio-extended",
      "operator": {
        "name": "Webz.io Ltd.",
        "url": "https://webz.io"
      },
      "description": "Second crawler in Webz.io's crawler pair, which performs ethical validation on the data collected by Webzio and tags it as usable or not usable for AI and machine-learning training. Webz.io publishes no IP ranges.",
      "docs": [
        "https://webz.io/blog/company/an-overview-of-the-webz-io-duo-of-crawlers/"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "webzio-extended"
        ],
        "instances": [
          "webzio-extended (+https://webz.io/bot.html)"
        ]
      },
      "behavior": {
        "respects_robots_txt": true
      },
      "verification": [],
      "tier": 3,
      "tier_label": "Listed only",
      "first_seen": "2026-07-26T13:26:40.725Z",
      "resolved_cidrs": []
    },
    {
      "id": "youbot",
      "name": "YouBot",
      "operator": {
        "name": "You.com",
        "url": "https://you.com"
      },
      "description": "You.com's crawler that indexes pages for its AI-powered search product. You.com documents a dedicated IP range, a reverse-DNS pattern, and support for signed-request verification via Web Bot Auth.",
      "docs": [
        "https://you.com/docs/youbot"
      ],
      "category": "ai-crawler",
      "user_agents": {
        "patterns": [
          "YouBot/\\d"
        ],
        "instances": [
          "Mozilla/5.0 AppleWebKit/537.36 (KHTML, like Gecko; compatible; YouBot/1.0; +https://docs.you.com/youbot; env:prod) Chrome/W.X.Y.Z Safari/537.36"
        ]
      },
      "behavior": {
        "respects_robots_txt": true,
        "robots_token": "YouBot"
      },
      "verification": [
        {
          "type": "static_cidrs",
          "cidrs": [
            "68.67.112.0/24"
          ]
        },
        {
          "type": "rdns",
          "masks": [
            "*.search.you.com"
          ]
        },
        {
          "type": "web_bot_auth",
          "signature_agent_url": "https://you.com/.well-known/http-message-signatures-directory"
        }
      ],
      "tier": 1,
      "tier_label": "Fully verifiable",
      "first_seen": "2026-07-08T19:13:49.571Z",
      "resolved_cidrs": [
        "68.67.112.0/24"
      ]
    }
  ]
}