{
  "@context": "https://schema.org",
  "@type": "ScholarlyArticle",
  "name": "AI Crawler, Training, and Retrieval Control Matrix",
  "url": "https://intelligencecompact.com/research/ai-crawler-control-matrix/",
  "description": "An independent taxonomy of training crawlers, search/retrieval crawlers, user-initiated fetchers, product control tokens, robots controls, and identity-verification concerns across major AI ecosystems.",
  "datePublished": "2026-09-04",
  "dateModified": "2026-09-04",
  "publisher": {
    "@type": "Organization",
    "name": "Intelligence Compact",
    "url": "https://intelligencecompact.com/"
  },
  "articleSection": [
    "AI crawlers",
    "GPTBot",
    "OAI-SearchBot",
    "ClaudeBot",
    "Google-Extended",
    "robots.txt"
  ],
  "wordCount": 5898,
  "identifier": {
    "@type": "PropertyValue",
    "propertyID": "sha256",
    "value": "fbf94abb079abfc4d8946ad0f12bd87f766dbf836f41684a47d310b8e65c2160"
  },
  "citation": [
    "https://www.anglera.com/glossary/gptbot",
    "https://www.xseek.io/docs/openai-crawlers-and-user-agents",
    "https://ppc.land/claudebot/",
    "https://developers.openai.com/api/docs/bots",
    "https://www.xseek.io/docs/claude-user-agents",
    "https://www.xseek.io/docs/llama-user-agents",
    "https://crawlercheck.com/directory/ai-bots/oai-searchbot",
    "https://www.searchenginejournal.com/anthropics-claude-bots-make-robots-txt-decisions-more-granular/568253/",
    "https://www.seroundtable.com/anthropic-updates-its-crawler-docs-40978.html",
    "https://www.soar.sh/blog/ai-bots-robots-txt-guide",
    "https://docs.mistral.ai/robots",
    "https://usehardal.com/blog/how-to-measure-claudebot-traffic",
    "https://www.tryrankly.com/agent-directory/operator/anthropic",
    "https://docs.peec.ai/connecting-your-data",
    "https://ppc.land/bingbot/",
    "https://www.transfon.com/blog/top-bots-2025",
    "https://developers.google.com/crawling/docs/crawlers-fetchers/google-common-crawlers",
    "https://support.apple.com/en-us/119829",
    "https://www.searchenginejournal.com/openais-crawler-docs-now-list-oai-adsbot-for-chatgpt-ads/572861/",
    "https://crawlercheck.com/directory/ai-bots/claudebot",
    "https://llmpulse.ai/ai-crawler-index/claudebot",
    "https://www.searchenginejournal.com/google-revamps-crawler-documentation/527424/",
    "https://www.seroundtable.com/google-updates-its-google-crawlers-and-fetchers-documentation-38073.html",
    "https://www.lumar.io/blog/best-practice/technical-geo-aeo-guide-for-ai-search-optimization/",
    "https://trakkr.ai/bots/google-extended",
    "https://www.stanventures.com/blog/googlebot-user-agent-string/",
    "https://www.botsights.com/bots",
    "https://www.switchtheweb.com/agents/bingbot",
    "https://www.getaiso.com/ai-bots/bingbot",
    "https://getfound3.com/blog/how-do-i-get-found-in-microsoft-copilot",
    "https://www.bing.com/webmasters/help/which-crawlers-does-bing-use-8c184ec0",
    "https://crawlercheck.com/directory/search-engines/applebot-extended",
    "https://www.seroundtable.com/apple-updates-applebot-docs-39310.html",
    "https://ppc.land/applebot/",
    "https://kitbase.dev/bot-directory/meta-externalads",
    "https://www.menra.ai/glossary/meta-externalagent",
    "https://datafa.st/crawlers/mistral-mistralai-user",
    "https://datadome.co/bots/mistralai-user/",
    "https://radar.cloudflare.com/bots/directory/mistralai-user",
    "https://arxiv.org/html/2606.10711v1",
    "https://tribune.com.pk/story/2472893/ai-firms-accused-of-scraping-publisher-sites-without-permission",
    "https://www.buzzstream.com/blog/publishers-block-ai-study/",
    "https://pressgazette.co.uk/platforms/third-party-scrapers-are-stealing-publisher-content-to-order-for-ai-companies/",
    "https://smarterarticles.co.uk/when-ai-devours-the-news-who-pays-for-truth",
    "https://help.aikido.dev/zen-firewall/miscellaneous/bot-protection-details",
    "https://www.browscap.org/stream?q=BrowsCapINI",
    "https://crawlercheck.com/directory/cloud-services/amazonbot",
    "https://www.humansecurity.com/learn/blog/crawlers-list-known-bots-guide/",
    "https://datadome.co/bots/claudebot/",
    "https://www.sorank.com/glossary-geo-seo/openai-crawlers",
    "https://www.anagram.ai/blog/gptbot-explained-how-chatgpt-crawls-sees-and-cites-your-site-in-2026",
    "https://thetechthinker.com/how-to-make-website-crawl-in-ai-engine/",
    "https://www.seosiri.com/2026/06/automated-llmstxt-centralized-sitemap-guide.html"
  ],
  "isBasedOn": {
    "@type": "CreativeWork",
    "name": "AI Crawler Control Matrix.md",
    "identifier": "fbf94abb079abfc4d8946ad0f12bd87f766dbf836f41684a47d310b8e65c2160"
  },
  "additionalProperty": [
    {
      "@type": "PropertyValue",
      "name": "source_completeness",
      "value": "received_complete"
    },
    {
      "@type": "PropertyValue",
      "name": "adoption_status",
      "value": "The taxonomy and identity-verification concerns inform operations. The report’s recommendation to block training crawlers is not adopted because it conflicts with the project’s current explicit training-eligibility objective."
    }
  ],
  "license": "https://intelligencecompact.com/content-use/"
}
