# robots.txt for mica-compliance.shop # Last updated: 2026-04-08 # # Strategy (must match .htaccess access policy): # - Allow only Google, Bing, Yandex, Mail.ru search engines # - Allow every AI crawler, training crawlers included (2026-08-28) # - Disallow every SEO backlink / audit crawler # - Disallow every download tool / bulk scraper # - Disallow common dev/build paths in the default rule # # .htaccess enforces the same policy at the HTTP layer (scrapers get 403, # allowed search engines get 301 to canonical target, real browsers get # served the page content). robots.txt is the polite-crawler declaration of # the same rules for crawlers that respect it. # ============================================================================= # ALLOWED SEARCH ENGINES (these are the only bots that should crawl) # ============================================================================= User-agent: Googlebot Allow: / User-agent: Googlebot-Image Allow: / User-agent: Googlebot-News Allow: / User-agent: Googlebot-Video Allow: / User-agent: Googlebot-Mobile Allow: / User-agent: APIs-Google Allow: / User-agent: Google-InspectionTool Allow: / User-agent: Storebot-Google Allow: / User-agent: bingbot Allow: / User-agent: BingPreview Allow: / User-agent: msnbot Allow: / User-agent: adidxbot Allow: / User-agent: YandexBot Allow: / User-agent: YandexImages Allow: / User-agent: YandexNews Allow: / User-agent: YandexMobileBot Allow: / User-agent: YandexMetrika Allow: / User-agent: Mail.RU_Bot Allow: / # ============================================================================= # AI CRAWLERS. Policy: FULLY OPEN, TRAINING INCLUDED. # User decision 2026-08-28, replacing the citation-only policy of 2026-06-14. # # These landings want to be read by neural networks and want to be part of # what those models are trained on. Every AI crawler below is allowed: the # model-training corpus crawlers as well as the live-answer and search-index # ones. .htaccess grants the same permission at the HTTP layer, and its # section 5c waves the crawlers whose User-Agent lacks a Mozilla prefix past # the catch-net, so this declaration is enforceable rather than decorative. # # Anthropic's asymmetric block is reversed by the same decision. # ============================================================================= # OpenAI User-agent: GPTBot Allow: / User-agent: OAI-SearchBot Allow: / User-agent: ChatGPT-User Allow: / # Anthropic User-agent: ClaudeBot Allow: / User-agent: Claude-User Allow: / User-agent: Claude-SearchBot Allow: / User-agent: claude-web Allow: / User-agent: anthropic-ai Allow: / # Google AI surfaces User-agent: Google-Extended Allow: / # Apple Intelligence User-agent: Applebot Allow: / User-agent: Applebot-Extended Allow: / # Perplexity User-agent: PerplexityBot Allow: / User-agent: Perplexity-User Allow: / # Common Crawl User-agent: CCBot Allow: / User-agent: CCResearchBot Allow: / # ByteDance User-agent: Bytespider Allow: / # Meta AI User-agent: Meta-ExternalAgent Allow: / User-agent: Meta-ExternalFetcher Allow: / User-agent: FacebookBot Allow: / # Amazon User-agent: Amazonbot Allow: / # Mistral User-agent: MistralAI-User Allow: / # DuckDuckGo AI assist User-agent: DuckAssistBot Allow: / # Other model and dataset crawlers User-agent: Diffbot Allow: / User-agent: Omgilibot Allow: / User-agent: Omgili Allow: / User-agent: cohere-ai Allow: / User-agent: cohere-training-data-crawler Allow: / User-agent: YouBot Allow: / User-agent: Bravebot Allow: / User-agent: Neevabot Allow: / User-agent: FriendlyCrawler Allow: / User-agent: ImagesiftBot Allow: / User-agent: img2dataset Allow: / User-agent: Timpibot Allow: / User-agent: PanguBot Allow: / User-agent: webzio-extended Allow: / User-agent: iaskspider Allow: / User-agent: AI2Bot Allow: / User-agent: VelenPublicWebCrawler Allow: / # Not neural-network crawlers, so the opening above does not reach them. # TurnitinBot builds a plagiarism corpus; the rest are social listening. User-agent: TurnitinBot Disallow: / User-agent: AwarioSmartBot Disallow: / User-agent: Kangaroo Bot Disallow: / User-agent: Scoop.it Disallow: / # ============================================================================= # SEO / BACKLINK / AUDIT CRAWLERS (disallow, they extract for competitors) # ============================================================================= User-agent: AhrefsBot Disallow: / User-agent: SemrushBot Disallow: / User-agent: Semrush Disallow: / User-agent: MJ12bot Disallow: / User-agent: DotBot Disallow: / User-agent: BLEXBot Disallow: / User-agent: rogerbot Disallow: / User-agent: MegaIndex Disallow: / User-agent: Serpstat Disallow: / User-agent: SerpstatBot Disallow: / User-agent: sistrix Disallow: / User-agent: SiteAuditBot Disallow: / User-agent: SeznamBot Disallow: / User-agent: spbot Disallow: / User-agent: linkdexbot Disallow: / User-agent: SpyFu Disallow: / User-agent: MajesticSEO Disallow: / User-agent: Majestic-12 Disallow: / User-agent: BarkRowler Disallow: / User-agent: SeekportBot Disallow: / User-agent: Exabot Disallow: / User-agent: Cliqzbot Disallow: / User-agent: Searchmetricsbot Disallow: / User-agent: BacklinkCrawler Disallow: / User-agent: LinkpadBot Disallow: / User-agent: Linkdex Disallow: / User-agent: Cocolyzebot Disallow: / User-agent: DataForSeoBot Disallow: / # ============================================================================= # NON-WHITELIST SEARCH ENGINES (disallow, user whitelisted only 4) # ============================================================================= User-agent: baiduspider Disallow: / User-agent: Baiduspider Disallow: / User-agent: sogou Disallow: / User-agent: coccoc Disallow: / User-agent: Naver Disallow: / User-agent: Yeti Disallow: / User-agent: DuckDuckBot Disallow: / User-agent: Qwantbot Disallow: / User-agent: Qwantify Disallow: / User-agent: Ecosia Disallow: / User-agent: PetalBot Disallow: / # ============================================================================= # DOWNLOAD TOOLS / OFFLINE BROWSERS / BULK SCRAPERS (disallow) # ============================================================================= User-agent: HTTrack Disallow: / User-agent: Wget Disallow: / User-agent: WebCopier Disallow: / User-agent: WebStripper Disallow: / User-agent: WebZIP Disallow: / User-agent: Teleport Disallow: / User-agent: TeleportPro Disallow: / User-agent: GetRight Disallow: / User-agent: FlashGet Disallow: / User-agent: LeechGet Disallow: / User-agent: MassDownloader Disallow: / User-agent: Harvester Disallow: / User-agent: EmailCollector Disallow: / User-agent: EmailSiphon Disallow: / User-agent: EmailWolf Disallow: / User-agent: ExtractorPro Disallow: / User-agent: LinkExtractorPro Disallow: / # ============================================================================= # ARCHIVE CRAWLERS (disallow, prevent Wayback caching of landing content) # ============================================================================= User-agent: ia_archiver Disallow: / User-agent: archive.org_bot Disallow: / User-agent: Archive-It Disallow: / User-agent: wayback Disallow: / # --- BEGIN generated block: master/data/bot-blacklist-additions.txt --- # ============================================================================= # BLACKLIST SYNC 2026-08-01 (SEO platforms, RU SEO, uptime monitoring, # social preview bots, AI aggregators, other crawlers) # # Mirrors the .htaccess section 4 blacklist. These crawlers get a 403 at the # HTTP layer; this is the polite declaration for the ones that read robots.txt. # ============================================================================= User-agent: AhrefsSiteAudit Disallow: / User-agent: Sitebulb Disallow: / User-agent: Oncrawl Disallow: / User-agent: deepcrawl Disallow: / User-agent: Lumar Disallow: / User-agent: RyteBot Disallow: / User-agent: WooRank Disallow: / User-agent: SERankingBot Disallow: / User-agent: Netpeak Disallow: / User-agent: WebCEO Disallow: / User-agent: SEOprofiler Disallow: / User-agent: RankRanger Disallow: / User-agent: AdvancedWebRanking Disallow: / User-agent: awrcloud Disallow: / User-agent: BrightEdge Disallow: / User-agent: ConductorSearchBot Disallow: / User-agent: seoClarityBot Disallow: / User-agent: SEOmonitor Disallow: / User-agent: SEO PowerSuite Disallow: / User-agent: LinkResearchTools Disallow: / User-agent: SimilarWeb Disallow: / User-agent: MozBot Disallow: / User-agent: MajesticBot Disallow: / User-agent: SEOstats Disallow: / User-agent: SEOdiver Disallow: / User-agent: SiteExplorer Disallow: / User-agent: Seopult Disallow: / User-agent: RookeeBot Disallow: / User-agent: WebArtexBot Disallow: / User-agent: Miralinks Robot Disallow: / User-agent: LinksMasterRoBot Disallow: / User-agent: StatOnlineRuBot Disallow: / User-agent: LinkStats Disallow: / User-agent: CNCat Disallow: / User-agent: Runet-Research-Crawler Disallow: / User-agent: SurdotlyBot Disallow: / User-agent: WebAlta Disallow: / User-agent: Sovetnik Disallow: / User-agent: UptimeRobot Disallow: / User-agent: Pingdom Disallow: / User-agent: StatusCake Disallow: / User-agent: Site24x7 Disallow: / User-agent: PRTG Disallow: / User-agent: Monitority Disallow: / User-agent: Netcraft Disallow: / User-agent: Xenu Disallow: / User-agent: W3C-checklink Disallow: / User-agent: deadlinkchecker Disallow: / User-agent: Upflow Disallow: / User-agent: SafeDNSBot Disallow: / User-agent: Virusdie_crawler Disallow: / User-agent: Twitterbot Disallow: / User-agent: Slackbot-LinkExpanding Disallow: / User-agent: facebookexternalhit Disallow: / User-agent: CCResearchBot Disallow: / User-agent: ISSCyberRiskCrawler Disallow: / User-agent: PiplBot Disallow: / User-agent: MojeekBot Disallow: / User-agent: DataparkSearch Disallow: / User-agent: SearchBlox Disallow: / User-agent: netEstate Disallow: / User-agent: Vagabondo Disallow: / User-agent: VoilaBot Disallow: / User-agent: Cityreview Disallow: / User-agent: Sysomos Disallow: / User-agent: Jyxobot Disallow: / User-agent: Twiceler Disallow: / User-agent: NextGenSearchBot Disallow: / User-agent: NimbleCrawler Disallow: / User-agent: WISENutbot Disallow: / User-agent: Snapbot Disallow: / User-agent: 360Spider Disallow: / User-agent: 80legs Disallow: / User-agent: Aboundex Disallow: / User-agent: SpiderLing Disallow: / User-agent: PixelTools Disallow: / User-agent: TprAdsTxtCrawler Disallow: / User-agent: weborama Disallow: / User-agent: meanpathbot Disallow: / User-agent: NetSeer Disallow: / User-agent: mfibot Disallow: / User-agent: Gluten Free Crawler Disallow: / User-agent: Site-Shot Disallow: / User-agent: Readability Disallow: / User-agent: Claude-User Disallow: / User-agent: Claude-SearchBot Disallow: / User-agent: Seekport Disallow: / # --- END generated block --- # ============================================================================= # DEFAULT (any bot not listed above) # # Allows crawling by polite crawlers that are not explicitly blacklisted, # but disallows dev/build artifacts. Note: the .htaccess Mozilla check # already 403s any UA that is neither a real browser nor an allowed bot, so # this default rule is a second line of defense for crawlers that DO spoof # Mozilla but respect robots.txt. # ============================================================================= User-agent: * Disallow: /.env Disallow: /.git/ Disallow: /.svn/ Disallow: /.hg/ Disallow: /.idea/ Disallow: /.vscode/ Disallow: /.DS_Store Disallow: /backups/ Disallow: /node_modules/ Disallow: /vendor/ Disallow: /deploy.sh Disallow: /up.sh Disallow: /package.json Disallow: /package-lock.json Disallow: /yarn.lock Disallow: /pnpm-lock.yaml Disallow: /README.md Disallow: /CHANGELOG.md Disallow: /LICENSE Disallow: /Makefile Disallow: /*.log$ Disallow: /*.bak$ Disallow: /*.swp$ Disallow: /*.swo$ Disallow: /*.sql$ Disallow: /*.sqlite$ Disallow: /*.env$ Disallow: /*.sh$ Disallow: /*.md$ Disallow: /*.yml$ Disallow: /*.yaml$ Disallow: /*.toml$ Disallow: /*.sample$ Disallow: /*.example$ # Crawl-delay for polite crawlers (Google ignores; Yandex + Bing respect) Crawl-delay: 2 # ============================================================================= # SITEMAP # ============================================================================= Sitemap: https://mica-compliance.shop/sitemap.xml