# robots.txt for imigo.ai User-agent: * Allow: / # Allow access to CSS and JS files Allow: /css/ Allow: /js/ Allow: /_next/ Allow: /images/ Allow: /assets/ Allow: /fonts/ # Allow specific pages and sections Allow: /en/ Allow: /ru/ Allow: /en/media/ Allow: /ru/media/ # Disallow admin and API endpoints Disallow: /admin/ Disallow: /api/ Disallow: /dashboard/ Disallow: /_next/webpack-hmr Disallow: /private/ # Clean-param directive for Yandex (ignore tracking and analytics parameters) Clean-param: utm_source&utm_medium&utm_campaign&utm_content&utm_term / Clean-param: fbclid&gclid&yclid&_ga&_gl&mc_cid&mc_eid / Clean-param: ref&referrer&source&campaign&medium / Clean-param: sessionid&sid&jsessionid&PHPSESSID / Clean-param: timestamp&t&_t&cache&v&ver&version / Clean-param: userId¤tTemplateId&etext&fbclid&gad_source&gad_campaignid&wbraid&gptImage&forSubRoute&callbackUrl&error&routeQuery&m-message-key-id&m-message-click-id&trainedModel&midjourney / # Disallow specific problematic parameters that create duplicate content Disallow: /*?print= Disallow: /*?pdf= Disallow: /*?share= Disallow: /*?download= Disallow: /*&print= Disallow: /*&pdf= Disallow: /*&share= Disallow: /*&download= # ── AI bots policy ────────────────────────────────────────────────────── # Two distinct categories that the previous "block them all" rule mixed up: # # 1) TRAINING crawlers — feed model training corpora. We block these to # avoid donating content for training without consent. They do NOT serve # query-time answers, so blocking them costs zero AI-search visibility. # # 2) RUNTIME / SEARCH crawlers — fetch pages on demand when a user asks # ChatGPT/Perplexity/Claude a question. Blocking them removes the site # from real-time AI answers and citations. We allow these (with a soft # Crawl-delay as a courtesy), and rely on edge rate-limiting and our # CDN cache to absorb load. # Training-only — blocked User-agent: GPTBot Disallow: / User-agent: CCBot Disallow: / User-agent: cohere-ai Disallow: / User-agent: anthropic-ai Disallow: / # Runtime/search — allowed with soft crawl-delay User-agent: ChatGPT-User Allow: / Crawl-delay: 1 User-agent: OAI-SearchBot Allow: / Crawl-delay: 1 User-agent: PerplexityBot Allow: / Crawl-delay: 1 User-agent: Perplexity-User Allow: / Crawl-delay: 1 User-agent: ClaudeBot Allow: / Crawl-delay: 1 User-agent: Claude-Web Allow: / Crawl-delay: 1 # Google-Extended controls inclusion in BOTH AI Overviews/AI Mode AND Gemini # grounding. Allowing it puts us in those answer surfaces. Googlebot is # already covered by the default "User-agent: * Allow: /" above and is # what actually serves Search; Google-Extended is the AI-specific opt-in. User-agent: Google-Extended Allow: / # Block common crawlers that might overload the server User-agent: AhrefsBot Crawl-delay: 10 User-agent: SemrushBot Crawl-delay: 10 User-agent: MJ12bot Crawl-delay: 10 # Regional specific directives for Yandex User-agent: YandexBot Allow: /ru/ Crawl-delay: 1 # Sitemap location Sitemap: https://imigo.ai/sitemap.xml