# Universal robots.txt base for public WordPress sites # Replace example.com with the canonical host of each site User-agent: * Disallow: /wp-admin/ Allow: /wp-admin/admin-ajax.php # Optional internal-search crawl control # Enable only when these URLs create a crawl trap or search-spam problem. # Make sure search result pages are already noindex or return 404 for junk queries. # Disallow: /*?s= # Disallow: /*&s= # Disallow: /search/ # WordPress duplicate and transient URLs Disallow: /*?replytocom= Disallow: /*&replytocom= Disallow: /*?preview= Disallow: /*&preview= Disallow: /*?customize_changeset_uuid= Disallow: /*&customize_changeset_uuid= # Yandex selects this group instead of User-agent: * User-agent: Yandex Disallow: /wp-admin/ Allow: /wp-admin/admin-ajax.php # Optional internal-search crawl control # Disallow: /*?s= # Disallow: /*&s= # Disallow: /search/ Disallow: /*?preview= Disallow: /*&preview= Disallow: /*?customize_changeset_uuid= Disallow: /*&customize_changeset_uuid= # Ignore tracking-only parameters and consolidate duplicate URLs Clean-param: replytocom&fbclid&gclid&yclid&_ga # Add only if these parameters are actually used as tracking-only parameters # Clean-param: msclkid&dclid&_gl&gbraid&wbraid&gad_source # Opt out of model-training crawlers without blocking their search crawlers User-agent: GPTBot Disallow: / User-agent: ClaudeBot Disallow: / User-agent: Applebot-Extended Disallow: / # Optional strict opt-out for Gemini training and grounding. # This does not affect ordinary Google Search, but can reduce Gemini visibility. # User-agent: Google-Extended # Disallow: / # Actual sitemap URL for this specific site Sitemap: https://cyclepedia.ru/sitemap_index.xml