# # robots.txt # # This file is to prevent the crawling and indexing of certain parts # of your site by web crawlers and spiders run by sites like Yahoo! # and Google. By telling these "robots" where not to go on your site, # you save bandwidth and server resources. # # This file will be ignored unless it is at the root of your host: # Used: http://example.com/robots.txt # Ignored: http://example.com/site/robots.txt # # For more information about the robots.txt standard, see: # http://www.robotstxt.org/robotstxt.html User-agent: * # CSS, JS, Images Allow: /core/*.css$ Allow: /core/*.css? Allow: /core/*.js$ Allow: /core/*.js? Allow: /core/*.gif Allow: /core/*.jpg Allow: /core/*.jpeg Allow: /core/*.png Allow: /core/*.svg Allow: /profiles/*.css$ Allow: /profiles/*.css? Allow: /profiles/*.js$ Allow: /profiles/*.js? Allow: /profiles/*.gif Allow: /profiles/*.jpg Allow: /profiles/*.jpeg Allow: /profiles/*.png Allow: /profiles/*.svg # Directories Disallow: /core/ Disallow: /profiles/ # Files Disallow: /README.md Disallow: /composer/Metapackage/README.txt Disallow: /composer/Plugin/ProjectMessage/README.md Disallow: /composer/Plugin/Scaffold/README.md Disallow: /composer/Plugin/VendorHardening/README.txt Disallow: /composer/Template/README.txt Disallow: /modules/README.txt Disallow: /sites/README.txt Disallow: /themes/README.txt Disallow: /web.config # Paths (clean URLs) Disallow: /admin/ Disallow: /comment/reply/ Disallow: /filter/tips Disallow: /node/add/ Disallow: /search/ Disallow: /user/register Disallow: /user/password Disallow: /user/login Disallow: /user/logout Disallow: /media/oembed Disallow: /*/media/oembed # Paths (no clean URLs) Disallow: /index.php/admin/ Disallow: /index.php/comment/reply/ Disallow: /index.php/filter/tips Disallow: /index.php/node/add/ Disallow: /index.php/search/ Disallow: /index.php/user/password Disallow: /index.php/user/register Disallow: /index.php/user/login Disallow: /index.php/user/logout Disallow: /index.php/media/oembed Disallow: /index.php/*/media/oembed # MGoBlog additions (appended to core's robots.txt by drupal-scaffold — # edit assets/scaffold/robots-additions.txt, not web/robots.txt, which is # regenerated on every composer build). # Comment permalinks render full thin-duplicate pages of their parent node # and account for the bulk of crawler load at the origin. Disallow: /comment/ # Junk multi-pager combos from the D6 era (?page=0,1,1,0,...) — Yahoo Slurp # grinds thousands of these into 404s daily. The comma in the pattern means # normal single-pager URLs (?page=2) stay crawlable. Slurp requests these # URLs with the comma percent-encoded (?page=0%2C0%2C...), and RFC 9309 # normalization of reserved characters can't be assumed for legacy # crawlers, so both spellings are listed. Disallow: /*?page=*,* Disallow: /*?page=*%2C* # Machine-readable content policy (contentsignals.org): allow search # indexing, disallow AI training. An express rights reservation under # Article 4 of EU Directive 2019/790. Content-Signal: search=yes, ai-train=no Sitemap: https://mgoblog.com/sitemap.xml # AI-training opt-outs. Google-Extended and Applebot-Extended are # robots.txt-only tokens (Googlebot/Applebot do the crawling; these # control AI-training use) — no WAF rule can express them. User-agent: Google-Extended Disallow: / User-agent: Applebot-Extended Disallow: / # AI-training crawlers. Most are also blocked at the Cloudflare WAF # ("AI Crawl Control" rule, which exempts /robots.txt so they can read # this); the robots entry asks them not to request pages at all. User-agent: CCBot Disallow: / User-agent: GPTBot Disallow: / User-agent: ChatGPT-User Disallow: / User-agent: ClaudeBot Disallow: / User-agent: Amazonbot Disallow: / User-agent: Bytespider Disallow: / User-agent: PetalBot Disallow: / User-agent: meta-externalagent Disallow: / User-agent: CloudflareBrowserRenderingCrawler Disallow: /