Sitemap: https://www.joopzy.com/wp-sitemap.xml # ============================================================================= # ONE group, applying to every crawler. # # DO NOT add a per-crawler group such as: # # User-agent: Googlebot # Allow: / # # A crawler obeys ONLY its most specific matching group and ignores the "*" # group entirely. A group like that therefore switches OFF every rule below for # that crawler — it does not "additionally allow" anything, because whatever is # not disallowed is already allowed by default. # # This file used to do exactly that. Googlebot's complete ruleset was "Allow: /", # so it was free to crawl wp-admin, the cart, the checkout and every parameter # URL on the site. Separately, the WooCommerce and crawler-trap rules were placed # after "User-agent: YoudaoBot/1.0" with no new User-agent line, so all ~26 of # them applied to YoudaoBot alone. Search Console reported 22M non-indexed pages, # millions of duplicate URLs and hundreds of thousands of 404s. # # NOTE ON FORMATTING: there must be NO BLANK LINE between a User-agent line and # its rules, or anywhere inside a group. A blank line ends the record for strict # parsers, which would orphan every rule after it. Comment lines are safe and are # used as separators below instead. # # Some rules are deliberately redundant: /*?* already covers the ?orderby=, # ?filter and ?s= patterns. They are kept so that relaxing /*?* later does not # silently unprotect everything else at the same time. # ============================================================================= User-agent: * # --- WordPress internals --- Allow: /wp-admin/admin-ajax.php Disallow: /wp-admin/ Disallow: /wp-includes/ Disallow: /readme.html Disallow: /license.txt Disallow: /xmlrpc.php Disallow: /wp-login.php Disallow: /wp-register.php # --- Media stays crawlable, so image search keeps working even though the # query-string rule below is broad. --- Allow: /wp-content/uploads/ # --- Any URL carrying a query string --- # Shop filters, sorting and pagination parameters generate an effectively # unbounded URL space. Those pages already return "noindex, follow"; this stops # them consuming crawl budget that belongs to the ~5,900 real product pages. Disallow: /*?* Disallow: /*? Disallow: /*~* Disallow: /*~ # --- Personal, transactional, or nothing to index --- Disallow: /cart/ Disallow: /checkout/ Disallow: /my-account/ Disallow: /search/ Disallow: /search # --- WooCommerce sorting, filtering and cart actions --- # (subsumed by /*?* above; kept as defence in depth) Disallow: /*?orderby=price Disallow: /*?orderby=price-desc Disallow: /*?orderby=rating Disallow: /*?orderby=date Disallow: /*?orderby=popularity Disallow: /*?orderby=title Disallow: /*?orderby=desc Disallow: /*?filter Disallow: /*add-to-cart=* Disallow: /*add_to_wishlist=* Disallow: /*?paged=&count=* Disallow: /*?count=* # --- Crawler traps --- Disallow: *?s=* Disallow: *?p=* Disallow: *&p=* Disallow: *&preview=* # ============================================================================= # Search, social and AI crawlers are NOT listed individually any more. # # Every one of them previously had a "User-agent: X / Allow: /" group, which # exempted it from all of the above. They now inherit the rules in "*", which # still leaves the whole catalogue, every category and the blog crawlable — only # parameter URLs, the cart, the checkout and account pages are off limits. That # is the intended behaviour for Google, Bing, Yandex, Baidu, the social preview # fetchers and the AI search/training crawlers alike. # ============================================================================= # --- Aggressive crawlers: blocked outright --- User-agent: Bytespider Disallow: / User-agent: CCBot Disallow: /