# https://www.robotstxt.org/robotstxt.html
User-agent: *
# Content signals — https://contentsignals.org/
# Declares how this site's content may be used, per crawler group.
# yes = permitted, no = not permitted. Signals are a statement of preference,
# not an access control; they express permission, they do not enforce it.
Content-Signal: search=yes, ai-input=yes, ai-train=yes
Allow: /
# TanStack Router route IDs leak into the SSR dehydration payload and the client
# bundle as literal strings ("/_default", "/_default/blog/$postId", ...). Googlebot
# extracts those as URLs and crawls them, which produced ~180 phantom 404s in
# Search Console. They are layout groups, never real URLs.
# Do NOT add /assets/ here — Google needs the JS and CSS to render the pages.
Disallow: /_default
Disallow: /_account
Disallow: /_checkout
Sitemap: https://warawul.coffee/sitemap.xml