# robots.txt for tw93.fun # Every crawler is allowed. The Content-Signal lines state, per tier, what each # group of crawlers may do with the content once it has been fetched. User-agent: * Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / Disallow: /_site/ Disallow: /css/ Disallow: /js/ Disallow: /.git/ Crawl-delay: 1 # Sitemaps # sitemap.xml covers posts and pages; sitemap-ai.xml covers the machine-readable # surfaces (projects/*.md, api/*.json, llms*.txt) that Jekyll ships as static # files and therefore leaves out of the generated sitemap. Sitemap: https://tw93.fun/sitemap.xml Sitemap: https://tw93.fun/sitemap-ai.xml # Schema Map (NLWeb Schema Feeds): the structured data feeds on this site. Schemamap: https://tw93.fun/schema-map.xml # Agent entry points, listed here so a crawler that reads only robots.txt still # finds them: # https://tw93.fun/llms.txt orientation, start here # https://tw93.fun/llms-full.txt full knowledge base # https://tw93.fun/index.md homepage as markdown # https://tw93.fun/pricing.md pricing and licensing # https://tw93.fun/openapi.json OpenAPI 3.1 description # https://tw93.fun/.well-known/api-catalog RFC 9727 API catalog # https://tw93.fun/.well-known/agent.json capabilities, limits, error handling # https://tw93.fun/.well-known/ai-plugin.json plugin manifest # --------------------------------------------------------------------------- # Tier 1: search and retrieval crawlers. # These fetch a page to answer a query and link back to it. # --------------------------------------------------------------------------- User-agent: Googlebot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Bingbot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Applebot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: DuckAssistBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / # --------------------------------------------------------------------------- # Tier 2: AI assistants fetching on behalf of a user, and AI search bots. # These are the crawlers the machine-readable surfaces here are written for. # --------------------------------------------------------------------------- User-agent: ChatGPT-User Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: OAI-SearchBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Claude-User Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Claude-SearchBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Claude-Web Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: PerplexityBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Perplexity-User Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / # --------------------------------------------------------------------------- # Tier 3: model training crawlers. # Training on this content is permitted, stated here explicitly rather than # left to the wildcard group. # --------------------------------------------------------------------------- User-agent: GPTBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: ClaudeBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: anthropic-ai Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Google-Extended Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: CCBot Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Bytespider Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: / User-agent: Meta-ExternalAgent Content-Signal: search=yes, ai-input=yes, ai-train=yes Allow: /