Sitemap: https://tomezone.com/sitemap.xml # TomeZone - crawler rules # # robots.txt is a REQUEST, not a control. Well-behaved crawlers obey it; a scraper # that means harm ignores it entirely. The rate limits on LibraryController are what # actually enforce this - see the "catalog" policy in Program.cs. This file's job is # to stop the polite majority from walking 72,000 titles by accident, and to keep # private pages out of search results. # ---------------------------------------------------------------------------- # Everyone # ---------------------------------------------------------------------------- User-agent: * # Signed-in areas. Nothing here belongs in a search index, and some of it is # personal - scholars' names, reading history, saved books. # # Every address the site emits is lowercase (Program.cs: LowercaseUrls), and robots # matching is case-sensitive, so the lowercase spelling is the one that matters. # The capitalised twins stay because old capitalised links still resolve. Disallow: /admin Disallow: /account Disallow: /study Disallow: /scholars Disallow: /students Disallow: /progress Disallow: /curriculum Allow: /reading-list/import/guide Disallow: /reading-list Disallow: /bookmarks Disallow: /support/thread Disallow: /offline Disallow: /billing Disallow: /Admin Disallow: /Account Disallow: /Study Disallow: /Scholars Disallow: /Students Disallow: /Progress Disallow: /Curriculum Disallow: /Reading-List Disallow: /Bookmarks Disallow: /Support/Thread Disallow: /Offline Disallow: /Billing # Fetch endpoints, not pages. Crawling these pulls whole books and audio files # through the server for no benefit to anyone. Disallow: /library/proxybook Disallow: /library/proxycover Disallow: /library/loadbook Disallow: /Library/ProxyBook Disallow: /Library/ProxyCover Disallow: /Library/LoadBook # Paged and filtered listings are crawlable: every listing link carries pageSize=, # so a rule on it blocked page 2 onward of every letter and category, while the # sitemap and the canonical tags invite the crawler in. The "catalog" rate limiter # (Program.cs) is what stops a sweep, not this file. # Be unhurried about it. Crawl-delay: 5 # ---------------------------------------------------------------------------- # AI training crawlers # ---------------------------------------------------------------------------- # The books are public domain and free to anyone. The catalogue - the matching, # normalising and curating that turned scattered sources into one usable library - # is the work, and there is no reason to hand it over as training data. # # Note on Google: "Google-Extended" governs AI training ONLY. It does NOT affect # Googlebot or search ranking, so this costs nothing in discoverability. The same # split applies to Applebot-Extended. # # Delete any line here to allow that crawler again. User-agent: GPTBot Disallow: / User-agent: OAI-SearchBot Disallow: / User-agent: ChatGPT-User Disallow: / User-agent: ClaudeBot Disallow: / User-agent: anthropic-ai Disallow: / User-agent: Claude-Web Disallow: / User-agent: Google-Extended Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: CCBot Disallow: / User-agent: PerplexityBot Disallow: / User-agent: Bytespider Disallow: / User-agent: Amazonbot Disallow: / User-agent: meta-externalagent Disallow: / User-agent: FacebookBot Disallow: / User-agent: Diffbot Disallow: / User-agent: cohere-ai Disallow: / User-agent: Omgilibot Disallow: / # The rest of the known training and bulk-scraping crawlers, added 2026-09-08 (the # owner's call: "any others you know of"). Search engines are NOT here on purpose. User-agent: Perplexity-User Disallow: / User-agent: Meta-ExternalFetcher Disallow: / User-agent: Google-CloudVertexBot Disallow: / User-agent: MistralAI-User Disallow: / User-agent: DuckAssistBot Disallow: / User-agent: YouBot Disallow: / User-agent: AI2Bot Disallow: / User-agent: Ai2Bot-Dolma Disallow: / User-agent: ImagesiftBot Disallow: / User-agent: Timpibot Disallow: / User-agent: Webzio-Extended Disallow: / User-agent: PanguBot Disallow: / User-agent: PetalBot Disallow: / User-agent: iaskspider/2.0 Disallow: / User-agent: Kangaroo Bot Disallow: / User-agent: Sidetrade indexer bot Disallow: / User-agent: img2dataset Disallow: / User-agent: magpie-crawler Disallow: / User-agent: Crawlspace Disallow: / User-agent: Brightbot 1.0 Disallow: / User-agent: VelenPublicWebCrawler Disallow: / User-agent: TikTokSpider Disallow: / User-agent: ISSCyberRiskCrawler Disallow: / User-agent: Cotoyogi Disallow: / User-agent: Factset_spyderbot Disallow: / User-agent: FirecrawlAgent Disallow: / User-agent: NovaAct Disallow: / User-agent: Operator Disallow: / User-agent: QualifiedBot Disallow: / User-agent: Scrapy Disallow: / User-agent: SemrushBot-OCOB Disallow: / User-agent: SemrushBot-SWA Disallow: / User-agent: Andibot Disallow: / User-agent: Devin Disallow: / User-agent: aiHitBot Disallow: / User-agent: bedrockbot Disallow: / User-agent: omgili Disallow: / User-agent: Poseidon Research Crawler Disallow: / User-agent: TerraCotta Disallow: / User-agent: Thinkbot Disallow: / User-agent: wpbot Disallow: / User-agent: YandexAdditional Disallow: /