# NOTE: robots.txt has no "include". A crawler obeys exactly ONE group — the one # matching its user-agent — and ignores every other group, including "*". So any # named group below must repeat the full disallow list or it gets a free pass. # This is what previously let GPTBot crawl /providers/*/claim and /_serverFn/. User-agent: * # Allow public content Allow: /providers/ Allow: /specialty/ Allow: /specialties/ Allow: /search Allow: /browse Allow: /blog/ Allow: /conditions/ Allow: /insurance/ Allow: /locations/ Allow: /services/ Allow: /compare/ Allow: /about Allow: /press/ Allow: /alternatives/ # Block auth + account flows Disallow: /auth/ Disallow: /login Disallow: /admin-login Disallow: /post-login Disallow: /signup Disallow: /claim Disallow: /review-heallexa Disallow: /r/ Disallow: /editor # Claim pages live under /providers//claim. Written out in full because # longest-match wins over "Allow: /providers/" — "/*/claim" would lose. Disallow: /providers/*/claim Disallow: /provider/*/claim # Block dashboard + admin (authenticated only) Disallow: /dashboard Disallow: /dashboard/ # Block internal + API routes Disallow: /api/ # Server-function RPC endpoints — never cacheable, every hit is a Lambda + DB # round trip. Nothing here is useful to a crawler. Disallow: /_serverFn/ # Router artifact producing 404s under provider pages Disallow: /providers/*/null # Block faceted search — prevents infinite crawl space from query param combinations Disallow: /search? Disallow: /*?q=* Disallow: /*?specialty=* Disallow: /*?insurance=* Disallow: /*?location=* # Block TanStack Router parameter artifact URLs (URL-encoded form: $ → %24) Disallow: /*%24* # Block utility files Disallow: /html/ Disallow: /sw.js Disallow: /apple-app-site-association # ── AI crawlers ────────────────────────────────────────────────────────────── # Strategy: allow AI search engines that surface results to users (AEO/GEO # visibility benefit), block pure training-data scrapers. # # Each group repeats the disallow list — see the note at the top of this file. User-agent: OAI-SearchBot User-agent: Google-Extended User-agent: PerplexityBot User-agent: anthropic-ai User-agent: ClaudeBot User-agent: Claude-SearchBot Allow: /providers/ Allow: /specialty/ Allow: /specialties/ Allow: /blog/ Allow: /conditions/ Allow: /insurance/ Allow: /locations/ Allow: /services/ Allow: /compare/ Allow: /about Allow: /press/ Allow: /alternatives/ Disallow: /auth/ Disallow: /login Disallow: /admin-login Disallow: /post-login Disallow: /signup Disallow: /claim Disallow: /providers/*/claim Disallow: /provider/*/claim Disallow: /providers/*/null Disallow: /review-heallexa Disallow: /r/ Disallow: /editor Disallow: /dashboard Disallow: /dashboard/ Disallow: /api/ Disallow: /_serverFn/ Disallow: /search Disallow: /search? Disallow: /*?q=* Disallow: /*?specialty=* Disallow: /*?insurance=* Disallow: /*?location=* Disallow: /*%24* Disallow: /html/ Disallow: /sw.js Disallow: /apple-app-site-association Crawl-delay: 5 # Block — pure training-data scrapers and SEO-tool crawlers with no answer-engine # benefit. meta-externalagent is Meta's crawler; FacebookBot is a different agent # and blocking it did nothing to stop this traffic. # # GPTBot is OpenAI's bulk crawler and was 49% of all traffic — a full sweep of # 6.18M provider URLs. OAI-SearchBot (above) is the agent that actually serves # ChatGPT Search results, so blocking GPTBot drops the crawl load while keeping # the answer-engine visibility. User-agent: GPTBot Disallow: / User-agent: ChatGPT-User Disallow: / User-agent: CCBot Disallow: / User-agent: FacebookBot Disallow: / User-agent: meta-externalagent Disallow: / User-agent: meta-externalfetcher Disallow: / User-agent: Bytespider Disallow: / User-agent: Amazonbot Disallow: / User-agent: DataForSeoBot Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: Baiduspider Disallow: / User-agent: Baiduspider-render Disallow: / # ── Aggressive SEO crawlers — rate-limit to protect Lambda concurrency ─────── # Kept crawlable so we can still audit our own SEO, but they get the full # disallow list — a Crawl-delay-only group is still a full-site pass. User-agent: AhrefsBot User-agent: SemrushBot User-agent: MJ12bot Disallow: /auth/ Disallow: /login Disallow: /admin-login Disallow: /post-login Disallow: /signup Disallow: /claim Disallow: /providers/*/claim Disallow: /provider/*/claim Disallow: /providers/*/null Disallow: /review-heallexa Disallow: /r/ Disallow: /editor Disallow: /dashboard Disallow: /dashboard/ Disallow: /api/ Disallow: /_serverFn/ Disallow: /search Disallow: /search? Disallow: /*?q=* Disallow: /*?specialty=* Disallow: /*?insurance=* Disallow: /*?location=* Disallow: /*%24* Disallow: /html/ Disallow: /sw.js Disallow: /apple-app-site-association Crawl-delay: 10 # Preferred domain for Yandex (non-www canonical) Host: heallexa.com Sitemap: https://heallexa.com/sitemap.xml