User-agent: *
Allow: /
Allow: /browse
Allow: /search
Disallow: /api/
# /api/ is fully disallowed with no Allow exceptions (removed 2026-08-30). Every code
# page is fully server-rendered -- bot and human HTML are identical and contain all
# relationships as plain links -- so no crawler needs any API response to understand a
# page. The old per-endpoint Allow list predated full SSR and caused Googlebot to spend
# a large share of its crawl budget re-fetching /api/ URLs its renderer discovered.
# The endpoints also send X-Robots-Tag: noindex. NOTE: robots.txt has no rule
# inheritance -- each named UA group repeats its full rule set.
# Search Result Hygiene - Prevent infinite crawl variations
# /browse?q=... stays disallowed: it canonicalizes to the bare /browse page, so crawling the
# permutations only burns crawl budget. The previous "Allow: /browse?q=*" line actively invited
# that -- robots.txt resolves conflicts by longest match, so it overrode the Disallow above it,
# and Search Console duly reported a stream of /browse?q=... URLs as "Crawled - currently not
# indexed". The bare /browse and /search pages stay crawlable via the Allow rules further up.
#
# /search?* is deliberately NOT disallowed any more. A results page had already been indexed
# (3 impressions, avg position 3.67), and the comment above used to claim these "can never be
# indexed" -- they could, because the page was self-canonical and carried no robots directive.
# It now serves , and a crawler has to be able to
# FETCH a page to see that directive. Disallowing the URL would hide the noindex and leave the
# already-indexed copies in place indefinitely. noindex+follow is the correct pair here: the
# results page stays out of the index while the links it makes to real code pages still count.
Disallow: /browse?*
# CPT is AMA-licensed and not published here. Every /cpt URL answers 410 Gone, but Google only
# learns that by fetching one, and hallucinated CPT URLs arrive faster than they can be retired
# (Search Console listed ~60 of them on 2026-08-16, most invented rather than ever linked).
# Disallowing the prefix stops the crawl spend at the door; the 410 stays for anything already
# known to Google, since a disallowed URL never reveals its status.
Disallow: /cpt
# Coder tools: the bare tool pages are real landing pages targeting high-intent queries
# ("can these ICD-10 codes be billed together", "compare ICD-10 codes"), so they MUST stay
# crawlable. Only their ?codes=... permutations are disallowed -- those are an unbounded
# crawl space (every code combination is a distinct URL) and all canonicalize to the bare
# page anyway. Note: "Disallow: /compare" previously blocked the tool page itself while
# sitemap-main.xml simultaneously submitted it -- a direct contradiction that made it
# impossible for /compare to ever rank.
#
# /claim-check?* IS NO LONGER DISALLOWED (2026-09-24), for the reason already worked through
# for /search?* below: a crawler has to be able to FETCH a URL to see a robots directive.
# Bing Webmaster Tools held 16 /claim-check?codes=... URLs as URL-only entries it could
# neither resolve to the bare page nor drop -- reported as crawl errors against the site for
# as long as the Disallow stood. Those URLs now serve alongside their existing canonical to /claim-check (server.ts, QUERY_IS_WORKING_
# STATE), so letting crawlers reach them is what clears them. /compare?* keeps its Disallow:
# nothing is stuck behind it.
Disallow: /compare?*
# Restricted areas - Conserve crawl budget
# (/registry-export is a scraper trap: never linked, never crawlable; do not fetch)
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /package.json
Disallow: /package-lock.json
Disallow: /tsconfig.json
Disallow: /vite.config.ts
Disallow: /.git*
# AI assistant crawlers: allowed since 2026-08-09 so Claude, ChatGPT and Perplexity can read
# and cite this content -- a real discovery channel, and one this site's official-data-only
# content is well suited to be cited accurately from.
#
# UPDATED 2026-08-23: ClaudeBot, GPTBot and Meta's meta-externalagent now have their own groups
# below (see the Applebot group's Crawl-delay comment for why) -- confirmed via Cloudflare
# analytics to be hitting the same host-level concurrent-connection throttle Bingbot was, at
# comparable or higher rates (ClaudeBot 56%, Applebot 42%, meta-externalagent 57% of requests
# 429'd/403'd over a 1h sample). CCSR, anthropic-ai, Claude-Web, ChatGPT-User and PerplexityBot
# still fall through to `User-agent: *`, which allows them -- no evidence yet ties them to the
# same problem specifically, so they are left there rather than guessing at groups for crawlers
# this site has no measurement on.
# Allow specialized hybrid search crawlers
User-agent: OAI-SearchBot
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
# Allow friendly consumer search engine bots with maximum priority and zero intentional delay
User-agent: Googlebot
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
#
# /cpt was TEMPORARILY OPEN for Googlebot (2026-09-01 to 2026-09-16, closed early at the owner's
# request; ~2026-09-29 was the plan): stale /cpt and /cptcode entries persisted in Google's index
# from before the CPT removal, and a disallowed URL can never be re-fetched to reveal its 410.
# Every such URL answers 410 Gone. The temporary prefix Removals for /cpt/ and /cptcode/ filed in
# GSC keep any stragglers out of results. Restored so hallucinated CPT URLs stay cheap again.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: bingbot
# NO Crawl-delay for bingbot (REMOVED 2026-09-26). History below, kept because it explains why
# the line was there and what to measure before ever putting it back.
#
# Why removed: Bing Webmaster Tools' URL Inspection, on a sampled page that was "Discovered but
# not crawled" (as 4 of 5 sampled sitemap pages were), lists as its FIRST recommendation "there is
# no crawl-delay in your robots.txt". While a Crawl-delay is present Bing also ignores the hourly
# crawl-rate schedule in Webmaster Tools > Crawl Control, so that is the lever to use instead if
# Bing's rate ever needs capping. And the prediction below did not come true: over the 24h to
# 2026-09-26 Bingbot made 2,737 requests (FEWER than the 3,269 before the delay) with 676 (25%)
# still 429'd -- ~2.7K/day against ~108K sitemap URLs is a ~40-day pass. The drop in the 429 share
# since 08-17 tracks the LiteSpeed denylist narrowing on 09-24 (see docs/HOSTINGER-429-TICKET.md),
# not the delay. Watch `bash scripts/cloudflare/cf-crawlers.sh 1` after this deploys: Bingbot's
# request count should rise; if its 429 share climbs well past ~25%, cap it in Crawl Control.
#
# ORIGINAL RATIONALE (2026-08-17):
# Measured over 23h on 2026-08-17 via Cloudflare analytics: Bingbot made 3,269 requests and
# 2,539 of them (78%) were answered 429 by the HOST, before Express was reached. Googlebot over
# the same window made 1,791 requests and received ZERO 429s -- so this is not a site-wide rate
# problem, it is Bingbot specifically.
#
# The host's limit is on CONCURRENT connections, not request rate (proven separately: 15
# simultaneous requests = 15/15 429, the same 15 sequential mostly 200). Crawl-delay makes Bing
# serialise, which is exactly the shape that limit wants.
#
# 2 seconds still permits ~43,000 requests/day, an order of magnitude above the ~3,300 Bing
# currently attempts, so this RAISES successful crawling by removing the 429s rather than
# throttling anything real.
#
# Google deliberately ignores Crawl-delay (it uses the Search Console crawl-rate setting), so
# this cannot slow Googlebot. It is placed inside the bingbot group because this robots.txt has
# NO rule inheritance -- directives must be repeated in each named group.
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: Applebot
# Crawl-delay, for Applebot. Same rationale as bingbot's below: measured via Cloudflare
# analytics 2026-08-23, Applebot got 429'd on 41.7% of a 930-request, 1h sample -- all from
# Hostinger's LiteSpeed layer (0-byte bodies, no app RateLimit-* headers, confirmed by
# Hostinger's own support as a host-level concurrent-connection limit with no per-domain
# exemption available). Crawl-delay only paces a crawler's own SEQUENTIAL requests and is
# voluntary; it will not stop Applebot from firing several requests at once within one crawl
# pass, which is the shape that actually trips the limit (proven: 15 simultaneous requests from
# one UA = 15/15 429, the same 15 sequential mostly succeed). Kept in because it costs nothing
# and may reduce sustained load even if it can't prevent a burst -- not a fix, a floor.
#
# RAISED 2 -> 10 on 2026-09-26, by the owner's choice, to leave the host's shared connection limit
# for Bingbot. Over the 24h to 09-26 Applebot made 5,625 requests (33% 429'd) against Bingbot's
# 2,737, and Bing (which also grounds several AI assistants) had indexed only ~3.3K of ~108K
# sitemap URLs. 10s still permits ~8,600 requests/day, above Applebot's volume.
Crawl-delay: 10
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: ClaudeBot
# Crawl-delay, for ClaudeBot. Same rationale as bingbot's and Applebot's above --
# measured via Cloudflare analytics 2026-08-23: 56.0% of a 7,526-request, 1h sample got
# 429'd, all from Hostinger's LiteSpeed layer, not this app. Not a fix (voluntary, paces
# only sequential requests, does not stop a burst within one crawl pass), but costs
# nothing to leave in.
#
# RAISED 2 -> 10 on 2026-09-26, by the owner's choice, to leave the host's shared connection limit
# for Bingbot: over the 24h to 09-26 ClaudeBot made 15,946 requests (28% 429'd, 13,978 reaching
# origin) -- nearly 6x Bingbot's 2,737 -- against the same limit. At 10s ClaudeBot can still make
# ~8,600 requests/day, enough to cover the site's core pages; it just stops crowding Bing out.
# The 2026-08-09 decision to ALLOW AI crawlers stands -- this paces ClaudeBot, it does not block it.
Crawl-delay: 10
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: ExaSearchBot
# Crawl-delay, for Exa's crawler (crawler.exa.ai), added 2026-10-01. It went from ~1k origin
# requests/day to 53,408 on 09-30 and 3,848 in one hour on 10-01 -- about one a second, almost all
# ICD-10-CM code pages, while the code-page payload cache was cold after the FY2027 loads, which
# held every code page to 30-40s at origin. Same pacing as ClaudeBot and Applebot: allowed, not
# blocked (the 2026-08-09 decision to allow AI crawlers stands); at 10s Exa can still make ~8,600
# requests/day. The crawl-hygiene rules are repeated verbatim below, as in every named group.
Crawl-delay: 10
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: GPTBot
# Crawl-delay, for GPTBot. Same host-level throttle as the other crawlers in this file --
# see the ClaudeBot group's comment above for the caveats on what this can and cannot fix.
Crawl-delay: 2
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: meta-externalagent
# Crawl-delay, for Meta's crawler (facebookexternalhit / meta-externalagent -- Facebook and
# Instagram link previews). Measured via Cloudflare analytics 2026-08-23: 57% of an
# 84-request, 1h sample got 429'd and another 19% got 403'd, both from Hostinger's
# LiteSpeed layer, not this app. Same caveats as the ClaudeBot group's comment above.
Crawl-delay: 2
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: Baiduspider
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
User-agent: Yandex
Disallow: /api/
Allow: /
# Crawl-hygiene rules, repeated verbatim in every named group. robots.txt has no rule
# inheritance: a crawler obeys exactly ONE group -- its most specific User-agent match -- and
# ignores 'User-agent: *' entirely. Because Googlebot (and bingbot/Applebot/Baiduspider/Yandex/
# OAI-SearchBot) each have their own group below, every Disallow in the * group was dead code
# for precisely the crawlers that matter. That silently un-did the /browse?* fix documented
# above: the URLs Search Console was reporting as 'Crawled - currently not indexed' were never
# actually blocked for Google at all.
Disallow: /cpt
Disallow: /browse?*
Disallow: /compare?*
Disallow: /registry-export
Disallow: /api/auth/
Disallow: /api/debug/
Disallow: /api/diagnostics
Disallow: /diagnostics
Disallow: /comparison/
Disallow: /search/
Disallow: /login
Disallow: /admin/
Disallow: /auth/
Disallow: /*.json$
Disallow: /*.php$
Disallow: /*.aspx$
Disallow: /.env*
Disallow: /server.ts
Disallow: /.git*
# Disallow aggressive general SEO auditing crawlers to save budget
User-agent: PetalBot
Disallow: /
User-agent: AhrefsBot
Disallow: /
User-agent: SemrushBot
Disallow: /
User-agent: DotBot
Disallow: /
User-agent: MJ12bot
Disallow: /
User-agent: BLEXBot
Disallow: /
# Reference Sitemaps. Only the INDEX: it enumerates all ~49 children itself, and the previous
# hand-picked subset (main, chapters, labs, ndc-1/2) read as a stale partial list -- it predated
# most of the children and never gained the ICD/HCPCS/PCS/MS-DRG files that carry 99% of URLs.
Sitemap: https://medcoder.ai/sitemap.xml