# UzExam.uz — robots.txt # Honest User-Agent identification: we are UzExam, an Uzbek edu-tech platform. # Scraper bots that ignore robots.txt are stopped by ContentRateLimitMiddleware. # # ⛔ P1-1 (2026-09-05): EVERY named User-agent group below repeats the SAME # rule block (source: templates/seo/_robots_rules.txt). RFC 9309 says a crawler # obeys only the most specific matching group and ignores `*` entirely, so a # named group with a lone `Allow: /` used to hand Claude-SearchBot / # ChatGPT-User / PerplexityBot / OAI-SearchBot / Applebot / YandexBot / # AhrefsSiteAudit full access to /auth/, /r/, /go/, /profile/ (observed in prod: # 1 253 GET /auth/login/ and 2 019 /u/N/ from one such bot). Never add a named # group with hand-written rules — always include the shared block, and edit the # rules only in that one file so the groups cannot drift apart again. # ── AI TRAINING crawlers — BLOCKED ──────────────────────────────────────── # Policy (2026-05-29): we WANT AI assistants to find + cite us (drives # discovery), but we do NOT want our 38k+ Q'lar harvested into a model's # training corpus for free. So: training/dataset crawlers are blocked here, # answer/search bots are explicitly allowed in the next block. User-agent: GPTBot Disallow: / User-agent: ClaudeBot Disallow: / User-agent: anthropic-ai Disallow: / User-agent: Google-Extended Disallow: / User-agent: Bytespider Disallow: / User-agent: CCBot Disallow: / User-agent: Applebot-Extended Disallow: / User-agent: Meta-ExternalAgent Disallow: / User-agent: meta-externalagent Disallow: / User-agent: Amazonbot Disallow: / # ── AI ANSWER / SEARCH bots — ALLOWED (same rules as everyone else) ──────── # These fetch a page to answer a live user question (and cite the source) or # to build a search index — NOT to train a base model. Letting them in is how # we get surfaced in ChatGPT, Perplexity, Claude, and Apple/Siri answers. # One group, many User-agent lines (RFC 9309 §2.2.1) — so the rule block below # applies to all of them and cannot drift per bot. User-agent: OAI-SearchBot User-agent: ChatGPT-User User-agent: PerplexityBot User-agent: Perplexity-User User-agent: Claude-User User-agent: Claude-SearchBot User-agent: Applebot # Shared rule block — identical in every group (see the file header). Allow: / Allow: /c/ Allow: /qollanma/ Allow: /rankings/ Allow: /privacy/ Allow: /reklama/ Allow: /terms/ Allow: /apps/ Allow: /sat/ Allow: /drills/ Allow: /vocab/ Allow: /speaking/ Allow: /universitetlar/ Allow: /sertifikat/ # User-specific dynamic paths — nothing for search engines here. # /sertifikat/urinish|mening = to'langan shaxsiy sahifalar; /verify/ = # sertifikat egasining ism/natijasi (kod bilganga ochiq, lekin qidiruvga # indekslanmasin — thin content + shaxsiy ma'lumot himoyasi). Disallow: /sertifikat/urinish/ Disallow: /sertifikat/mening/ Disallow: /verify/ Disallow: /auth/ Disallow: /login/ Disallow: /webapp/ Disallow: /api/ Disallow: /contribute/ Disallow: /jk-panel/ Disallow: /profile/ Disallow: /reports/ Disallow: /go/ Disallow: /r/ # Public CSS/JS must remain crawlable so search engines can render pages. # Mini-app runner / attempt / session pages — user-specific state, no SEO value. Disallow: /sat/a/ Disallow: /sat/t/*/start/ Disallow: /drills/s/ Disallow: /drills/start/ Disallow: /vocab/d/*/study/ Disallow: /vocab/d/*/review/ Disallow: /speaking/q/*/note/ Disallow: /essay-checker/attempts/ Disallow: /t/ # Tenant subdomain URLs — owned by orgs, not part of UzExam apex SEO surface. # Each tenant nginx serves its own robots if needed. Disallow: /live/ Disallow: /assignments/ Disallow: /studio/ Disallow: /invites/ Disallow: /groups/ Disallow: /students/ Disallow: /teachers/ Disallow: /my-results/ Disallow: /billing/ Disallow: /settings/ Crawl-delay: 2 # ── Ahrefs SITE AUDIT — o'z auditimiz (AWT, bepul) — RUXSAT ─────────────── # AhrefsBot (competitor research) pastda blokda qoladi; AhrefsSiteAudit esa # faqat o'zimiz tasdiqlagan proyekt buyurtmasi bilan crawl qiladi (2026-08-04: # "crawl error" mailining sababi shu blok edi). User-agent: AhrefsSiteAudit # Shared rule block — identical in every group (see the file header). Allow: / Allow: /c/ Allow: /qollanma/ Allow: /rankings/ Allow: /privacy/ Allow: /reklama/ Allow: /terms/ Allow: /apps/ Allow: /sat/ Allow: /drills/ Allow: /vocab/ Allow: /speaking/ Allow: /universitetlar/ Allow: /sertifikat/ # User-specific dynamic paths — nothing for search engines here. # /sertifikat/urinish|mening = to'langan shaxsiy sahifalar; /verify/ = # sertifikat egasining ism/natijasi (kod bilganga ochiq, lekin qidiruvga # indekslanmasin — thin content + shaxsiy ma'lumot himoyasi). Disallow: /sertifikat/urinish/ Disallow: /sertifikat/mening/ Disallow: /verify/ Disallow: /auth/ Disallow: /login/ Disallow: /webapp/ Disallow: /api/ Disallow: /contribute/ Disallow: /jk-panel/ Disallow: /profile/ Disallow: /reports/ Disallow: /go/ Disallow: /r/ # Public CSS/JS must remain crawlable so search engines can render pages. # Mini-app runner / attempt / session pages — user-specific state, no SEO value. Disallow: /sat/a/ Disallow: /sat/t/*/start/ Disallow: /drills/s/ Disallow: /drills/start/ Disallow: /vocab/d/*/study/ Disallow: /vocab/d/*/review/ Disallow: /speaking/q/*/note/ Disallow: /essay-checker/attempts/ Disallow: /t/ # Tenant subdomain URLs — owned by orgs, not part of UzExam apex SEO surface. # Each tenant nginx serves its own robots if needed. Disallow: /live/ Disallow: /assignments/ Disallow: /studio/ Disallow: /invites/ Disallow: /groups/ Disallow: /students/ Disallow: /teachers/ Disallow: /my-results/ Disallow: /billing/ Disallow: /settings/ Crawl-delay: 2 # ── SEO competitor crawlers — BLOCKED ───────────────────────────────────── # These don't drive traffic, only competitor research; cost us bandwidth. User-agent: AhrefsBot Disallow: / User-agent: SemrushBot Disallow: / User-agent: MJ12bot Disallow: / User-agent: DotBot Disallow: / # SE Ranking backlink-indeksi (2026-09-10: 5.9.120.8 dan soatiga 1 000+ `/q/N/` # savol-sahifasi, harvest-detektor 500-chegarani urdi) — bizga trafik bermaydi. User-agent: SERankingBacklinksBot Disallow: / # Awario/WebMeUp brend-monitoring (2026-09-11: 65.109.20.216 dan `/q/N/` ni # sitemap-tartibida 30/daq olib harvest-detektor 500-chegarasini urdi; robots.txt # ni har 2 daqiqada qayta o'qiydi — blok darhol kuchga kiradi) — trafik bermaydi. User-agent: AwarioBot Disallow: / # ── Yandex — explicit allow + Host directive ────────────────────────────── # Uzbek user pool uses Yandex heavily. Host directive is Yandex-specific # (canonical mirror declaration). User-agent: YandexBot User-agent: Yandex Host: uzexam.uz # Clean-param — tracking/UTM/referral parametrli URL'larni bitta canonical'ga # yig'adi (duplicate-content + crawl-budget isrofini kamaytiradi). /go/ marketing # redirectlari, UTM, referral va analitika parametrlari shu yerda. Clean-param: utm_source&utm_medium&utm_campaign&utm_content&utm_term&ref&fbclid&gclid&yclid&ysclid&source&start # Shared rule block — identical in every group (see the file header). Allow: / Allow: /c/ Allow: /qollanma/ Allow: /rankings/ Allow: /privacy/ Allow: /reklama/ Allow: /terms/ Allow: /apps/ Allow: /sat/ Allow: /drills/ Allow: /vocab/ Allow: /speaking/ Allow: /universitetlar/ Allow: /sertifikat/ # User-specific dynamic paths — nothing for search engines here. # /sertifikat/urinish|mening = to'langan shaxsiy sahifalar; /verify/ = # sertifikat egasining ism/natijasi (kod bilganga ochiq, lekin qidiruvga # indekslanmasin — thin content + shaxsiy ma'lumot himoyasi). Disallow: /sertifikat/urinish/ Disallow: /sertifikat/mening/ Disallow: /verify/ Disallow: /auth/ Disallow: /login/ Disallow: /webapp/ Disallow: /api/ Disallow: /contribute/ Disallow: /jk-panel/ Disallow: /profile/ Disallow: /reports/ Disallow: /go/ Disallow: /r/ # Public CSS/JS must remain crawlable so search engines can render pages. # Mini-app runner / attempt / session pages — user-specific state, no SEO value. Disallow: /sat/a/ Disallow: /sat/t/*/start/ Disallow: /drills/s/ Disallow: /drills/start/ Disallow: /vocab/d/*/study/ Disallow: /vocab/d/*/review/ Disallow: /speaking/q/*/note/ Disallow: /essay-checker/attempts/ Disallow: /t/ # Tenant subdomain URLs — owned by orgs, not part of UzExam apex SEO surface. # Each tenant nginx serves its own robots if needed. Disallow: /live/ Disallow: /assignments/ Disallow: /studio/ Disallow: /invites/ Disallow: /groups/ Disallow: /students/ Disallow: /teachers/ Disallow: /my-results/ Disallow: /billing/ Disallow: /settings/ Crawl-delay: 2 # ── Default policy (Google, Bing, generic crawlers) ─────────────────────── User-agent: * # Content Signals (contentsignals.org / draft-romm-aipref-contentsignals) — # machine-readable AI-usage policy. Mirrors the per-bot rules above: # search=yes → search indexing OK # ai-input=yes → AI assistants may read + cite us in live answers (discovery) # ai-train=no → our 38k+ Q'lar must NOT be harvested into a training corpus Content-Signal: search=yes, ai-input=yes, ai-train=no # Shared rule block — identical in every group (see the file header). Allow: / Allow: /c/ Allow: /qollanma/ Allow: /rankings/ Allow: /privacy/ Allow: /reklama/ Allow: /terms/ Allow: /apps/ Allow: /sat/ Allow: /drills/ Allow: /vocab/ Allow: /speaking/ Allow: /universitetlar/ Allow: /sertifikat/ # User-specific dynamic paths — nothing for search engines here. # /sertifikat/urinish|mening = to'langan shaxsiy sahifalar; /verify/ = # sertifikat egasining ism/natijasi (kod bilganga ochiq, lekin qidiruvga # indekslanmasin — thin content + shaxsiy ma'lumot himoyasi). Disallow: /sertifikat/urinish/ Disallow: /sertifikat/mening/ Disallow: /verify/ Disallow: /auth/ Disallow: /login/ Disallow: /webapp/ Disallow: /api/ Disallow: /contribute/ Disallow: /jk-panel/ Disallow: /profile/ Disallow: /reports/ Disallow: /go/ Disallow: /r/ # Public CSS/JS must remain crawlable so search engines can render pages. # Mini-app runner / attempt / session pages — user-specific state, no SEO value. Disallow: /sat/a/ Disallow: /sat/t/*/start/ Disallow: /drills/s/ Disallow: /drills/start/ Disallow: /vocab/d/*/study/ Disallow: /vocab/d/*/review/ Disallow: /speaking/q/*/note/ Disallow: /essay-checker/attempts/ Disallow: /t/ # Tenant subdomain URLs — owned by orgs, not part of UzExam apex SEO surface. # Each tenant nginx serves its own robots if needed. Disallow: /live/ Disallow: /assignments/ Disallow: /studio/ Disallow: /invites/ Disallow: /groups/ Disallow: /students/ Disallow: /teachers/ Disallow: /my-results/ Disallow: /billing/ Disallow: /settings/ Crawl-delay: 2 Sitemap: https://uzexam.uz/sitemap.xml