- robotsTxtUrl
- https://smartdraftboard.com/robots.txt
- exists
- true
- rawRobotsTxt
- # As a condition of accessing this website, you agree to abide by the following
# content signals:
# (a) If a Content-Signal = yes, you may collect content for the corresponding
# use.
# (b) If a Content-Signal = no, you may not collect content for the
# corresponding use.
# (c) If the website operator does not include a Content-Signal for a
# corresponding use, the website operator neither grants nor restricts
# permission via Content-Signal with respect to the corresponding use.
# The content signals and their meanings are:
# search: building a search index and providing search results (e.g., returning
# hyperlinks and short excerpts from your website's contents). Search does not
# include providing AI-generated search summaries.
# ai-input: inputting content into one or more AI models (e.g., retrieval
# augmented generation, grounding, or other real-time taking of content for
# generative AI search answers).
# ai-train: training or fine-tuning AI models.
# use: how AI systems may consume the content (immediate, reference, or full).
# ANY RESTRICTIONS EXPRESSED VIA CONTENT SIGNALS ARE EXPRESS RESERVATIONS OF
# RIGHTS UNDER ARTICLE 4 OF THE EUROPEAN UNION DIRECTIVE 2019/790 ON COPYRIGHT
# AND RELATED RIGHTS IN THE DIGITAL SINGLE MARKET.
# BEGIN Cloudflare Managed content
User-agent: *
Content-Signal: search=yes,ai-train=no,use=reference
Allow: /
User-agent: Amazonbot
Disallow: /
User-agent: Applebot-Extended
Disallow: /
User-agent: Bytespider
Disallow: /
User-agent: CCBot
Disallow: /
User-agent: ClaudeBot
Disallow: /
User-agent: CloudflareBrowserRenderingCrawler
Disallow: /
User-agent: Google-Extended
Disallow: /
User-agent: GPTBot
Disallow: /
User-agent: meta-externalagent
Disallow: /
# END Cloudflare Managed Content
User-agent: *
Allow: /
Allow: /afl-draft-rankings
Allow: /nrl-draft-rankings
Allow: /nfl-draft-rankings
Allow: /sleeper-draft-rankings
Allow: /espn-fantasy-draft-rankings
Allow: /nba-draft-rankings
Allow: /football-draft-rankings
Allow: /afl-draft-cheat-sheet
Allow: /nrl-draft-cheat-sheet
Allow: /fpl-draft-cheat-sheet
Allow: /afl-salary-cap
Allow: /nrl-salary-cap
Allow: /football-salary-cap
Allow: /salary-cap/afl/supercoach
Allow: /salary-cap/afl/fantasy
Allow: /salary-cap/nrl/supercoach
Allow: /salary-cap/football/fplclassic
Allow: /afl-in-season
Allow: /nrl-in-season
Allow: /fpl-in-season
Allow: /nfl-in-season
Allow: /matchday-live
Allow: /draft-board
Allow: /draft-board/afl
Allow: /draft-board/nrl
Allow: /draft-board/epl
Allow: /fpl-draft-waiver-picks
Allow: /fpl-captain-picks
Allow: /fpl-draft-rankings
Allow: /fpl-draft-tips
Allow: /fpl-draft-tools
Allow: /fpl-price-changes
Allow: /fpl-fixture-difficulty
Allow: /afl-supercoach-waiver-wire
Allow: /afl-fantasy-waiver-wire
Allow: /nrl-supercoach-waiver-wire
Allow: /fantasy-projections
Allow: /afl-projections
Allow: /nrl-projections
Allow: /fpl-projections
Allow: /projections/methodology
Allow: /projections/fpl-methodology
Allow: /stats/
Allow: /pro
Allow: /community
Allow: /players/
Allow: /terms
Allow: /privacy-policy
# New expansion pages
Allow: /glossary
Allow: /afl-injuries
Allow: /nrl-injuries
Allow: /afl-team-lists
Allow: /scouting/
Allow: /compare/
Allow: /teams/
Allow: /fpl-players/
Allow: /matchup/
Allow: /gameweek/
Allow: /briefing/
Allow: /rate-my-team
Allow: /creators
Allow: /fpl/transfer-momentum
Allow: /fantrax/auction
Allow: /fixture-runs/
Allow: /positions/
# FPL + Fantrax acquisition surfaces
Allow: /fantrax-draft-rankings
Allow: /fpl/
Allow: /fpl-leaderboards
Allow: /fpl-intelligence
Allow: /mock-draft/
# Embed widgets not for indexing
Disallow: /embed/
# Scraper honeypot — NOT a real page. Every honest crawler learns to avoid it
# from this line, which is the whole point: fetching it after being told not
# to is what identifies an unauthorised one. The link exists only as a hidden,
# nofollow anchor in index.html, so nothing that respects either signal ever
# arrives. Repeated in every explicit User-agent group below, because a bot
# obeys ONLY its most specific matching group — a Disallow that lives solely
# in `*` is invisible to GPTBot, ClaudeBot and the rest.
# Enforced by api-server/src/botDefense/traps.ts.
Disallow: /data-export/
Disallow: /admin
Allow: /api/public/
Disallow: /api/
# /shared/ deliberately NOT disallowed: shareShell exists to hand Twitterbot,
# facebookexternalhit, LinkedInBot and Slack per-share OG cards, and those
# unfurlers respect robots.txt — a Disallow here starved the exact audience
# the SSR was built for. Indexing is prevented where it belongs: every
# /shared/ render carries <meta name="robots" content="noindex,follow">.
Disallow: /profile
Disallow: /reset-password
Disallow: /attached_assets/
Disallow: /scripts/
Disallow: /tasks/
Disallow: /src/
Disallow: /server/
Disallow: /node_modules/
# AI crawlers — explicitly welcome. Answer-engine referrals (ChatGPT,
# Perplexity, Claude, Google AI Overviews) are an acquisition channel;
# llms.txt carries the product summary these bots should read. The
# `User-agent: *` block above already allows them — these blocks are
# declarative insurance so a future tightening of `*` can't silently
# cut off AI ingestion. Fetches are logged by aiBotClassifier.ts.
#
# These crawlers are also EXEMPT from every rate limit and abuse control the
# server applies, but the exemption is earned at the network layer, not from
# the User-Agent string below: reverse DNS (Google, Bing, Apple, Yandex,
# Baidu) or the operator's own published IP ranges (OpenAI, Perplexity).
# See api-server/src/botDefense/. If you operate a crawler and want the same
# exemption, publish your ranges or set PTR records and tell us — an
# unverifiable crawler is still served in full, just under a rate budget.
#
# Every group repeats the honeypot Disallow: a robots.txt group REPLACES the
# `*` group rather than extending it, so the line above does not reach here.
User-agent: GPTBot
Allow: /
Disallow: /data-export/
User-agent: OAI-SearchBot
Allow: /
Disallow: /data-export/
User-agent: ChatGPT-User
Allow: /
Disallow: /data-export/
User-agent: ClaudeBot
Allow: /
Disallow: /data-export/
User-agent: Claude-SearchBot
Allow: /
Disallow: /data-export/
User-agent: Claude-User
Allow: /
Disallow: /data-export/
User-agent: PerplexityBot
Allow: /
Disallow: /data-export/
User-agent: Perplexity-User
Allow: /
Disallow: /data-export/
User-agent: Google-Extended
Allow: /
Disallow: /data-export/
User-agent: Applebot-Extended
Allow: /
Disallow: /data-export/
User-agent: MistralAI-User
Allow: /
Disallow: /data-export/
# Commercial crawl-and-resell operations. These are not search engines and
# not answer engines: they copy a site wholesale and sell the result back as
# a competitive-intelligence dataset, which is the exact channel by which
# this product's rankings and projections get replicated. Refused here as a
# statement of terms, and refused by name in botDefense/clientSignals.ts
# because compliance with this file is optional and they know it.
User-agent: AhrefsBot
Disallow: /
User-agent: SemrushBot
Disallow: /
User-agent: MJ12bot
Disallow: /
User-agent: DotBot
Disallow: /
User-agent: BLEXBot
Disallow: /
User-agent: DataForSeoBot
Disallow: /
User-agent: SerpstatBot
Disallow: /
User-agent: ZoominfoBot
Disallow: /
User-agent: Barkrowler
Disallow: /
User-agent: Diffbot
Disallow: /
# Single sitemap index only. sitemap.xml is a <sitemapindex> generated by
# generateSitemapIndex() (api-server/src/sitemapGenerator.ts) that always lists
# the current sub-sitemap set. Listing sub-sitemaps individually here drifted:
# it still pointed at the legacy single-file sitemap-players.xml /
# sitemap-comparisons.xml (now split into -priority/-longtail) and omitted
# sitemap-leaders.xml + sitemap-positions.xml. Pointing only at the index means
# robots.txt can never drift from the generator again.
Sitemap: https://smartdraftboard.com/sitemap.xml