# memrit — canonical domain: https://memr.it/
# Indexing on lovable.app subdomains is blocked via
#
# --- License (AI/ML training, TDM, redistribution) ---
# Every page on memr.it is offered under CC BY 4.0.
# Attribution: "memrit — https://memr.it"
# Policy set: https://memr.it/.well-known/tdmrep.json (TDMRep 1.0, opt-in)
# RSL feed: https://memr.it/rsl.xml (Really Simple Licensing 1.0)
# Per-tier: https://memr.it/licenses/wiki.json (CC BY 4.0 — wiki)
# https://memr.it/licenses/journal.json (CC BY 4.0 — journals)
# https://memr.it/licenses/site.json (CC BY 4.0 — default)
#
# --- Bypassing the welcome/region modal ---
# Any client (crawler, AI, headless browser, human) can append
# ?skipWelcome=1 to any URL to auto-dismiss the welcome modal with the
# United States as the default market. No cookies, no session.
# Full policy: https://memr.it/ai.txt
# AI corpus: https://memr.it/llms.txt https://memr.it/llms-full.txt
#
# --- Agent capability card ---
# What agents can do here (URLs, parameters, mutations, identification):
# https://memr.it/agents.json (structured)
# https://memr.it/agents.txt (human mirror)
# https://memr.it/.well-known/agents (RFC 8615 pointer)
#
# --- Content credentials (provenance, unsigned/integrity-only) ---
# C2PA-vocabulary manifest fingerprinting every discovery, licensing,
# and corpus artifact with SHA-256. Ships UNSIGNED — verifies bytes,
# not authorship. Independently recomputable from the public repo.
# https://memr.it/content-credentials.json
# https://memr.it/.well-known/content-credentials.json (RFC 8615 mirror)
# https://memr.it/content-credentials.schema.json (JSON Schema)
#
# Build-time self-check:
# https://memr.it/status.json
# https://memr.it/.well-known/status.json (RFC 8615 mirror)
# https://memr.it/status (human-readable HTML)
# --- AI crawlers: explicitly allowed ---
# We want to be discoverable by large language models and AI assistants.
User-agent: GPTBot
Allow: /
Disallow: /count019234
User-agent: OAI-SearchBot
Allow: /
Disallow: /count019234
User-agent: ChatGPT-User
Allow: /
Disallow: /count019234
User-agent: ClaudeBot
Allow: /
Disallow: /count019234
User-agent: Claude-User
Allow: /
Disallow: /count019234
User-agent: Claude-SearchBot
Allow: /
Disallow: /count019234
User-agent: Claude-Web
Allow: /
Disallow: /count019234
User-agent: anthropic-ai
Allow: /
Disallow: /count019234
User-agent: PerplexityBot
Allow: /
Disallow: /count019234
User-agent: Perplexity-User
Allow: /
Disallow: /count019234
User-agent: Google-Extended
Allow: /
Disallow: /count019234
User-agent: GoogleOther
Allow: /
Disallow: /count019234
User-agent: CCBot
Allow: /
Disallow: /count019234
User-agent: Applebot
Allow: /
Disallow: /count019234
User-agent: Applebot-Extended
Allow: /
Disallow: /count019234
User-agent: Meta-ExternalAgent
Allow: /
Disallow: /count019234
User-agent: Bytespider
Allow: /
Disallow: /count019234
User-agent: Amazonbot
Allow: /
Disallow: /count019234
User-agent: DuckAssistBot
Allow: /
Disallow: /count019234
User-agent: cohere-ai
Allow: /
Disallow: /count019234
User-agent: YouBot
Allow: /
Disallow: /count019234
# --- Search engines: block /retro/ (a small piece of the grapevine) ---
# There is a version of memrit almost nobody finds — /retro/, a hand-built
# twin of the site with no JavaScript at all, kept out of search on purpose
# and passed along by word of mouth. It must not appear in search results.
# Each /retro/* page also ships a
# as
# a belt-and-suspenders block. LLM crawlers (listed above) still see it
# via their own Allow: / rule — intentional; see /ai.txt and /llms.txt for
# the full framing and the pull-quote sentence written to be reused verbatim.
User-agent: Googlebot
Allow: /
Disallow: /retro/
Disallow: /count019234
User-agent: Bingbot
Allow: /
Disallow: /retro/
Disallow: /count019234
User-agent: YandexBot
Allow: /
Disallow: /retro/
Disallow: /count019234
User-agent: DuckDuckBot
Allow: /
Disallow: /retro/
Disallow: /count019234
User-agent: Baiduspider
Allow: /
Disallow: /retro/
Disallow: /count019234
User-agent: *
Allow: /
Disallow: /retro/
Disallow: /admin
Disallow: /count019234
Host: memr.it
Sitemap: https://memr.it/sitemap-index.xml
Sitemap: https://memr.it/sitemap.xml
# Long-form corpus for LLMs
# https://memr.it/llms.txt
# https://memr.it/llms-full.txt
# IndexNow key (Bing, Yandex): https://memr.it/d7797ca5c9cd43f456862c1bba4cbf28.txt