{
  "registry": "Sources cited by the uptake manual",
  "compiled": "2026-09-18",
  "licence": "CC BY-SA 4.0",
  "note": "Every id here is referenced from the manual text as [S<id>]. Each entry was fetched and returned HTTP 200 on the compiled date; a link that dies later is a link that died later.",
  "sources": [
    {"id": 1, "title": "REST API endpoints for repository metrics — traffic", "publisher": "GitHub", "url": "https://docs.github.com/en/rest/metrics/traffic", "kind": "documentation", "used_for": "The clone and view endpoints, and the fourteen-day retention window."},
    {"id": 2, "title": "About CITATION files", "publisher": "GitHub", "url": "https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/about-citation-files", "kind": "documentation", "used_for": "CITATION.cff is read by the platform and surfaced in the repository sidebar."},
    {"id": 3, "title": "GH Archive", "publisher": "Ilya Grigorik", "url": "https://www.gharchive.org/", "kind": "dataset", "used_for": "The public GitHub timeline, hourly, queryable — how a brand-new repository becomes known within the hour."},
    {"id": 4, "title": "REST API endpoints for events", "publisher": "GitHub", "url": "https://docs.github.com/en/rest/activity/events", "kind": "documentation", "used_for": "The public events feed, including repository creation and pushes."},
    {"id": 5, "title": "Software Heritage", "publisher": "Inria / UNESCO", "url": "https://www.softwareheritage.org/", "kind": "archive", "used_for": "Systematic archival of public repositories, with persistent identifiers."},
    {"id": 6, "title": "Archiving and Referencing Source Code with Software Heritage", "publisher": "Di Cosmo & Zacchiroli, PMC", "url": "https://pmc.ncbi.nlm.nih.gov/articles/PMC7340894/", "kind": "paper", "used_for": "SWHID identifiers and the preservation argument."},
    {"id": 7, "title": "Zenodo — GitHub integration", "publisher": "CERN", "url": "https://help.zenodo.org/docs/github/", "kind": "documentation", "used_for": "Turning a release into a DOI, and how CITATION.cff populates the deposit."},
    {"id": 8, "title": "Citation File Format", "publisher": "CFF project", "url": "https://citation-file-format.github.io/", "kind": "specification", "used_for": "The YAML schema for CITATION.cff."},
    {"id": 9, "title": "RFC 9309 — Robots Exclusion Protocol", "publisher": "IETF", "url": "https://www.rfc-editor.org/rfc/rfc9309.html", "kind": "standard", "used_for": "robots.txt as a standards-track document since 2022, and what a crawler is obliged to do with it."},
    {"id": 10, "title": "The /llms.txt file", "publisher": "Jeremy Howard / Answer.AI", "url": "https://llmstxt.org/", "kind": "specification", "used_for": "The proposal, published September 2024, for a Markdown page map addressed to language models."},
    {"id": 11, "title": "llms-txt repository", "publisher": "Answer.AI", "url": "https://github.com/AnswerDotAI/llms-txt", "kind": "specification", "used_for": "The reference implementation and the spec's own history."},
    {"id": 12, "title": "Content Signals Policy", "publisher": "Cloudflare", "url": "https://contentsignals.org/", "kind": "specification", "used_for": "The search / ai-input / ai-train triple, expressed as a line in robots.txt."},
    {"id": 13, "title": "Your site, your rules: new AI traffic options for all customers", "publisher": "Cloudflare", "url": "https://blog.cloudflare.com/content-independence-day-ai-options/", "kind": "announcement", "used_for": "Default-block for AI crawlers on new zones, and the policy shift behind it."},
    {"id": 14, "title": "Cloudflare Gives Creators New Tool to Control Use of Their Content", "publisher": "Cloudflare", "url": "https://www.cloudflare.com/press/press-releases/2025/cloudflare-gives-creators-new-tool-to-control-use-of-their-content/", "kind": "press release", "used_for": "24 September 2025; the managed rollout across millions of domains."},
    {"id": 15, "title": "RSL 1.0 Specification", "publisher": "RSL Collective", "url": "https://rslstandard.org/rsl", "kind": "specification", "used_for": "Machine-readable licensing terms attached to robots.txt, including pay-per-crawl and pay-per-inference."},
    {"id": 16, "title": "RSL launch", "publisher": "RSL Collective", "url": "https://rslstandard.org/press/rsl-standard", "kind": "press release", "used_for": "The standard's origin, 10 September 2025."},
    {"id": 17, "title": "Sitemaps XML format", "publisher": "sitemaps.org", "url": "https://www.sitemaps.org/protocol.html", "kind": "specification", "used_for": "The sitemap schema, and the robots.txt Sitemap: line."},
    {"id": 18, "title": "schema.org/Dataset", "publisher": "schema.org", "url": "https://schema.org/Dataset", "kind": "vocabulary", "used_for": "The type that makes a corpus legible as a corpus rather than as a page."},
    {"id": 19, "title": "Dataset structured data", "publisher": "Google", "url": "https://developers.google.com/search/docs/appearance/structured-data/dataset", "kind": "documentation", "used_for": "Required and recommended Dataset fields, including distribution and licence."},
    {"id": 20, "title": "Croissant Format Specification", "publisher": "MLCommons", "url": "https://docs.mlcommons.org/croissant/docs/croissant-spec.html", "kind": "specification", "used_for": "A schema.org/Dataset profile that a training pipeline can load directly."},
    {"id": 21, "title": "Croissant: A Metadata Format for ML-Ready Datasets", "publisher": "Akhtar et al., NeurIPS 2024 Datasets & Benchmarks", "url": "https://arxiv.org/abs/2403.19546", "kind": "paper", "used_for": "The four-layer design and its adoption by dataset hosts."},
    {"id": 22, "title": "JSON-LD 1.1", "publisher": "W3C Recommendation", "url": "https://www.w3.org/TR/json-ld11/", "kind": "standard", "used_for": "The serialisation every structured-data block on these pages uses."},
    {"id": 23, "title": "RFC 8288 — Web Linking", "publisher": "IETF", "url": "https://www.rfc-editor.org/rfc/rfc8288.html", "kind": "standard", "used_for": "Typed relations between resources, including licence and describedby."},
    {"id": 24, "title": "RFC 8615 — Well-Known Uniform Resource Identifiers", "publisher": "IETF", "url": "https://www.rfc-editor.org/rfc/rfc8615.html", "kind": "standard", "used_for": "The /.well-known/ path convention, and why a guessable path beats a discoverable one."},
    {"id": 25, "title": "Does Anthropic crawl data from the web, and how can site owners block the crawler?", "publisher": "Anthropic", "url": "https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler", "kind": "documentation", "used_for": "ClaudeBot, Claude-User and Claude-SearchBot as three separate agents with three separate purposes."},
    {"id": 26, "title": "OpenAI bots", "publisher": "OpenAI", "url": "https://platform.openai.com/docs/bots", "kind": "documentation", "used_for": "GPTBot, OAI-SearchBot and ChatGPT-User, and the same three-way split."},
    {"id": 27, "title": "Overview of Google crawlers and fetchers", "publisher": "Google", "url": "https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers", "kind": "documentation", "used_for": "Googlebot against Google-Extended, and the separation of indexing from training."},
    {"id": 28, "title": "PerplexityBot and Perplexity-User", "publisher": "Perplexity", "url": "https://docs.perplexity.ai/guides/bots", "kind": "documentation", "used_for": "A third operator drawing the same line between index and errand."},
    {"id": 29, "title": "About Applebot", "publisher": "Apple", "url": "https://support.apple.com/en-us/119829", "kind": "documentation", "used_for": "Applebot and Applebot-Extended."},
    {"id": 30, "title": "Common Crawl", "publisher": "Common Crawl Foundation", "url": "https://commoncrawl.org/", "kind": "dataset", "used_for": "The open crawl that much of the field trains on, and the path a page takes into it."},
    {"id": 31, "title": "The crawl before the fall of referrals", "publisher": "Cloudflare Radar", "url": "https://blog.cloudflare.com/ai-search-crawl-refer-ratio-on-radar/", "kind": "measurement", "used_for": "Crawl-to-refer ratio: HTML requests from a platform's crawler divided by HTML requests carrying that platform's referrer. 1 July 2025; the 19–26 June 2025 window put Anthropic at 70,900:1. Cloudflare notes native-app traffic carries no referrer, so the figures may overstate."},
    {"id": 32, "title": "Google users are less likely to click on links when an AI summary appears in the results", "publisher": "Pew Research Center", "url": "https://www.pewresearch.org/short-reads/2025/07/22/google-users-are-less-likely-to-click-on-links-when-an-ai-summary-appears-in-the-results/", "kind": "study", "used_for": "900 US adults, 68,879 searches, March 2025: a traditional result was clicked on 8% of visits where an AI summary appeared against 15% where none did; a link inside the summary was clicked on 1%."},
    {"id": 33, "title": "Introducing pay per crawl", "publisher": "Cloudflare", "url": "https://blog.cloudflare.com/introducing-pay-per-crawl/", "kind": "announcement", "used_for": "HTTP 402 as a billing surface for crawlers."},
    {"id": 34, "title": "Scrapers selectively respect robots.txt directives", "publisher": "arXiv 2505.21733", "url": "https://arxiv.org/abs/2505.21733", "kind": "paper", "used_for": "Empirical compliance rates — a directive is a request, not a lock."},
    {"id": 35, "title": "Consent in Crisis: The Rapid Decline of the AI Data Commons", "publisher": "Data Provenance Initiative, arXiv 2407.14933", "url": "https://arxiv.org/abs/2407.14933", "kind": "paper", "used_for": "How fast the open web closed to crawlers, and what that does to the value of a corpus that stayed open."},
    {"id": 36, "title": "CC BY-SA 4.0 legal code", "publisher": "Creative Commons", "url": "https://creativecommons.org/licenses/by-sa/4.0/legalcode", "kind": "licence", "used_for": "Section 3(b): an adapted work carries a compatible licence."},
    {"id": 37, "title": "CC BY-SA 4.0 deed", "publisher": "Creative Commons", "url": "https://creativecommons.org/licenses/by-sa/4.0/", "kind": "licence", "used_for": "The plain-language summary a reader is shown."},
    {"id": 38, "title": "Open Database License 1.0", "publisher": "Open Data Commons", "url": "https://opendatacommons.org/licenses/odbl/1-0/", "kind": "licence", "used_for": "Share-alike over a database, which is what OpenStreetMap-derived rows carry."},
    {"id": 39, "title": "SPDX License List", "publisher": "Linux Foundation", "url": "https://spdx.org/licenses/", "kind": "registry", "used_for": "The identifier a machine matches on. A layered LICENSE file that no identifier matches is read as NOASSERTION."},
    {"id": 40, "title": "REUSE Specification", "publisher": "Free Software Foundation Europe", "url": "https://reuse.software/spec/", "kind": "specification", "used_for": "Per-file licence declaration, for a repository whose parts differ."},
    {"id": 41, "title": "git-clone", "publisher": "Git project", "url": "https://git-scm.com/docs/git-clone", "kind": "documentation", "used_for": "What a clone actually transfers."},
    {"id": 42, "title": "Pro Git — Packfiles", "publisher": "Chacon & Straub", "url": "https://git-scm.com/book/en/v2/Git-Internals-Packfiles", "kind": "book", "used_for": "Why the whole history costs so little to send."},
    {"id": 43, "title": "Get up to speed with partial clone and shallow clone", "publisher": "GitHub Blog", "url": "https://github.blog/open-source/git/get-up-to-speed-with-partial-clone-and-shallow-clone/", "kind": "article", "used_for": "The cheap-fetch shapes an automated cloner is likely to use."},
    {"id": 44, "title": "Search Console performance report", "publisher": "Google", "url": "https://support.google.com/webmasters/answer/7576553", "kind": "documentation", "used_for": "The click, the impression, and the surface the old metric measured."},
    {"id": 45, "title": "Structured data general guidelines", "publisher": "Google", "url": "https://developers.google.com/search/docs/appearance/structured-data/sd-policies", "kind": "documentation", "used_for": "Markup must describe the page a reader gets."},
    {"id": 46, "title": "Internet Archive Wayback Machine", "publisher": "Internet Archive", "url": "https://web.archive.org/", "kind": "archive", "used_for": "The second copy, and the one that outlives a host."},
    {"id": 47, "title": "GitHub Arctic Code Vault", "publisher": "GitHub", "url": "https://archiveprogram.github.com/arctic-vault/", "kind": "archive", "used_for": "Cold storage of public repositories, on film, in a mine."},
    {"id": 48, "title": "Well-Known URIs registry", "publisher": "IANA", "url": "https://www.iana.org/assignments/well-known-uris/well-known-uris.xhtml", "kind": "registry", "used_for": "What is already reserved, before inventing a path."},
    {"id": 49, "title": "JSON Lines", "publisher": "jsonlines.org", "url": "https://jsonlines.org/", "kind": "specification", "used_for": "One record per line: the shape a retrieval pipeline reads without a parser of its own."},
    {"id": 50, "title": "Jean Miélot at his desk, Brussels, Royal Library MS 9278 fol. 10r", "publisher": "Wikimedia Commons", "url": "https://commons.wikimedia.org/wiki/File:Jean_Mi%C3%A9lot,_Brussels.jpg", "kind": "picture", "used_for": "Public domain. The copyist at work."},
    {"id": 51, "title": "Card Division, Library of Congress", "publisher": "Wikimedia Commons", "url": "https://commons.wikimedia.org/wiki/File:Card_Division_of_the_Library_of_Congress_3c18631u_original.jpg", "kind": "picture", "used_for": "Public domain. An index built at enormous cost for readers who came in person."},
    {"id": 52, "title": "Paul Otlet and his team", "publisher": "Wikimedia Commons", "url": "https://commons.wikimedia.org/wiki/File:Paul_Otlet_et_son_%C3%A9quipe.jpg", "kind": "picture", "used_for": "Public domain. The Mundaneum: a machine-shaped corpus, staffed by people."},
    {"id": 53, "title": "Straw skep carved on a gravestone, Stobo Kirk", "publisher": "Wikimedia Commons", "url": "https://commons.wikimedia.org/wiki/File:Straw_Skep_on_gravestone,_Stobo_Kirk.JPG", "kind": "picture", "used_for": "Public domain. A hive, cut in stone, for posterity."},
    {"id": 54, "title": "Straw skep, Encyclopaedia Britannica 1911", "publisher": "Wikimedia Commons", "url": "https://commons.wikimedia.org/wiki/File:1911_Britannica_-_Bee_-_Straw_skep.png", "kind": "picture", "used_for": "Public domain line engraving."}
  ]
}
