{
  "methodology_version": "2026-09-13.1",
  "generated_utc": "2026-09-20T03:08:03.986Z",
  "coverage": {
    "tranco_top_n": 50000,
    "robots_parsed": 30742,
    "crawlers_tracked": 18
  },
  "evidence": {
    "version": "2026-09-13.1",
    "crawler_identity": "The robots.txt census and the reachability sweep run as CrawlPriceIndexBot/1.0, self-identified, requests cryptographically signed (RFC 9421 / Web Bot Auth; public key directory: https://crawlpriceindex.com/.well-known/http-message-signatures-directory). The wide wire probe is a disclosed measurement study: one homepage request presenting a named AI crawler's published user-agent string (GPTBot this edition) and one as our own self-identified crawler (User-Agent CrawlPriceIndexBot/1.0, unsigned on this instrument — the honest-identity control), once per domain per edition. The wide probe's control has been this identified crawler since 2 September 2026; no leg of the wide probe is a browser. A second wire instrument, the identity matrix, makes the opposite trade: 331 domains rather than tens of thousands, each asked under six identities, one request per identity per domain per edition. The six are our own CrawlPriceIndexBot/1.0 (the honest-identity control); a current Chrome user-agent string, which is a string and not a browser; GPTBot's published user-agent string; ClaudeBot's published user-agent string; ClaudeBot's published user-agent string carrying the request header crawler-max-price: 0.001; and, from 17 September 2026, our own CrawlPriceIndexBot/1.0 carrying that same header and that same value. Those last two state a price a crawler would be prepared to pay, in the units that header defines, and they exist to observe whether a door answers differently when the request carries one. The sixth was added because the fifth varies two things at once against the plain ClaudeBot leg — an impersonated identity and an offer to pay — so nothing in the data could say which produced a different answer; with a declared identity carrying the identical header value, the price signal is isolated from who is asking. Its readings are archived from the day it starts and its level is published from its first edition; no edition-over-edition movement is claimed for it until its own series is long enough to support one, and the surface states how many editions it has. No payment is offered in any binding sense and none has ever been made — there is no wallet, no credential and no code path in this pipeline that could settle one. Presenting another operator's user-agent string means the response recorded is the response that string received, not one that operator received; the requests originate from an ordinary consumer connection and not from any published crawler address range, so a door that verifies its callers by address treats these knocks as unverified, which is itself part of what the matrix measures. Described in full at https://crawlpriceindex.com/methodology#identities, named there on 10 September 2026; the instrument itself is older.",
    "signal_classes": {
      "robots_stance": {
        "what": "Per-crawler directive parsed from the site's own robots.txt",
        "method": "GET /robots.txt with our identified user agent, parsed for User-agent/Disallow/Allow groups",
        "evidence_type": "observed",
        "values": {
          "blocked": "an explicit Disallow rule applies to this crawler",
          "allowed": "an explicit Allow rule applies",
          "partial": "some paths disallowed, not the whole site",
          "unlisted": "robots.txt exists but names no rule for this crawler",
          "no_robots": "no robots.txt was served"
        },
        "caveat": "A declaration is a request to crawlers, not an access control, and not evidence of what any crawler did. See probe_panel_observations."
      },
      "observed_price": {
        "what": "A price a site quotes to an AI crawler",
        "method": "Recorded only when the site returns it in a machine-readable payment response (HTTP 402 with a crawler-price/payment header, or an equivalent x402 offer)",
        "evidence_type": "observed",
        "caveat": "We publish only prices we have actually been quoted. We do not estimate, model or extrapolate prices for domains that do not quote one."
      },
      "payment_signal": {
        "what": "Paywall / licensing / toll signals other than an explicit price",
        "method": "HTTP status and headers on identified-crawler requests to the homepage (402 responses, TollBit-style token walls, licensing redirects, payment:free declarations)",
        "evidence_type": "observed"
      },
      "block_rates": {
        "what": "Share of scanned domains blocking a given crawler",
        "method": "Aggregated from robots_stance across all parsed domains in the current sweep",
        "evidence_type": "derived",
        "caveat": "Derived arithmetic over observations. Denominator is robots_parsed, not tranco_top_n."
      },
      "cctld_editions": {
        "what": "Block rates segmented by country-code top-level domain suffix",
        "method": "ccTLD suffix of the domain, aggregated where at least 8 domains are present",
        "evidence_type": "derived",
        "caveat": "A ccTLD is a domain-suffix classification, not a country. It is not operator location, ownership, audience or hosting, and generic suffixes (.com, .org, .io) carry no geography at all. We group by suffix and label by suffix."
      },
      "probe_panel_observations": {
        "what": "What a request presenting a named AI crawler received from the wide-probe set, counted against what each domain's robots.txt declared for that crawler",
        "method": "Compare this edition's robots.txt reading for the presented crawler against the HTTP status returned to the crawler-presenting request, on the wide probe's purposive set (top 2,000 by rank plus every domain blocking a tracked crawler)",
        "evidence_type": "observed (purposive set, counts only)",
        "caveat": "Published as counts and never as a rate. robots.txt is advisory, not a server access control, so 'declared block, page served' is protocol-normal and not a compliance failure. One vantage point on one dual-stack residential connection; each reached door knocked three ways in one sweep -- our identified crawler, a named crawler's user-agent string, a browser's user-agent string -- on a set built to over-represent blockers. A refusal is not evidence of intent and not proof any crawler was denied; we have no crawler-side logs."
      },
      "trends": {
        "what": "Edition-over-edition movement of the above",
        "method": "Difference between consecutive published editions held in our own archive",
        "evidence_type": "derived",
        "caveat": "History cannot be backfilled: series begin at the edition in which we first observed them."
      }
    },
    "not_measured": [
      "Actual revenue earned by any site",
      "Traffic volumes of any site",
      "Prices for domains that do not publish one",
      "Private licensing deal values",
      "Whether an AI company honoured a payment request",
      "Whether any named crawler obeys any site's robots.txt directives"
    ],
    "reproducibility": "Every figure is regenerated from a full census, twice a week (an edition on Sunday and one on Wednesday). Method, crawler identity and signal definitions are published at https://crawlpriceindex.com/methodology.html and versioned above.",
    "changelog_url": "https://crawlpriceindex.com/changelog.html — dated record of every methodology change; consult it to explain shifts between editions.",
    "corrections_url": "https://crawlpriceindex.com/corrections.json",
    "corrections_policy": "Errors are corrected in the next edition (the census runs twice a week) and listed in corrections[] with the date, what changed and why — the same records, as JSON, at https://crawlpriceindex.com/corrections.json and in the free feed at https://crawlpriceindex.com/index.json under `corrections`. We do not silently amend past editions."
  },
  "human_readable": "https://crawlpriceindex.com/methodology.html",
  "crawler_key_directory": "https://crawlpriceindex.com/.well-known/http-message-signatures-directory",
  "contact": "hello@crawlpriceindex.com"
}