{
    "_schema": "research-ledger/v1",
    "_type": "research-ledger",
    "title": "AI and Search Discovery Research Ledger",
    "purpose": "A running record of what we know about how AI systems and search engines discover, retrieve, interpret, and cite websites ... with every claim held at its true evidence level. Maintained by the weekly RealSEOLife.com research brief routine. The point is that a working theory never quietly becomes a fact just because it got repeated.",
    "maintained_by": "Weekly research brief routine, RealSEOLife.com",
    "created": "2026-09-08",
    "updated": "2026-10-02",
    "levels": {
        "established": "Documented by the platform itself, or independently reproduced enough that it is not seriously disputed. Primary vendor documentation, standards bodies.",
        "external-finding": "A credible third party measured or reported it. We have not verified it ourselves. Source and date required.",
        "hypothesis": "Our working theory. Not yet tested. Stated so that it could be shown false.",
        "demonstrated": "We tested or measured it with Digital Karma data and it held. Experiment slug and numbers required."
    },
    "promotion_rules": [
        "A hypothesis becomes demonstrated only through an actual experiment recorded on RealSEOLife.com with real warehouse numbers behind it.",
        "An external finding never becomes established just because time passed. It is promoted only when primary documentation confirms it, or when we reproduce it ourselves.",
        "A demonstrated entry states what was measured, not what it means. The interpretation goes in the linked article, not in the claim.",
        "Any entry not reviewed in 90 days gets re-checked before it is cited again."
    ],
    "entries": [
        {
            "id": "ai-crawler-vendor-expansion-2026-08",
            "level": "demonstrated",
            "kind": "observation",
            "claim": "Three AI crawlers expanded their portfolio-wide request volume by roughly an order of magnitude in the 28 days ending 2026-09-08, against the preceding 28 days.",
            "detail": "cohere-ai went from 1,444 to 12,811 requests, GrokBot from 1,121 to 11,590, and Claude-User from 921 to 10,924. The site spread differs sharply and matters more than the volume: GrokBot touched 71 properties and Claude-User 74, while cohere-ai concentrated on just 16. Established crawlers grew but did not step-change ... Amazonbot 48,782 to 74,311, Bytespider 15,779 to 32,882, OAI-SearchBot 9,403 to 14,816, and ClaudeBot was flat to slightly down at 41,863 to 40,249. Google-Agent rose from 1,358 to 9,473 across only 5 properties.",
            "evidence": "Digital Karma Data Warehouse, ai_crawler_daily joined to bot_fingerprints, portfolio-wide across roughly 130 properties.",
            "source": "warehouse-snapshot.json, generated 2026-09-08",
            "caveat": "Counts only crawlers that declare a known fingerprint. A bot that does not identify itself is absent from these numbers entirely, so this is a floor, not a total. ClaudeBot being flat while Claude-User multiplied 11.9x is the interesting split: those are different access modes, not one number.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "crawler-mode-split-hypothesis"
            ]
        },
        {
            "id": "realseolife-publish-burst-impressions-2026-08",
            "level": "demonstrated",
            "kind": "observation",
            "claim": "RealSEOLife.com weekly GSC impressions rose roughly 40x in the week of an eight-article publishing burst, while average position stayed flat and clicks did not move.",
            "detail": "Weekly impressions ran 144, 273, 75, 40, 47 through mid-August. Eight articles were published between 2026-08-16 and 2026-08-21. The week beginning 2026-08-17 recorded 1,936 impressions, then 2,561, then 2,296. Average position over the same weeks moved from the 40s and 50s to 57.0, 54.8, 55.4 ... essentially unchanged. Weekly clicks stayed at 0 to 2 throughout.",
            "evidence": "Digital Karma Data Warehouse, gsc_query_daily for site_id 24, eight weeks ending 2026-09-08.",
            "source": "warehouse-snapshot.json, generated 2026-09-08",
            "caveat": "This is an eligibility and surface effect, not a ranking effect, and it must not be described as one. More pages became eligible for more queries at roughly the same average depth. Nothing here shows the site ranking better for anything it already ranked for. The most recent week is always partial because GSC lags 2 to 3 days.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-23",
            "properties": [
                "realseolife.com"
            ],
            "related": [
                "publish-burst-surface-hypothesis"
            ]
        },
        {
            "id": "crawler-mode-split-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "Training-corpus crawlers and live retrieval fetchers respond to different site properties, so a property can be well covered by one and invisible to the other.",
            "detail": "ClaudeBot was flat while Claude-User multiplied nearly 12x over the same window on the same portfolio. If those two access modes tracked the same site signals they should move together. They did not.",
            "test": "Segment portfolio properties by the ratio of retrieval-mode requests to training-mode requests over 28 days, then compare the two groups on structured data coverage, /ai/ endpoint completeness, and content recency. Falsified if the ratio shows no relationship to any of those, or if the split is explained by property age or raw page count alone.",
            "metric": "ai_crawler_daily requests by bot_fingerprint, segmented by visitor_class",
            "properties": [
                "portfolio-wide"
            ],
            "window": "28 days, compared against the prior 28",
            "status": "open",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-10-02",
            "review_notes": [
                {
                    "date": "2026-10-02",
                    "note": "Left open and complicated rather than advanced. This run produced a clean behavioural split between crawlers on identical URLs, but it does not run along the axis this hypothesis assumes. Across fifteen BellyUp category subdomains over 2026-09-27 to 2026-10-02, under robots.txt reading Allow: / with no Disallow, GPTBot enumerated query-parameter combinations at 99.9 percent of 565,258 requests, meta-externalagent at 94.6 percent of 88,559 and Amazonbot at 99.0 percent of 28,283, while ClaudeBot (528 requests), OAI-SearchBot (302) and PerplexityBot (2) made zero query-string requests between them. Our own bot_fingerprints table classifies ClaudeBot as crawler_purpose training, identical to GPTBot, Amazonbot and meta-externalagent, so a training crawler sat on the non-enumerating side with the two ai_search crawlers. The separating variable in this window is the vendor, not the declared access mode. The new entry vendor-not-purpose-predicts-parameter-enumeration-hypothesis carries that reading forward as its own testable claim rather than overwriting this one."
                }
            ]
        },
        {
            "id": "publish-burst-surface-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "On a low-authority property, impression volume is governed mainly by indexed surface area, and average position is close to independent of publishing volume in the short term.",
            "detail": "Derived from the RealSEOLife.com burst: impressions moved 40x, position did not move at all, clicks stayed at zero. If true, publishing volume buys query eligibility but not competitiveness, and the two need to be measured and reported separately rather than as one number.",
            "test": "Repeat a comparable burst on a second low-authority property and check whether impressions rise while average position stays within a few points. Falsified if position improves materially alongside impressions, or if impressions stay flat.",
            "metric": "gsc_query_daily impressions and avg position, weekly, before and after",
            "properties": [
                "realseolife.com",
                "second property to be selected"
            ],
            "window": "8 weeks before, 8 weeks after",
            "status": "open",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-10-02",
            "review_notes": [
                {
                    "date": "2026-09-26",
                    "note": "Left open, not advanced. Adjacent evidence only. In the week 2026-09-19 to 2026-09-25, while RealSEOLife.com gained a large amount of new crawlable surface, its crawler volume rose sharply against the prior week: GPTBot 642 to 3,714, ClaudeBot 227 to 2,557, PerplexityBot 122 to 1,024, while the site's dominant crawler meta-externalagent stayed flat at 1,522 to 1,462. Warehouse events 938, 962, 1031 and 1072 record the surface added in that week, and events 946 and 1010 record two portfolio-wide server changes in the same window, a rewrite fix that had been returning HTTP 500 under /ai/ on 101 sites and the rollout of compression and static cache headers, so the crawler rise has at least three plausible causes and no clean attribution. It does not advance this hypothesis in any case, because this hypothesis concerns Search Console impressions against average position, not crawler requests. The test still needs a second low-authority property and the snapshot still carries no publishing-volume field to select one with."
                },
                {
                    "date": "2026-10-02",
                    "note": "Re-checked and left untouched. Nothing this run bears on Search Console impressions against average position, which is what this hypothesis measures. All of this week's evidence is server-log crawler behaviour. The test still needs a second low-authority property and the snapshot still carries no publishing-volume field to select one with."
                }
            ]
        },
        {
            "id": "google-user-triggered-fetchers-outside-robots-2026-09",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google documents user-triggered fetchers as a separate class from its common crawlers and states that this class generally ignores robots.txt rules. Google-Agent belongs to that class.",
            "detail": "Google's crawler and fetcher reference now sits in a dedicated Crawling infrastructure section at developers.google.com/crawling, framed as shared across Search, Gemini, Shopping, AdSense, News and Gemini Notebook. Google-Agent does not appear on the common crawlers page, which lists Googlebot, Googlebot-Image, Googlebot-Video, Googlebot-News, Storebot-Google, Google-InspectionTool, GoogleOther, GoogleOther-Image, GoogleOther-Video, Google-CloudVertexBot and Google-Extended. It appears on the user-triggered fetchers page as \"used by agents hosted on Google infrastructure to navigate the web and perform actions upon user request\" and uses IP ranges from user-triggered-agents.json. That page states of the class: \"Because the fetch was requested by a user, these fetchers generally ignore robots.txt rules.\" Google-NotebookLM is listed as a former agent, \"supported until August 2026,\" replaced by Google-GeminiNotebook, which \"requests individual URLs that Gemini Notebook users have provided as sources for their projects.\" The Google-Agent page carries a last updated stamp of 2026-08-19. OpenAI documents the equivalent boundary for ChatGPT-User: \"Because these actions are initiated by a user, robots.txt rules may not apply.\"",
            "evidence": "Primary vendor documentation, fetched 2026-09-08.",
            "source": "https://developers.google.com/search/docs/crawling-indexing/google-user-triggered-fetchers ... https://developers.google.com/crawling/docs/crawlers-fetchers/google-agent ... https://developers.google.com/search/docs/crawling-indexing/google-common-crawlers ... https://developers.openai.com/api/docs/bots",
            "caveat": "This records what the documentation says, not observed behaviour. We have not independently tested whether any of these fetchers actually ignore a Disallow directive on our own properties, and the warehouse currently carries no per-property robots.txt extract to test it with. The Google-Agent page's last updated stamp of 2026-08-19 dates the page, not necessarily the policy.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-08",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "crawler-mode-split-hypothesis",
                "agentic-fetcher-target-selection-hypothesis"
            ]
        },
        {
            "id": "cloudflare-crawler-purpose-defaults-2026-09-15",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "From 2026-09-15, Cloudflare applies defaults to new domains that block Training and Agent crawlers on pages displaying ads while leaving Search allowed, and multi-purpose crawlers combining Search and Training are caught by the Training block.",
            "detail": "Cloudflare's developer changelog entry dated 2026-07-01 defines three independent controls. Search is \"crawlers that index your content so they can answer questions about it later.\" Agent is \"automated activity acting in real time on a person's behalf, such as chat fetch bots.\" Training is \"crawlers that take your content to train or fine-tune a model.\" For each category a customer can block on all pages, block only on pages that display ads, or not block. The changelog states that from 2026-09-15 new domains onboarding receive defaults where \"Bots classified as Training or as Agent are blocked on pages that display ads, while Search remains allowed,\" and that \"Multi-purpose crawlers that combine Search and Training will be affected by the new defaults to block Training.\" The options themselves are available to all customers including the Free plan.",
            "evidence": "Cloudflare developer changelog, entry dated 2026-07-01, fetched 2026-09-08.",
            "source": "https://developers.cloudflare.com/changelog/post/2026-07-01-ai-traffic-options/",
            "caveat": "The new defaults apply to new domains onboarding, not retroactively to every existing zone, and the changelog is the record of intent rather than of observed enforcement. We have not verified which of our own properties sit behind Cloudflare, and the snapshot carries no hosting or CDN field, so we cannot currently say what this changes for us specifically. Effects on the wider bot mix should not be attributed to this date without a clean before-and-after baseline.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-08",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "crawler-mode-split-hypothesis",
                "agentic-fetcher-target-selection-hypothesis"
            ]
        },
        {
            "id": "google-eea-aggregator-supplier-units-2026-09",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google documents EEA-only aggregator units and supplier units as search surfaces that provide visibility to ecosystem participants, with no publisher markup required and no stated criteria for how a site is classified as one or the other.",
            "detail": "Google's documentation updates log carries an entry dated 2026-09-08: \"Added documentation about regional differences in Search experience, which includes information about search experiences available in certain countries, such as aggregator units, supplier units, and carousels.\" The new page defines an aggregator unit as \"A unit that provides visibility for ecosystem participants (aggregators) on the search results page\" and a supplier unit as \"A unit that provides visibility for ecosystem participants (direct suppliers) on the search results page.\" Both are documented as EEA only and cover hotels, flights, ground transportation and products. The page does not state what determines aggregator versus direct supplier treatment, and names no structured data or markup requirement.",
            "evidence": "Google Search Central documentation updates log and the aggregator features page, fetched 2026-09-08.",
            "source": "https://developers.google.com/search/updates ... https://developers.google.com/search/docs/appearance/aggregator-features",
            "caveat": "This is a documentation addition, not necessarily a product launch on that date. Scope is EEA only and four verticals only, none of which the portfolio currently operates in, so it is carried as a classification precedent rather than as an eligibility opportunity. Nothing here says the classification is used anywhere outside these units.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": []
        },
        {
            "id": "geo-misinformation-defense-gap-2026-09",
            "level": "external-finding",
            "kind": "measurement-study",
            "claim": "Off-the-shelf LLM guardrails reduce the success rate of information-distorting generative engine optimization attacks by at most 5.7 percent relative, according to one benchmark published 2026-09-02.",
            "detail": "Counter-GEO-Bench, arXiv 2609.02316, submitted 2026-09-02 by Bing Zheng, Zongyao Zhao and Wenming Yang. The benchmark pairs 247 human-verified, quality-gated queries with information-preserving and information-distorting GEO rewrites, and scores defenses on attack success rate, false positive rate and answer quality across three victim LLMs. Granite Guardian, Llama Guard 3 and NeMo Self-Check Fact-Checking reduced attack success rate by at most 5.7 percent relative, with Granite Guardian's reduction reported as not statistically significant. The paper's own baseline, C-GEO Guard, reduced it by 47.6 percent relative with near-zero utility loss. The stated explanation: \"Safety-taxonomy guardrails target policy violations, while GEO misinformation passes through them as fluent informational content.\"",
            "evidence": "arXiv preprint abstract, fetched 2026-09-08. Not reproduced by us.",
            "source": "https://arxiv.org/abs/2609.02316",
            "caveat": "A preprint, one team, one benchmark, three victim models, 247 queries, published days before this entry and not independently reproduced. It measures the defenses tested, not generative search products in production, and says nothing about what Google, OpenAI or Anthropic actually run in front of their retrieval stacks. Do not read it as a statement that retrieval can be gamed at will, and do not cite it as established.",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-09-08",
            "properties": [
                "portfolio-wide"
            ],
            "related": []
        },
        {
            "id": "agentic-fetcher-target-selection-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "Agentic and user-triggered fetchers are pointed at a small, selected set of properties by something other than general crawlability, so a property can be near-universally crawled by bulk crawlers and still receive zero agentic fetches.",
            "detail": "In the snapshot generated 2026-09-08, recent 28 days against prior 28: Google-Agent 1,358 to 9,473 requests across only 5 properties, Applebot-Extended 446 to 5,746 across only 4, cohere-ai 1,444 to 12,811 across 16. Over the same window Amazonbot went 48,782 to 74,311 across 127 properties, GPTBot 21,554 to 29,202 across 127, and OAI-SearchBot 9,403 to 14,816 across 129. The counter-example is carried on purpose: Claude-User multiplied 921 to 10,924 across 74 properties and GrokBot 1,121 to 11,590 across 71, so narrow footprint does not by itself predict growth. Inside the portfolio the asymmetry is concrete. aiwebsitesystems.com logged 400 Google-Agent requests over 28 days while realseolife.com logged no Google-Agent rows at all, and realseolife.com logged 1,701 Applebot-Extended requests, placing it inside Apple's 4-property set and outside Google's 5-property set on the same infrastructure.",
            "test": "Add a per-property robots.txt directive extract to the warehouse and join it to the crawler rollups so permissiveness can be held constant instead of assumed. Then ship a matched machine-readable surface onto a property currently receiving zero Google-Agent and zero Applebot-Extended requests and compare 28 days after against 28 days before, recording the 4 weeks either side of 2026-09-15 separately because Cloudflare's default change lands inside the window. Falsified if sites_touched_recent for Google-Agent and Applebot-Extended rises broadly across the portfolio with no change on our side, or if the treated property gains no agentic fetches while comparable untreated properties gain them anyway, or if the whole difference is explained by robots.txt permissiveness once that field exists.",
            "metric": "ai_crawler_daily requests by bot fingerprint plus sites_touched per fingerprint, segmented by declared purpose class, recent 28 days against prior 28",
            "properties": [
                "portfolio-wide",
                "realseolife.com",
                "aiwebsitesystems.com"
            ],
            "window": "28 days after the change, against the 28 days before it",
            "status": "open",
            "first_recorded": "2026-09-08",
            "last_reviewed": "2026-10-02",
            "review_notes": [
                {
                    "date": "2026-10-02",
                    "note": "Left open, and its missing instrumentation is worse than previously recorded. This hypothesis asks for a per-property robots.txt directive extract so permissiveness can be held constant instead of assumed. This run established that robots.txt is not merely unrecorded but actively overwritten: thirty of the portfolio's one hundred and four per-site sitemap generators write robots.txt from a fixed string on every run, and on 2026-10-02 that erased a Disallow: /*? rule which had been in place on thirty BellyUp category subdomains since 2026-09-11 and had measurably held GPTBot to zero filter-URL requests. The rule was only recoverable from weekly backup archives. Historical robots.txt permissiveness therefore cannot be reconstructed from live state at all, and a directive extract added today would have no backfill. A second confound is now explicit: reachable URL count varies by four orders of magnitude between portfolio properties, from 17 real URLs against 82,913 filter combinations on bars.bellyupjax.com to sites with no query-string surface at all, and that was never held constant in the original per-property comparison."
                }
            ]
        },
        {
            "id": "crawler-rollup-not-reproducible-2026-09-11",
            "level": "demonstrated",
            "kind": "observation",
            "claim": "Two Digital Karma warehouse extracts generated three days apart disagree by up to a factor of ten on AI crawler request counts over 28-day windows that share 25 of 28 days, while the Google Search Console series in the same two files reproduces to the digit.",
            "detail": "Snapshot A was generated 2026-09-08 and its numbers are preserved verbatim in this ledger under ai-crawler-vendor-expansion-2026-08 and agentic-fetcher-target-selection-hypothesis. Snapshot B was generated 2026-09-11T23:40:01+00:00. Recent-28d requests, A then B: OAI-SearchBot 14,816 then 15,071 across 129 then 130 properties, Bytespider 32,882 then 29,599, Applebot-Extended 5,746 then 4,586 on 4 properties in both, ClaudeBot 40,249 then 29,438, cohere-ai 12,811 across 16 then 8,142 across 7, Amazonbot 74,311 then 32,799 across the identical 127 properties, GrokBot 11,590 across 71 then 4,652 across 15, Claude-User 10,924 across 74 then 1,149 across 24, GPTBot 29,202 then 68,391. Google-Agent, recorded in A at 9,473 requests across 5 properties, has no row at all in B, which lists AI2Bot at 1 prior-window request, so B's fingerprint list is not truncated by volume. Over the same two files, RealSEOLife.com weekly GSC impressions read 144, 273, 75, 40, 47, 1,936, 2,561, 2,296 in A and 72, 273, 75, 40, 47, 1,936, 2,561, 2,669 in B, with average positions 57.0, 54.8, 55.4 in A and 56.97, 54.79, 55.63 in B. Six of eight impression values are identical; the 2026-08-31 week rose 16 percent, which is the documented GSC lag filling in a partial week. B is internally consistent: its by-fingerprint recent-28d requests sum to 243,670 against 225,578 across the four full weeks from 2026-08-17, with the window reaching two days into the week of 2026-08-10, and all seven named properties reconcile the same way.",
            "evidence": "Direct comparison of warehouse-snapshot.json generated 2026-09-11T23:40:01+00:00 against the 2026-09-08 figures recorded verbatim in this ledger on 2026-09-08.",
            "source": "warehouse-snapshot.json generated 2026-09-11 ... https://www.realseolife.com/data/research/ledger.json entries ai-crawler-vendor-expansion-2026-08 and agentic-fetcher-target-selection-hypothesis",
            "caveat": "What is demonstrated is that the two extracts disagree, not which one is right. Neither pull has been checked against raw log_requests, which the warehouse retains for only about 60 days and which this routine cannot reach from the sandbox. Server logs do not arrive late, so reporting lag cannot explain the crawler divergence, but a rollup rebuild, a backfill, a dedup change or a fingerprint dictionary change all could. Every count on both sides sees only crawlers that declare a known fingerprint, so both are floors. The GSC series reproducing is evidence that the problem is localised to the crawler rollup, not that the rest of the warehouse is verified.",
            "first_recorded": "2026-09-12",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide",
                "realseolife.com"
            ],
            "related": [
                "ai-crawler-vendor-expansion-2026-08",
                "crawler-mode-split-hypothesis",
                "agentic-fetcher-target-selection-hypothesis",
                "crawler-fingerprint-classification-drift-hypothesis"
            ]
        },
        {
            "id": "crawler-fingerprint-classification-drift-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "The instability between warehouse extracts is in bot fingerprint classification, not in raw request capture, so total AI-plausible request volume is roughly stable across two pulls while the per-fingerprint split is not.",
            "detail": "Inside snapshot B alone, four fingerprints carry real prior-window volume and almost none in the recent window: Claude-SearchBot 19,051 to 1,820 on 1 property, AgentTrust 18,978 to 381, DatasetSEO-AI-Observatory 1,249 to 0, Google-Extended 714 to 0. Across the two snapshots, the fingerprints that moved most also lost property coverage, Claude-User 74 to 24, GrokBot 71 to 15, cohere-ai 16 to 7, and Google-Agent 5 to absent, while Amazonbot held the identical 127 properties and OAI-SearchBot 129 then 130. A bot that slows down loses volume and keeps its property spread. A fingerprint that stops being matched loses both together. No vendor announced a retirement in the window: Anthropic's crawler page still lists all three Claude agents and is stamped 2026-04-07, Google's Search Central updates log carries nothing after 2026-09-08, and OpenAI's bot page still documents four bots.",
            "test": "Add three fields to the extract and re-pull the same window twice a week apart. First, a snapshot_id and an immutable stored copy of every extract. Second, a total AI-plausible request count including requests whose user agent matched no known fingerprint, reported alongside the sum of named fingerprints. Third, a version stamp for the fingerprint dictionary. Then compare two extracts whose recent windows overlap by at least 25 of 28 days.",
            "metric": "requests_recent_28d and sites_touched_recent per bot fingerprint, plus the unclassified residual, compared between two extracts over an 89 percent overlapping window",
            "properties": [
                "portfolio-wide"
            ],
            "window": "Two consecutive weekly extracts, confirmed against a third",
            "status": "open",
            "falsified_if": "Two extracts a week apart agree within 10 percent on every fingerprint over an 89 percent overlapping window, which would mean the 2026-09-08 to 2026-09-11 divergence was a one-off rather than drift. Also falsified if the unclassified residual stays flat while named fingerprints move by the same factors seen here, since that would put the instability in ingestion or in the rollup arithmetic rather than in classification.",
            "first_recorded": "2026-09-12",
            "last_reviewed": "2026-10-02",
            "related": [
                "crawler-rollup-not-reproducible-2026-09-11",
                "ai-crawler-vendor-expansion-2026-08",
                "multi-vendor-identity-spoof-non-get-2026-09-25"
            ],
            "review_notes": [
                {
                    "date": "2026-09-26",
                    "note": "Left open. This run supplies a concrete mechanism by which per-fingerprint counts can move without any vendor changing behaviour: on 2026-09-25 fifteen addresses each presented nine to eleven vendors' crawler identities, so a single actor can add volume to many fingerprints at once. It is a mechanism, not the explanation. The demonstrated spoof is confined to non-GET traffic and totals 1,065 requests against 317,280 GET requests over the same 28 days, which is 0.33 percent, far too small to account for the per-fingerprint swings this hypothesis was opened to explain. The three instrumentation fields it asks for, a snapshot id with immutable stored extracts, an unclassified residual count, and a fingerprint dictionary version, still do not exist, so the fault still cannot be located in classification rather than ingestion."
                },
                {
                    "date": "2026-10-02",
                    "note": "Left open, with the first independent corroboration of a large per-fingerprint move. GPTBot rose from 28,033 to 594,124 requests over the recent 28 days against the prior 28, a factor of 21, which is exactly the shape this hypothesis was opened to distrust. This time the move was checked against evidence outside the fingerprint dictionary: 572,212 of 573,723 GPTBot-labelled requests over 2026-09-27 to 2026-10-02 came from addresses inside the twelve CIDR ranges published at openai.com/gptbot.json and fetched into vendor_ip_ranges on 2026-09-30 01:55:56 UTC, with the remaining 1,511 outside them. The rise is therefore a real traffic change, not a classification artefact, and the method generalises: any suspect per-fingerprint swing can be tested against published ranges where a vendor publishes them, which currently covers Google, OpenAI and Perplexity only. The three instrumentation fields this hypothesis asks for, a snapshot id with immutable stored extracts, an unclassified residual count and a fingerprint dictionary version, still do not exist."
                }
            ]
        },
        {
            "id": "q2d-web-agentic-retrieval-benchmark-2026-09",
            "level": "external-finding",
            "kind": "measurement-study",
            "claim": "A benchmark published 2026-09-08 reports that first-stage retriever rankings are largely insensitive to which relevance judgment set is used, but diverge substantially across topical domains, query languages and query types, and that agentic pipelines retrieve against AI system-written query reformulations whose distribution differs from human search behaviour.",
            "detail": "Q2D-Web (Query2Doc-Web), arXiv 2609.08887, submitted 2026-09-08 by Maximilian Schall, Sedigheh Eslami, Markus Krimmel, Antoine Chaffin, Louis Milliken, Bo Wang and Denis Bykov. The benchmark pairs a 190M-document web corpus with 70k agentic search queries in ten languages, reformulated from real user queries in production systems, and supplies three fixed relevance judgment sets: agent citations, production rankings, and a combined set that unions both and adds LLM judgments of unlabeled pooled documents. The paper states its motivation directly: \"most benchmarks assess human-written queries, while the first-stage retrievers in agentic RAG pipelines serve AI system-written reformulations whose distribution differs from human search behavior.\" Thirteen retrievers were benchmarked across lexical, dense and late-interaction families. Retaining a third of the corpus by reciprocal rank fusion over pooled retriever runs preserved the full-corpus model ranking under the combined judgments while raising absolute Recall@1000 by only 3 to 7 points.",
            "evidence": "arXiv abstract, fetched 2026-09-12. Not reproduced by us.",
            "source": "https://arxiv.org/abs/2609.08887",
            "caveat": "A preprint, one team, four days old at the time of this entry, not independently reproduced. It evaluates first-stage retrievers on a fixed corpus, not any production generative search product, and says nothing about what Google, OpenAI, Anthropic or Perplexity actually run. The claim that agent reformulations differ distributionally from human queries is the paper's framing and its corpus construction choice, not a separately measured result reported in the abstract. Nothing here licenses a statement about what any individual site should change.",
            "first_recorded": "2026-09-12",
            "last_reviewed": "2026-09-12",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "searchatlas-agentic-evidence-graphs-2026-09"
            ]
        },
        {
            "id": "searchatlas-agentic-evidence-graphs-2026-09",
            "level": "external-finding",
            "kind": "measurement-study",
            "claim": "A framework published 2026-09-09 reconstructs how evidence propagates through LLM search agent trajectories and reports that process failures, including fragmented answer support and unverified parametric knowledge entering responses, are strongly associated with incorrect answers.",
            "detail": "SearchAtlas: Analyzing Agentic Search Strategies via Evidential Query Graphs, arXiv 2609.10901, submitted 2026-09-09 by Jiacheng Sang, Mengyuan Li, Sanxing Chen, Yukun Huang, Yu Feng and Bhuwan Dhingra, accepted to Findings of EMNLP 2026. The framework converts search trajectories into graphs whose edges represent how evidence moves from the query that retrieved it to the final answer. The automated parsing pipeline reports a mean edge F1 of 86.0 percent against human-annotated graphs. Five search agents were analysed on three benchmarks, showing what the paper calls \"systematic differences in search scale and evidence aggregation\". The abstract states that SearchAtlas \"exposes fragmented answer support, question constraints that do not reach the answer, and unverified parametric knowledge entering the response\", and that these process failures associate with incorrect answers more strongly than an LLM judge given the raw trajectory or the ordered query list.",
            "evidence": "arXiv abstract, fetched 2026-09-12. Not reproduced by us.",
            "source": "https://arxiv.org/abs/2609.10901",
            "caveat": "A preprint with conference acceptance, three days old at the time of this entry, five agents, three benchmarks, not reproduced by us. It measures research agents on benchmark tasks, not commercial answer engines on live web queries. It says nothing about which sites get retrieved or cited, and carries no implication for how any page should be written. Its usefulness here is the distinction it operationalises: a document being retrieved into a trajectory is not the same as that document supporting the answer.",
            "first_recorded": "2026-09-12",
            "last_reviewed": "2026-09-12",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "q2d-web-agentic-retrieval-benchmark-2026-09"
            ]
        },
        {
            "id": "gsc-page-indexing-report-no-backfill-2026-09",
            "level": "external-finding",
            "kind": "platform-statement",
            "claim": "On 2026-09-11 a Google Search Advocate stated that the June 2026 gap now visible in the Search Console page indexing report is permanent, because Google does not back-fill indexing data.",
            "detail": "Search Engine Roundtable and Search Engine Land both reported on 2026-09-11 that the page indexing report is missing a block of June 2026 data across accounts. John Mueller responded on Bluesky: \"This is likely from the time in June where the data was delayed - there just isn't any data for that period, the page indexing report wasn't updated. We don't back-fill indexing data.\" Both reports note he said he would check with the team. The underlying June event was a multi-week delay in the page indexing report, which was reported as resolved in early July 2026 without the missing period being restored.",
            "evidence": "Search Engine Roundtable and Search Engine Land, both dated 2026-09-11, fetched 2026-09-12. Not confirmed against Google documentation.",
            "source": "https://www.seroundtable.com/google-search-console-indexing-report-missing-data-42073.html ... https://searchengineland.com/google-search-console-indexing-report-missing-june-data-488261",
            "caveat": "This is a Googler's reply on a social platform relayed by two trade publications, not Google documentation, which is why it sits at level two and not level one. We have not seen the Bluesky post ourselves and the two outlets render the quote with slightly different punctuation. The statement concerns the page indexing report only. Nothing here says anything about the Search performance report, about impressions or clicks data, or about whether any page's actual index status changed. Do not read a reporting gap as a crawling or indexing event.",
            "first_recorded": "2026-09-12",
            "last_reviewed": "2026-09-12",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "crawler-rollup-not-reproducible-2026-09-11"
            ]
        },
        {
            "id": "crawler-rollup-reproduces-2026-09-18",
            "level": "demonstrated",
            "kind": "observation",
            "claim": "A third Digital Karma warehouse extract, generated seven days after the second, reproduces it on nine of the eleven AI crawler fingerprints recorded verbatim in this ledger, all within 0.85x to 1.29x, and sides against the 2026-09-08 extract on every fingerprint where the first two disagreed.",
            "detail": "Extract C was generated 2026-09-18T23:40:01+00:00. Recent 28 day requests, extract B (2026-09-11) then extract C: GPTBot 68,391 then 88,479; ClaudeBot 29,438 then 33,354; Bytespider 29,599 then 29,070; Amazonbot 32,799 then 28,493 across 127 then 128 properties; OAI-SearchBot 15,071 then 18,862 across 130 then 130; cohere-ai 8,142 then 6,932 across 7 then 20; GrokBot 4,652 then 3,946 across 15 then 24; Claude-User 1,149 then 1,145 across 24 then 38; AgentTrust 381 then 405. Two fall outside: Applebot-Extended 4,586 across 4 then 1,662 across 3, and Claude-SearchBot 1,820 across 1 then 834 across 14. Both are accounted for inside extract C itself, which reports Applebot-Extended at 4,530 requests in its prior 28 day window, within 1.2 percent of what B called recent, and Claude-SearchBot at 20,269 prior against 834 recent, the signature of a real fall-off caught at a window boundary rather than a classification failure. Three fingerprints carry identical prior-window counts in B and C, AgentTrust 18,978, Google-Extended 714 and DatasetSEO-AI-Observatory 1,249, which is what results when all of a bot's activity sits inside the stretch the two prior windows share. Google-Agent, recorded in extract A at 9,473 requests across 5 properties, has no row in B or C. Put all three extracts in sequence and the pattern is consistent: GPTBot 29,202 then 68,391 then 88,479; Amazonbot 74,311 then 32,799 then 28,493; Claude-User 10,924 then 1,149 then 1,145. Extract C is internally consistent on the same test B passed: by-fingerprint recent 28 day requests sum to 266,469 against 258,672 across the four weekly buckets from 2026-08-24, and all seven named properties reconcile between 1.00 and 1.07. The Google Search Console control still holds: RealSEOLife.com weekly impressions for every complete week present in both files read 75, 40, 47, 1,936, 2,561, 2,669 in each, with C adding 2,084 for the week of 2026-09-07 and a partial 515 for 2026-09-14.",
            "evidence": "Direct comparison of warehouse-snapshot.json generated 2026-09-18T23:40:01+00:00 against the 2026-09-11 and 2026-09-08 figures recorded verbatim in this ledger.",
            "source": "warehouse-snapshot.json generated 2026-09-18 ... https://www.realseolife.com/data/research/ledger.json entries crawler-rollup-not-reproducible-2026-09-11, crawler-fingerprint-classification-drift-hypothesis and ai-crawler-vendor-expansion-2026-08",
            "caveat": "What is demonstrated is that two of three extracts agree, not that they are correct. Three pulls from one pipeline can be wrong together. Neither has been checked against raw log_requests, which the warehouse retains about 60 days and which this routine cannot reach. The recent windows of B and C are seven days apart and therefore overlap by 21 of 28 days, below the 25 of 28 that crawler-fingerprint-classification-drift-hypothesis names for its own falsification test, so that test is still unrun and this entry does not satisfy it. Agreement here is within 30 percent, not the 10 percent that clause asks for. All counts on all three sides see only crawlers declaring a known fingerprint, so every number is a floor. The extract used here was generated 2026-09-18 and this comparison was run 2026-09-23, so five days of activity are outside it.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide",
                "realseolife.com"
            ],
            "related": [
                "crawler-rollup-not-reproducible-2026-09-11",
                "crawler-fingerprint-classification-drift-hypothesis",
                "ai-crawler-vendor-expansion-2026-08",
                "fetcher-property-preference-stability-hypothesis"
            ]
        },
        {
            "id": "google-search-profiles-badge-2026-09-16",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google documents a Search profile as a first-party object aggregating a publisher's content from across the web and social platforms, with a badge declared by a plain HTML link and no structured data requirement, and states that following a profile makes the linked content more likely to appear for that audience in Google Discover.",
            "detail": "Google Search Central's documentation updates log carries an entry dated 2026-09-16 adding a guide on adding \"a Search profile badge to your website\". The page defines the object: \"A Search profile brings together your content from across the web and social platforms into a single destination on Google.\" On the distribution effect: \"When readers follow your Search profile, it makes your content that's linked on your Search profile ... more likely to appear for your audience on Google Discover.\" The linked platforms named are Instagram, TikTok, YouTube, X, Facebook and the publisher's own website. Implementation is a standard HTML anchor element to a profile URL of the form https://profile.google.com/@handle paired with an SVG image, with a text link alternative offered. No structured data and no markup are required; the page states the profile must be claimed. Brand guidance includes touch targets of at least 48 x 48 dp on Android and 44 x 44 px on iOS and web, and keeping the Super G icon unmodified. The page is stamped 2026-09-16 UTC.",
            "evidence": "Primary vendor documentation, fetched 2026-09-23.",
            "source": "https://developers.google.com/search/docs/appearance/search-profiles ... https://developers.google.com/search/updates",
            "caveat": "This records what the documentation says, not observed behaviour, and we have not claimed or tested a Search profile on any portfolio property. The badge documentation names no follower threshold and states only that the profile must be claimed; Search Engine Journal's 2026-09-16 article by Roger Montti references separate coverage putting availability at 10,000 followers, which is trade reporting and not carried here as established. The documented effect is on Google Discover for that profile's followers. Nothing on the page says a Search profile is used as an entity signal anywhere in Search ranking or in AI surfaces, and it must not be described that way.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "google-aggregator-supplier-units-local-business-2026-09-18"
            ]
        },
        {
            "id": "google-aggregator-supplier-units-local-business-2026-09-18",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "On 2026-09-18 Google extended its EEA aggregator and supplier unit documentation to cover local business queries, ten days after first publishing it, with the definitions unchanged and still no stated criteria for how a site is classified as an aggregator or a direct supplier.",
            "detail": "Google's documentation updates log carries an entry dated 2026-09-18 updating the aggregator and supplier unit documentation to \"include support for local business queries\". The definitions are identical to the 2026-09-08 version recorded in google-eea-aggregator-supplier-units-2026-09: an aggregator unit is \"A unit that provides visibility for ecosystem participants (aggregators) on the search results page\" and a supplier unit is \"A unit that provides visibility for ecosystem participants (direct suppliers) on the search results page\". Both remain European Economic Area only, now covering local businesses alongside hotels, flights, ground transportation and products. The page states no eligibility criteria for the units beyond regional availability and names no markup requirement for them. On the same page, structured data carousels for local business queries are documented as available in the EEA, South Africa and Turkiye, and those do require structured data. Page stamped 2026-09-18 UTC.",
            "evidence": "Primary vendor documentation, fetched 2026-09-23.",
            "source": "https://developers.google.com/search/docs/appearance/aggregator-features ... https://developers.google.com/search/updates",
            "caveat": "A documentation update, not necessarily a product change on that date. Scope is still EEA only, and none of the covered verticals is one the portfolio operates in, so this stays a classification precedent and not an eligibility opportunity. The contrast on the page, where adjacent carousels require structured data and these units require none, is what the entry is for; nothing here says the aggregator or supplier classification is used anywhere outside these units. Filed as a separate entry rather than written into google-eea-aggregator-supplier-units-2026-09, which stays as published.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "google-eea-aggregator-supplier-units-2026-09",
                "google-search-profiles-badge-2026-09-16"
            ]
        },
        {
            "id": "conversational-platform-domain-preference-2026-09",
            "level": "external-finding",
            "kind": "measurement-study",
            "claim": "A study published 2026-09-16 across ChatGPT, Claude, Grok and DeepSeek reports that each platform's search returns results from its own preferred domains, that more frequent web search invocation does not necessarily improve response quality, and that some claims in responses rest on uncited search results.",
            "detail": "arXiv 2609.19244, Characterizing Web Search by Conversational LLM Agents: From Search Decisions and Strategies to Results and Responses, submitted 2026-09-16 by Mahsa Amani, Seungeon Lee, Abhisek Dash, Asmaa El Fraihi, Yunah Jang, Elisabeth Kirsten, Qinyuan Wu, Krishna P. Gummadi, Manish Gupta, Abhilasha Ravichander, Muhammad Bilal Zafar and Soumi Das, in cs.AI and cs.IR. The abstract describes \"the first study of Web search across four major conversational platforms (ChatGPT, Claude, Grok, and DeepSeek), combining real-world user interactions (invivo) with controlled experiments using the same platform's models by their APIs (invitro)\". Reported findings, quoted: web-search decisions \"vary substantially across platforms and models, while more frequent Web-search invocation does not necessarily yield better response quality\"; agents \"employ different complex querying strategies and ... platform specific search engines return search results from their preferred domains\"; and \"although responses are largely grounded in search results, some claims rely on uncited search results, raising concerns about attribution and reliability\".",
            "evidence": "arXiv abstract, fetched 2026-09-23. Not reproduced by us.",
            "source": "https://arxiv.org/abs/2609.19244",
            "caveat": "A preprint, one team, four platforms, one week old at the time of this entry, not independently reproduced. The domain preference result is reported in the abstract without the magnitude, the domain list, or the query sample being visible there, so nothing here licenses a statement about which domains any platform prefers. It says nothing about our properties. Do not read it as evidence that a site can be optimised per platform; it reports that preferences differ, not what produces them.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "q2d-web-agentic-retrieval-benchmark-2026-09",
                "searchatlas-agentic-evidence-graphs-2026-09",
                "fetcher-property-preference-stability-hypothesis"
            ],
            "review_notes": [
                {
                    "date": "2026-09-26",
                    "note": "Re-reviewed and held at external-finding. The hypothesis we derived from it on our own data, fetcher-property-preference-stability-hypothesis, is now falsified, but that outcome says nothing about the original arXiv result. It says our instrument was wrong. The external finding is unaffected and is not downgraded."
                }
            ]
        },
        {
            "id": "terms-txt-agentic-access-protocol-2026-09",
            "level": "external-finding",
            "kind": "protocol-proposal",
            "claim": "A preprint submitted 2026-09-10 proposes terms.txt, a robots.txt-style file expressing per-purpose bot access terms, enforced at the origin through Web Bot Auth signatures, signed intent, delegation tokens, HTTP 402 negotiation and signed receipts.",
            "detail": "arXiv 2609.11152, terms.txt: A Consent and Compensation Protocol for Agentic Web Access, submitted 2026-09-10 by Rajarshi Chowdhury, in cs.NI, cs.AI, cs.CR and cs.CY. The abstract argues the existing control is inadequate: robots.txt \"cannot express identity, purpose, terms, or price, can be circumvented, and newer alternatives are largely proprietary CDN features\". It specifies \"terms.txt, a robots.txt-style file for per-path, per-purpose bot access terms, plus an origin-enforced exchange using Web Bot Auth signatures, signed intent, delegation tokens, HTTP 402 negotiation, and signed receipts\", and defines \"what the exchange can enforce, audit, and leave to contract\". A dependency-free implementation is reported as adding 0.20 to 0.65 ms per request on one vCPU. The abstract's premise cites public measurements that \"automated clients now make up most requests, training dominates Cloudflare-classified crawling, and the largest AI platforms fetch thousands of pages for each visitor they return\".",
            "evidence": "arXiv abstract, fetched 2026-09-23. Not reproduced by us, and no implementation tested.",
            "source": "https://arxiv.org/abs/2609.11152",
            "caveat": "A single-author preprint with no standards body behind it and no adoption signal of any kind. It is not an IETF draft and must not be confused with the AIPREF working group documents. The traffic-ratio figures in its premise are the author's characterisation of third-party measurements, not results of this paper, and are not carried here as findings. Nothing here changes what any site should do today; it is filed as a proposal to track alongside AIPREF and Cloudflare Content Signals.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "cloudflare-crawler-purpose-defaults-2026-09-15",
                "google-user-triggered-fetchers-outside-robots-2026-09"
            ]
        },
        {
            "id": "gsc-crawl-stats-gap-2026-09-15",
            "level": "external-finding",
            "kind": "platform-statement",
            "claim": "The Search Console Crawl Stats report was missing 2026-09-15 across all Search Console profiles as reported on 2026-09-20, and was restored by 2026-09-22. Separately, on 2026-09-16 a Google Search Advocate stated that position counting for AI Mode and AI Overviews is not intended as an absolute truth and will keep changing.",
            "detail": "Barry Schwartz, Search Engine Roundtable, 2026-09-20: \"The crawl stats report is missing a day of data again, this time, the missing data is from September 15th\", affecting all Search Console profiles, with the clarification that \"This is not an issue with your site, it is an issue with Google's reporting in Google Search Console.\" The article carries an update showing the data restored by 2026-09-22. Separately, Schwartz reported on 2026-09-16 that John Mueller, answering a LinkedIn question about whether individual link citations inside an AI Mode response are counted as sequential positions or as a single block, said tracking will \"evolve over time\" as Google Search, AI Mode and AI Overviews \"evolve too\", that \"there are edge-cases\", that \"The goal is not a written-in-stone absolute truth for position counting (that's impossible)\", and that Google would \"update the documentation when / if there are significant changes\".",
            "evidence": "Search Engine Roundtable, articles dated 2026-09-20 and 2026-09-16, fetched 2026-09-23. Not confirmed against Google documentation.",
            "source": "https://www.seroundtable.com/google-search-console-crawl-stats-missing-42120.html ... https://www.seroundtable.com/google-search-console-ai-reporting-change-42099.html",
            "caveat": "Trade reporting plus a Googler's reply on a social platform, which is why this sits at level two. We have not seen the LinkedIn thread ourselves and have not checked a Search Console property to confirm the Crawl Stats gap or its restoration. A reporting gap is not a crawling or indexing event and must not be read as one. The Crawl Stats gap recovered, unlike the June page indexing gap recorded in gsc-page-indexing-report-no-backfill-2026-09, and nothing states a general rule about which reports back-fill. The coincidence in timing with our own warehouse anomaly is a coincidence in timing and nothing more; no common cause is claimed or implied.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-23",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "gsc-page-indexing-report-no-backfill-2026-09",
                "crawler-rollup-reproduces-2026-09-18"
            ]
        },
        {
            "id": "fetcher-property-preference-stability-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "The per-property mix of AI fetcher fingerprints reflects a stable platform-side preference for particular properties rather than extract noise, so a property will keep drawing the same fetchers across consecutive weekly extracts with nothing changed on our side.",
            "detail": "Derived from arXiv 2609.19244, which reports that \"platform specific search engines return search results from their preferred domains\", crossed with the per-property tables in the 2026-09-18 extract. The frozen baseline, recent 28 days, top fingerprints per property. realseolife.com: meta-externalagent 4,705, GPTBot 1,207, ClaudeBot 1,123, Bytespider 818, YouBot 635, Applebot-Extended 625, Amazonbot 500, OAI-SearchBot 276, PerplexityBot 236, CCBot 68, ChatGPT-User 41, AgentTrust 12, Claude-User 1. aisymantix.com: Claude-User 587, Amazonbot 481, CCBot 258, OAI-SearchBot 190, ClaudeBot 120, GPTBot 107, ChatGPT-User 66, YouBot 54, meta-externalagent 20, PerplexityBot 16, Bytespider 10, Claude-SearchBot 2. quickrankai.com: ClaudeBot 499, OAI-SearchBot 318, Bytespider 294, GPTBot 180, Amazonbot 92, Claude-User 16, AgentTrust 12, ChatGPT-User 7, YouBot 4, CCBot 4, meta-externalagent 1. seo.krisada.com: meta-externalagent 2,178, ClaudeBot 601, Amazonbot 444, Bytespider 364, GPTBot 273, YouBot 149, CCBot 124, OAI-SearchBot 56. datasetseo.com: ClaudeBot 877, Amazonbot 489, GPTBot 351, OAI-SearchBot 189, ChatGPT-User 36, meta-externalagent 9, CCBot 7, Bytespider 3, cohere-ai 3, AI2Bot 3, YouBot 2, DeepSeekBot 1. signalarchitectgroup.com: GPTBot 397, ClaudeBot 357, Amazonbot 176, OAI-SearchBot 170, Claude-User 33, AgentTrust 12, PerplexityBot 7, ChatGPT-User 4, YouBot 2, meta-externalagent 1. aiwebsitesystems.com: meta-externalagent 1,466, GPTBot 283, OAI-SearchBot 268, ClaudeBot 220, Amazonbot 169, YouBot 28, ChatGPT-User 7, CCBot 4, Bytespider 3. The asymmetries that motivate the question: meta-externalagent 4,705 on realseolife.com against 1 on quickrankai.com and 1 on signalarchitectgroup.com, and Claude-User 587 on aisymantix.com against 1 on realseolife.com. These properties share infrastructure, owner and broad subject matter.",
            "test": "Change nothing on the properties. Compare the baseline above against the next two weekly extracts, 2026-09-25 and 2026-10-02, on per-property fingerprint share and rank order. Only if the mix holds across all three does the preference question become testable, at which point per-property robots.txt, CDN and schema coverage fields must exist so permissiveness and markup can be held constant rather than assumed.",
            "metric": "Per-property share and rank order of ai_crawler_daily requests by bot fingerprint, recent 28 days, across three consecutive weekly extracts",
            "properties": [
                "realseolife.com",
                "aisymantix.com",
                "quickrankai.com",
                "seo.krisada.com",
                "datasetseo.com",
                "signalarchitectgroup.com",
                "aiwebsitesystems.com"
            ],
            "window": "Three consecutive weekly extracts: 2026-09-18 baseline, then 2026-09-25 and 2026-10-02",
            "status": "falsified",
            "falsified_if": "The rank order of the top three fingerprints on any property reshuffles between extracts with nothing changed on that property. Also falsified if the asymmetries disappear once per-property robots.txt and CDN fields exist and are held constant, or if meta-externalagent's uneven spread tracks hosting, sitemap structure or raw page count rather than anything platform-side. A single extract failing to reproduce returns this behind crawler-fingerprint-classification-drift-hypothesis, which is its prerequisite.",
            "first_recorded": "2026-09-23",
            "last_reviewed": "2026-09-26",
            "related": [
                "conversational-platform-domain-preference-2026-09",
                "crawler-fingerprint-classification-drift-hypothesis",
                "crawler-rollup-reproduces-2026-09-18",
                "agentic-fetcher-target-selection-hypothesis",
                "rank-order-crawler-mix-test-measures-margin-2026-09-25",
                "rolling-window-turns-single-week-burst-into-trend-2026-09-25"
            ],
            "status_history": [
                {
                    "date": "2026-09-23",
                    "status": "open"
                },
                {
                    "date": "2026-09-26",
                    "status": "falsified"
                }
            ],
            "resolved": "2026-09-26",
            "resolution": "Falsified at test two of three, on its own stated criterion. The criterion was that the rank order of the top three fingerprints reshuffling on any property with nothing changed would kill it. Four of seven properties reshuffled. aisymantix.com, quickrankai.com, datasetseo.com and signalarchitectgroup.com all changed their top three; realseolife.com, seo.krisada.com and aiwebsitesystems.com held.",
            "resolution_evidence": "Both 28-day windows were reconstructed from ai_crawler_daily directly rather than compared across two snapshot files, using weekly buckets anchored to 2026-09-25, so the baseline window is 2026-08-22 to 2026-09-18 and the current window is 2026-08-29 to 2026-09-25. The reconstruction reproduces the frozen baseline in this entry closely and two of its figures exactly, Claude-User 587 on aisymantix.com and ClaudeBot 877 on datasetseo.com. Baseline against current, top three per property: realseolife.com meta-externalagent 4,917 to 5,061, GPTBot 1,275 to 4,679, ClaudeBot 1,167 to 3,452, order held. seo.krisada.com meta-externalagent 2,247 to 2,574, ClaudeBot 617 to 781, Amazonbot 482 to 506, order held. aiwebsitesystems.com meta-externalagent 1,505 to 1,490, GPTBot 334 to 415, OAI-SearchBot 294 to 365, order held. aisymantix.com baseline Claude-User 587, Amazonbot 500, CCBot 258; current Amazonbot 444, CCBot 258, ClaudeBot 205. quickrankai.com baseline ClaudeBot 524, OAI-SearchBot 335, Bytespider 326; current ClaudeBot 551, OAI-SearchBot 417, GPTBot 309. datasetseo.com baseline ClaudeBot 877, Amazonbot 512, GPTBot 353; current ClaudeBot 441, Amazonbot 404, OAI-SearchBot 272. signalarchitectgroup.com baseline GPTBot 400, ClaudeBot 377, Amazonbot 189; current ClaudeBot 369, GPTBot 326, OAI-SearchBot 218.",
            "what_it_taught": "More useful than the hypothesis was. The failure is not evidence against platform preference; it is evidence that the chosen instrument could not have measured preference either way. See rank-order-crawler-mix-test-measures-margin-2026-09-25 and rolling-window-turns-single-week-burst-into-trend-2026-09-25, which together show that the test tracked margin width and that its headline example was a single week."
        },
        {
            "id": "webbotauth-httpsig-wg-draft-2026-09-01",
            "level": "established",
            "kind": "standards-document",
            "claim": "Crawler identity now has an adopted IETF working-group draft. Web Bot Auth lets an automated client cryptographically prove which key signed its request, so a site can verify identity instead of inferring it from a user agent string.",
            "detail": "The document is draft-ietf-webbotauth-httpsig-protocol-00, titled \"HTTP Message Signatures for automated traffic\", dated 2026-09-01, and it replaces the individual submission draft-meunier-webbotauth-httpsig-protocol, which is what working-group adoption looks like on the datatracker. Datatracker state is I-D Exists with consensus boilerplate Unknown and no IESG state or telechat date assigned. It specifies asymmetric signing of outbound HTTP requests, a Signature-Agent header field for in-band key discovery, a JSON Web Key Set directory format, the well-known URI /.well-known/http-message-signatures-directory, and three discovery types named directory, jwks_uri and CIMD. Stated aims include regulatory transparency, resource management, anti-impersonation and differentiating human from automated traffic. Platform support predates adoption: Google's own crawling changelog records \"Added Web Bot Auth authentication documentation\" on 2026-05-04, and Cloudflare publishes bot-verification documentation for it.",
            "evidence": "Primary standards document and primary vendor changelog, both fetched 2026-09-26.",
            "source": "https://datatracker.ietf.org/doc/draft-ietf-webbotauth-httpsig-protocol/ ... https://developers.google.com/crawling/docs/changelog",
            "caveat": "An Internet-Draft is not a published RFC and can still change. A valid signature proves that the holder of a key published at a given URL signed the message; it does not prove the agent is well behaved, that it honours robots.txt, or that it is entitled to the content. It answers who, not whether. We have no evidence yet about how many senders actually deploy it, because we do not capture the Signature-Agent header and cannot currently tell whether any request we have received carried one.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "signed-agent-identity-separates-spoof-from-shift-hypothesis",
                "multi-vendor-identity-spoof-non-get-2026-09-25",
                "crawler-fingerprint-classification-drift-hypothesis"
            ]
        },
        {
            "id": "akamai-ai-crawler-post-shift-2026-09-22",
            "level": "external-finding",
            "kind": "vendor-measurement",
            "claim": "Akamai reports that verified AI crawlers have moved beyond reading pages into sending POST requests that carry out actions such as logins, cart additions and checkouts.",
            "detail": "From a 30-day analysis of Akamai's global customer traffic, as reported by Help Net Security on 2026-09-22: ecommerce accounted for 44.8 percent of AI bot POST transactions, travel climbed to 30 percent in a single month, and Model Context Protocol traffic made up 4.1 percent. ChatGPT is named among the verified crawlers involved. Akamai representatives named in the piece, Steve Winterfeld and Ryan Gao, declined to say which transaction types were most common, saying it varies by the retailer's own logs. No overall GET against POST split was disclosed.",
            "evidence": "Trade press coverage of a vendor report. Help Net Security article fetched 2026-09-26.",
            "source": "https://www.helpnetsecurity.com/2026/09/22/ai-crawler-traffic-online-stores/",
            "caveat": "Akamai's own report was not linked from the coverage and was not fetched, so the method, the exact date range and the definition of a verified crawler are all as summarised by the publication rather than read at source. The measurement is drawn from ecommerce and travel traffic, and our portfolio is content properties, so the trend has no particular reason to appear on our logs. It is filed here because it is the claim our own non-GET measurement this week was made against, not because we expect to reproduce it.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "multi-vendor-identity-spoof-non-get-2026-09-25"
            ]
        },
        {
            "id": "datadome-bot-detection-failure-2026-09-23",
            "level": "external-finding",
            "kind": "vendor-measurement",
            "claim": "DataDome reports that about 65.3 percent of tested popular websites detected none of ten simulated bot types, and that traffic from AI agents and large language model crawlers rose 82.3 percent over twelve months.",
            "detail": "DataDome's State of Bot and Agent Security Report 2026, as reported by Help Net Security on 2026-09-23, analysed trillions of requests across more than 75,000 customer websites over the twelve months from July 2025 to June 2026, and separately ran ten simulated bot types against 21,491 popular website homepages. Of those homepages, about 65.3 percent detected none of the ten and only 2.4 percent stopped or challenged every type. Over the same twelve months, AI agent and large language model crawler traffic rose 82.3 percent, malicious bot activity rose 124 percent, and human traffic grew 13.2 percent. Bots and AI agents generated roughly 26.5 percent of all traffic, scraping accounted for 70.9 percent of bad bot traffic, and AI bot traffic to login pages rose more than eightfold in the first half of 2026.",
            "evidence": "Trade press coverage of a vendor report. Help Net Security article fetched 2026-09-26.",
            "source": "https://www.helpnetsecurity.com/2026/09/23/datadome-growing-bad-bot-traffic-report/",
            "caveat": "DataDome's own report was not fetched; every figure here is as summarised by the publication. The vendor sells bot detection, so a detection-failure headline is favourable to it, and the simulated-bot test is DataDome's own construction rather than an independent protocol. The figure that matters most for this ledger is the detection failure rate, because it bears directly on how much confidence any published per-vendor bot statistic deserves, including ours.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "multi-vendor-identity-spoof-non-get-2026-09-25",
                "webbotauth-httpsig-wg-draft-2026-09-01"
            ]
        },
        {
            "id": "google-videoobject-creator-interactionstatistic-2026-09-24",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google added support for the creator property in VideoObject structured data and clarified which interaction types are supported for interactionStatistic.",
            "detail": "Google's Search Central documentation updates log records on 2026-09-24 that it added support for the creator property, clarified the supported interaction types for interactionStatistic, and documented that both creator and author are supported in VideoObject structured data.",
            "evidence": "Primary vendor documentation. developers.google.com/search/updates fetched 2026-09-26.",
            "source": "https://developers.google.com/search/updates",
            "caveat": "This records a documentation change, not an observed ranking or appearance effect. Google adding support for a property does not establish that it uses the property for anything we can measure. No property in the portfolio currently emits VideoObject at scale, so we have no way to test it here.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": []
        },
        {
            "id": "google-mediapartners-generalised-2026-09-17",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google generalised its Mediapartners-Google documentation to state that the crawler affects various ad-related products rather than AdSense alone, which means one fingerprint now stands for several products.",
            "detail": "Google's crawling documentation changelog records on 2026-09-17 that it generalised the Mediapartners-Google crawler documentation to clarify that it affects various ad-related products beyond AdSense. The entry sits on the separate crawling documentation site Google stood up in November 2025, whose own stated reason is that \"Google's crawling infrastructure is shared across a variety of Google products beyond Search, including Google Shopping, News, Gemini, AdSense, and more\". The same changelog places Google-Agent and its IP ranges for user-triggered agents at 2026-03-20, the NotebookLM user agent rename to Google-GeminiNotebook at 2026-07-16, and Web Bot Auth documentation at 2026-05-04.",
            "evidence": "Primary vendor documentation. developers.google.com/crawling/docs/changelog fetched 2026-09-26.",
            "source": "https://developers.google.com/crawling/docs/changelog",
            "caveat": "The changelog states which documentation changed, not when the underlying crawler behaviour changed, and the two are not the same date. This entry extends google-user-triggered-fetchers-outside-robots-2026-09 rather than replacing it.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "google-user-triggered-fetchers-outside-robots-2026-09"
            ]
        },
        {
            "id": "multi-vendor-identity-spoof-non-get-2026-09-25",
            "level": "demonstrated",
            "kind": "observation",
            "claim": "Across the portfolio, 902 of the 1,065 non-GET requests classified as declared AI crawlers over the 28 days ending 2026-09-25 came from addresses that each presented three or more different vendors' crawler identities on a single day, 2026-09-25.",
            "detail": "Method breakdown for visitor_class ai_crawler over 2026-08-29 to 2026-09-25: GET 317,280 requests across 169 sites, POST 955 across 43 sites, DELETE 91 across 28 sites, HEAD 19 across 2 sites. Grouping the non-GET requests by address and counting distinct vendors per address per day returns fifteen addresses presenting nine to eleven vendors and ten to fifteen distinct fingerprint names each, all of them on 2026-09-25 and all of them in the 34.x and 35.x ranges that belong to Google Cloud. The busiest presented eleven vendors and fifteen fingerprint names in 206 requests to one site. Accounting for all 1,065 non-GET requests: 902 from addresses carrying three or more vendor identities on 2026-09-25, 11 from other addresses on that same day, and 152 across all other days combined. The fifteen borrowed names were Amazonbot, GPTBot, ClaudeBot, Claude-User, Claude-SearchBot, PerplexityBot, meta-externalagent, GrokBot, DeepSeekBot, OAI-SearchBot, CCBot, YouBot, Bytespider, cohere-ai and ChatGPT-User. The strongest single signal is the method itself: 91 DELETE requests were attributed to crawler fingerprints, and no search crawler, training crawler or retrieval fetcher has any reason to send DELETE to a content site.",
            "evidence": "Digital Karma Data Warehouse, log_requests joined to bot_fingerprints, queried live on 2026-09-26 across all logged portfolio properties. Three separate queries: method totals by visitor_class, non-GET totals by fingerprint and method with first and last observed day, and a per-address distinct-vendor count restricted to non-GET requests.",
            "source": "Live warehouse query, 2026-09-26. Article: https://www.realseolife.com/article/fifteen-ai-crawlers-one-bot",
            "caveat": "This is 1,065 requests against 317,280 GET requests over the same window, which is 0.33 percent of declared AI crawler volume. It demonstrates that borrowed crawler identity is real and concentrated on our own infrastructure and that the method column exposes it. It demonstrates nothing at all about the other 99.67 percent, whose identity still rests entirely on unverifiable user agent strings. The multi-vendor rule is a detection heuristic, not proof that shared infrastructure can never exist, as the warehouse alignment agreement already records. Three fingerprints do carry non-GET activity spanning multiple days and are therefore not accounted for by the single-day sweep: Claude-User POST across 2026-09-07 to 2026-09-25, ClaudeBot POST from 2026-09-05, and meta-externalagent POST from 2026-09-16. Those are the only non-GET traffic on the portfolio that could plausibly relate to the Akamai finding, and 154, 57 and 65 requests are too few to conclude anything from.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "portfolio-wide"
            ],
            "related": [
                "akamai-ai-crawler-post-shift-2026-09-22",
                "datadome-bot-detection-failure-2026-09-23",
                "webbotauth-httpsig-wg-draft-2026-09-01",
                "signed-agent-identity-separates-spoof-from-shift-hypothesis",
                "crawler-fingerprint-classification-drift-hypothesis"
            ]
        },
        {
            "id": "rank-order-crawler-mix-test-measures-margin-2026-09-25",
            "level": "demonstrated",
            "kind": "methodology",
            "claim": "Testing per-property AI crawler mix by whether the top three fingerprints keep their rank order measures how far apart the leaders are, not whether the mix is stable. On the portfolio the split was exact: every property whose order held had its leader ahead by at least 3.64 times, and every property that reshuffled had its leaders within 1.71 times.",
            "detail": "Seven properties, two 28-day windows a week apart, both reconstructed from ai_crawler_daily rather than compared across snapshot files. Order held on three: realseolife.com with the leader 3.86 times second place and 46.0 percent of the property's AI crawler requests, seo.krisada.com at 3.64 times and 51.7 percent, aiwebsitesystems.com at 4.51 times and 58.7 percent. Order reshuffled on four: datasetseo.com at 1.71 times and 43.7 percent, quickrankai.com at 1.56 times and 32.2 percent, aisymantix.com at 1.17 times and 30.1 percent, signalarchitectgroup.com at 1.06 times and 33.1 percent. There is no overlap between the two groups and the gap between them is wide, 1.71 against 3.64. The mechanism is arithmetic rather than statistical: when two counts sit within a few percent of each other, an ordinary week of variation reorders them, and when one count is several times the other, nothing short of a real change reorders them. A rank-order criterion therefore reports stability wherever a property happens to have a dominant crawler, regardless of whether anything about platform preference is true.",
            "evidence": "Digital Karma Data Warehouse, ai_crawler_daily joined to bot_fingerprints and sites, queried live 2026-09-26. Weekly buckets anchored to 2026-09-25 so that the baseline window 2026-08-22 to 2026-09-18 and the current window 2026-08-29 to 2026-09-25 are built from the same rows. The reconstruction reproduces the frozen baseline held in fetcher-property-preference-stability-hypothesis closely and reproduces Claude-User 587 on aisymantix.com and ClaudeBot 877 on datasetseo.com exactly.",
            "source": "Live warehouse query, 2026-09-26. Article: https://www.realseolife.com/article/fifteen-ai-crawlers-one-bot",
            "caveat": "Seven properties is a small sample and the separation being perfect at n equals 7 should not be read as a validated threshold. The 3.64 figure is where this portfolio happens to split, not a general constant. What is not sample-dependent is the mechanism, since close values reordering under small changes is a property of arithmetic rather than an empirical correlation. The replacement instrument is not established here: share with a stated margin of error is the obvious candidate but has not been tested.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "realseolife.com",
                "aisymantix.com",
                "quickrankai.com",
                "seo.krisada.com",
                "datasetseo.com",
                "signalarchitectgroup.com",
                "aiwebsitesystems.com"
            ],
            "related": [
                "fetcher-property-preference-stability-hypothesis",
                "rolling-window-turns-single-week-burst-into-trend-2026-09-25"
            ]
        },
        {
            "id": "rolling-window-turns-single-week-burst-into-trend-2026-09-25",
            "level": "demonstrated",
            "kind": "methodology",
            "claim": "A 28-day rolling window reported a single week of crawler activity as a four-week standing pattern and then as a collapse, with nothing changed on the property. The Claude-User preference for aisymantix.com that motivated an entire hypothesis was 586 of its 587 requests inside one week.",
            "detail": "Claude-User requests to aisymantix.com by week, buckets anchored to 2026-09-25: 586 in 2026-08-22 to 2026-08-28, then 0, then 1, then 0, then 0. The 28-day total of 587 held in the frozen 2026-09-18 baseline is therefore one week of activity and one stray request. Because the window is a rolling 28 days, that burst was reported as a standing 587 for four consecutive weeks and then as a fall to a single request the moment the burst rolled out of the back of the window, producing two false readings from one event. The same decomposition shows a genuine decay that a rolling window would have flattened in the other direction: ClaudeBot on datasetseo.com ran 436, 280, 84, 77, 0 across the same five weeks, which is a real fall-off rather than a window artefact.",
            "evidence": "Digital Karma Data Warehouse, ai_crawler_daily joined to bot_fingerprints and sites, weekly decomposition queried live 2026-09-26 over 2026-08-22 to 2026-09-25.",
            "source": "Live warehouse query, 2026-09-26. Article: https://www.realseolife.com/article/fifteen-ai-crawlers-one-bot",
            "caveat": "The general point about rolling windows is well understood and is not presented as a discovery. What is demonstrated here is specific and local: this ledger's own published baseline carried a single-week burst as a per-property preference, and four weeks of this routine's reasoning rested on it. The corrective is cheap and should now be standing practice: decompose any 28-day crawler figure into its four weeks before calling it a pattern.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "properties": [
                "aisymantix.com",
                "datasetseo.com",
                "portfolio-wide"
            ],
            "related": [
                "fetcher-property-preference-stability-hypothesis",
                "rank-order-crawler-mix-test-measures-margin-2026-09-25",
                "crawler-rollup-reproduces-2026-09-18"
            ]
        },
        {
            "id": "signed-agent-identity-separates-spoof-from-shift-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "Serving a Web Bot Auth key directory and recording signature presence will separate verifiable vendor fetches from borrowed-identity traffic, and the borrowed share of declared AI crawler requests will prove larger than zero on the GET side too, rather than staying confined to the non-GET traffic where we found it.",
            "detail": "Derived from webbotauth-httpsig-wg-draft-2026-09-01 crossed with multi-vendor-identity-spoof-non-get-2026-09-25. We have now demonstrated borrowed crawler identity on our own infrastructure, but only inside 0.33 percent of declared AI crawler volume, because the method column was the only identity evidence available. Every one of the 317,280 GET requests in the same window rests on a user agent string alone. Web Bot Auth is the first mechanism that would let the site verify rather than infer, and the same measurement would tell us how much of our crawler reporting has been costume all along.",
            "test": "Generate an Ed25519 key pair and publish /.well-known/http-message-signatures-directory on realseolife.com first, then portfolio-wide. Add Signature-Agent and Signature header capture to log ingestion and store a per-request verification result alongside bot_fingerprint_id. Report weekly, per property, the share of ai_crawler requests carrying a resolvable and valid signature against the share resting on the user agent alone. Cross the signature result against vendor IP range verification, which the warehouse already performs, so the two identity signals can be compared rather than assumed to agree.",
            "metric": "Share of ai_crawler requests carrying a verifiable HTTP Message Signature with a resolvable Signature-Agent key, against the share whose identity rests on the user agent string alone, weekly per property.",
            "properties": [
                "realseolife.com",
                "portfolio-wide"
            ],
            "window": "90 days from the first day the directory and the header capture are both live, reported weekly.",
            "status": "open",
            "falsified_if": "No request in 90 days carries a Signature-Agent header, which would mean the protocol has no deployed senders reaching us and the fingerprint remains the only identity available. Also falsified if signature presence shows no relationship to vendor IP range verification, since that would mean the signature adds no identity evidence beyond what the address already gives. A third failure mode: if the unsigned share turns out to be indistinguishable from the signed share on request pattern, method mix and page selection, then identity verification is buying no analytical separation even when it works.",
            "first_recorded": "2026-09-26",
            "last_reviewed": "2026-09-26",
            "related": [
                "webbotauth-httpsig-wg-draft-2026-09-01",
                "multi-vendor-identity-spoof-non-get-2026-09-25",
                "crawler-fingerprint-classification-drift-hypothesis",
                "datadome-bot-detection-failure-2026-09-23"
            ]
        },
        {
            "id": "robots-query-string-block-compliance-split-2026-10-02",
            "level": "demonstrated",
            "kind": "measurement",
            "claim": "A wildcard query-string robots.txt rule took OpenAI's crawlers to zero filter-URL requests, reduced Amazonbot by 92 percent, and had no restraining effect on Meta's crawler, which increased elevenfold with the rule in place.",
            "detail": "A rule reading Disallow: /*? was added to robots.txt on the BellyUp Atlanta category subdomains on 2026-09-11 at 23:30 UTC, following the crawl-trap diagnosis published 2026-09-09. The file was recovered from weekly backup archives stamped 2026-09-11 23:30, 2026-09-19 22:46 and 2026-09-26 04:07, so the wording and dates are evidence rather than recollection. Query-string requests on that network, measuring the full raw retention window before the change (2026-08-28 to 2026-09-11) against after it (2026-09-12 to 2026-10-02): GPTBot 63,649 to 0, OAI-SearchBot 19 to 0, Amazonbot 545 to 45, meta-externalagent 937 to 10,566. GPTBot continued making 575 clean-URL requests during the blocked period, so it kept visiting the site and skipped only the grid. Canonical tags pointing at the unfiltered hub were present on every filter URL throughout both windows and did not change. Disallow: /*? depends on wildcard syntax, a widely supported extension rather than part of the original robots exclusion standard, so indifference and an unimplemented pattern cannot be distinguished from server logs. The operational effect is identical either way.",
            "metric": "log_requests filtered to visitor_class ai_crawler on *.bellyupatl.com, counted by bot fingerprint and split on whether the requested URL carries a query string, either side of 2026-09-11",
            "properties": [
                "bars.bellyupatl.com",
                "nightlife.bellyupatl.com",
                "family.bellyupatl.com",
                "vegan.bellyupatl.com",
                "bbq.bellyupatl.com"
            ],
            "window": "15 days before the change against 21 days after it, inside the 35-day raw retention window",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "generated-robots-txt-overwrites-crawl-directives-2026-10-02",
                "vendor-not-purpose-predicts-parameter-enumeration-hypothesis",
                "the-bots-that-ask-permission-and-the-bots-that-dont"
            ]
        },
        {
            "id": "generated-robots-txt-overwrites-crawl-directives-2026-10-02",
            "level": "demonstrated",
            "kind": "measurement",
            "claim": "A per-site sitemap generator rewrites robots.txt from a fixed string on every run, so a hand-added crawl directive has an undeclared expiry and leaves no trace in live state once it lapses.",
            "detail": "The per-site deploy/generate-sitemap.php builds the sitemap and then writes robots.txt from a fixed string containing User-agent: *, Allow: / and a sitemap line, with no conditional and no merge. Checked 2026-10-02, none of the thirty BellyUp category subdomains carries the Disallow: /*? rule any longer: every file is the bare permissive version at 72 to 86 bytes. The Atlanta files were rewritten at 21:08 UTC, Charlotte and Tampa at 21:09, and Jacksonville, Orlando and St Augustine at 04:08 by the nightly cron. Thirty of the portfolio's one hundred and four sitemap generators of that name write robots.txt this way. The cron log confirms the generator has run on bellyupatl.com.bars every night since 2026-09-08, and the rule nonetheless survived from 2026-09-11 to 2026-10-01, so the overwrite is not purely a function of the nightly schedule and the exact trigger on 2026-10-02 at 21:08 is not established. The rule was recoverable only from weekly backup archives. No warehouse table records what crawl directives any site was serving on a given date, so an audit reading current state cannot distinguish a directive that was never added from one that was added and overwritten.",
            "metric": "robots.txt byte size, content and mtime across 30 BellyUp category subdomain docroots, against the robots.txt writing code in each deploy/generate-sitemap.php",
            "properties": [
                "30 BellyUp category subdomains",
                "30 of 104 portfolio sitemap generators"
            ],
            "window": "State as of 2026-10-02, against weekly backup archives from 2026-09-13, 2026-09-20 and 2026-09-27",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "robots-query-string-block-compliance-split-2026-10-02",
                "agentic-fetcher-target-selection-hypothesis"
            ]
        },
        {
            "id": "gptbot-rate-limited-eleven-hour-session-2026-10-02",
            "level": "demonstrated",
            "kind": "measurement",
            "claim": "A verified GPTBot crawl session ran at a fixed 1.999 requests per second for eleven consecutive hours and then stopped, twice in three days, and spent 99.99 percent of it on query-string filter combinations.",
            "detail": "On bars.bellyupjax.com, which declares 22 sitemap URLs and exposes 82,913 reachable filter combinations from twelve venues, GPTBot made 206,380 requests between 2026-09-27 and 2026-10-02. 206,356 carried a query string and 24 did not, reaching 17 distinct real URLs with each of the four Jacksonville bar guides read exactly once, the homepage three times and the sitemap six times. By hour: 2026-09-28 ran 7,038 to 7,198 requests per hour for eleven consecutive hours from 03:00 then dropped to 2,420 at 14:00 and 1 by 16:00; 2026-09-30 ran 7,183 to 7,198 per hour for eleven consecutive hours from 09:00 then stopped. 7,196 requests per hour is 1.999 per second, held within nine requests per hour across twenty-two hours on two separate days. Identity verified: 572,212 of 573,723 GPTBot-labelled portfolio requests in the window matched the twelve CIDR ranges at openai.com/gptbot.json fetched 2026-09-30 01:55:56 UTC. Network-wide across fifteen unprotected subdomains declaring 215 sitemap URLs in total, the three enumerating crawlers made 682,100 requests and were served 3,783 MB.",
            "metric": "log_requests for GPTBot on bars.bellyupjax.com, grouped by hour and by presence of a query string, plus bytes served and distinct URL counts",
            "properties": [
                "bars.bellyupjax.com",
                "15 BellyUp category subdomains"
            ],
            "window": "2026-09-27 to 2026-10-02, six complete log days",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "six-crawler-parameter-enumeration-split-2026-10-02",
                "canonical-does-not-control-crawling-2025-12-18"
            ]
        },
        {
            "id": "six-crawler-parameter-enumeration-split-2026-10-02",
            "level": "demonstrated",
            "kind": "measurement",
            "claim": "Under identical permissive robots.txt and identical canonical tags, three AI crawlers enumerated a query-parameter filter grid at above 94 percent of their requests and three made zero query-string requests, and the split cuts across our own crawler purpose classification.",
            "detail": "Across the fifteen Jacksonville, Orlando and St Augustine BellyUp category subdomains, 2026-09-27 to 2026-10-02, serving robots.txt reading Allow: / with no Disallow. Enumerated: GPTBot 565,258 requests at 99.9 percent query-string share and 3,137 MB, meta-externalagent 88,559 at 94.6 percent and 494 MB, Amazonbot 28,283 at 99.0 percent and 152 MB. Zero query-string requests: ClaudeBot 528, OAI-SearchBot 302, PerplexityBot 2. The warehouse bot_fingerprints table classifies ClaudeBot, GPTBot, Amazonbot and meta-externalagent all as crawler_purpose training, and OAI-SearchBot and PerplexityBot as ai_search, so a training crawler sits on the non-enumerating side alongside both ai_search crawlers. The site is held constant and only the crawler varies. The same vendor ordering appears independently in robots-query-string-block-compliance-split-2026-10-02, which reaches it through a rule change on one network rather than a comparison across crawlers.",
            "metric": "log_requests filtered to visitor_class ai_crawler on the fifteen unprotected subdomains, by bot fingerprint, split on query-string presence, with bytes served",
            "properties": [
                "15 BellyUp category subdomains across bellyupjax.com, bellyuporlando.com and bellyupstaugustine.com"
            ],
            "window": "2026-09-27 to 2026-10-02, six complete log days",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "crawler-mode-split-hypothesis",
                "vendor-not-purpose-predicts-parameter-enumeration-hypothesis",
                "robots-query-string-block-compliance-split-2026-10-02"
            ]
        },
        {
            "id": "canonical-does-not-control-crawling-2025-12-18",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google documents that rel=\"canonical\" on a faceted navigation URL may, over time, decrease the crawl volume of non-canonical versions, and ranks it below the two methods that prevent the fetch.",
            "detail": "Google's crawling documentation on managing faceted navigation URLs, last updated 2025-12-18 UTC, states that using rel=\"canonical\" to specify which URL is canonical \"may, over time, decrease the crawl volume\" of the non-canonical versions. It presents this as the weaker option beneath its two primary recommendations, preventing crawling via robots.txt or moving filter state into URL fragments, and separately recommends returning HTTP 404 for filter combinations that produce empty result sets. Our own measurement this week is a direct instance: every one of the 82,913 enumerated filter URLs on bars.bellyupjax.com carried a correct canonical to the unfiltered hub and was requested anyway, 206,356 times in six days. Three nonsense filter combinations tested on that site on 2026-10-02 each returned HTTP 200 with a full page rather than 404.",
            "metric": null,
            "properties": [
                "portfolio-wide, any property exposing a filter or sort control"
            ],
            "window": null,
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "source": "https://developers.google.com/crawling/docs/faceted-navigation",
            "related": [
                "gptbot-rate-limited-eleven-hour-session-2026-10-02",
                "fragment-filters-beat-robots-disallow-hypothesis"
            ]
        },
        {
            "id": "google-cross-product-crawling-docs-2026-10-02",
            "level": "established",
            "kind": "platform-documentation",
            "claim": "Google publishes its crawler and fetcher documentation as one cross-product surface covering Search, Gemini, Shopping, Ads and News, so a single robots.txt decision is documented as spanning search and AI product visibility together.",
            "detail": "developers.google.com/crawling scopes itself across multiple Google products rather than Search alone: Googlebot for Search, Google-Extended for Gemini apps and the Vertex AI API, Storebot-Google for Shopping, AdsBot-Google for ad quality checks, Googlebot-News for News, and Google-GeminiNotebook as a user-triggered fetcher for URLs a person supplies. Its sections cover managing crawler and fetcher interactions, verifying authentic Google requests, reducing crawl rates, managing crawling via robots.txt, and how crawling preferences affect visibility across Google products. Fetched 2026-10-02. No migration notice or migration date was visible on the page, so the date this consolidation happened is not established here.",
            "metric": null,
            "properties": [
                "portfolio-wide"
            ],
            "window": null,
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "source": "https://developers.google.com/crawling",
            "related": [
                "canonical-does-not-control-crawling-2025-12-18",
                "google-user-triggered-fetchers-outside-robots-2026-09"
            ]
        },
        {
            "id": "aipref-vocab-draft-08-2026-09-14",
            "level": "established",
            "kind": "standards-draft",
            "claim": "The IETF AI Preferences working group vocabulary draft 08 defines three usage categories, AI Training, AI Use and Search, and states explicitly that its contents do not reflect working group consensus.",
            "detail": "draft-ietf-aipref-vocab-08, published 2026-09-14 with an expiry of 2027-03-18, is a standards-track Internet-Draft of the IETF AI Preferences working group, authored by Paul Keller of Open Future with Martin Thomson of Mozilla as editor. It defines AI Training as using an asset to modify the learned parameters of a generative AI model, AI Use as using an asset as input to a generative AI model where the asset is not directly provided by the user, and Search as applications whose primary purpose is selecting and directing users to asset locations, conditioned on direct references and relevant excerpts. The draft states its contents do not reflect consensus of the working group either in whole or in part, so the category boundaries can still move. The three-way split is close to the training, ai_search and user_retrieval grouping our warehouse already uses.",
            "metric": null,
            "properties": [
                "portfolio-wide"
            ],
            "window": null,
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "source": "https://datatracker.ietf.org/doc/html/draft-ietf-aipref-vocab",
            "related": [
                "terms-txt-agentic-access-protocol-2026-09",
                "webbotauth-httpsig-wg-draft-2026-09-01",
                "robots-query-string-block-compliance-split-2026-10-02"
            ]
        },
        {
            "id": "openai-crawlers-render-javascript-2026-09-25",
            "level": "external-finding",
            "kind": "third-party-measurement",
            "claim": "An independent server-log analysis reports OpenAI's OAI-SearchBot and GPTBot began rendering pages with JavaScript on 2026-09-25, and that most of the resulting request increase was prefetch traffic rather than pages actually read.",
            "detail": "Hisashi Space, posting 2026-09-30, reports from server logs that OpenAI crawlers started rendering with JavaScript on 2026-09-25, measuring OAI-SearchBot up 5 to 15x on 2026-09-24 and 2026-09-25 on Next.js sites and GPTBot rendered pages up well over 100x between mid and late September. Method: user agents matched against OpenAI's published IP ranges, rendering identified by Next.js _rsc= prefetch parameters, and pages actually rendered counted by distinct referrers. The author's own caution is the valuable part: most of the 150,000-plus daily requests were prefetch rather than full renders, and the distinct-referrer count showed roughly 4,400 pages actually rendered by GPTBot on 2026-09-28. The author also notes rendering switched on only for some sites. Not verified by us and not claimed as the explanation for our own step change, which began 2026-09-27 on flat PHP sites with no prefetch mechanism and was driven by query-parameter enumeration instead.",
            "metric": null,
            "properties": [
                "external"
            ],
            "window": null,
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "source": "https://dev.to/hisashispace/traffic-from-chatgpt-jumped-is-it-because-its-crawlers-now-run-javascript-5961",
            "related": [
                "gptbot-rate-limited-eleven-hour-session-2026-10-02",
                "same-28-days-two-pulls-two-different-webs"
            ]
        },
        {
            "id": "fragment-filters-beat-robots-disallow-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "Moving a directory filter grid from query-string URLs to URL fragments, inside the generator rather than in robots.txt, will cut filter-URL requests from every enumerating crawler including those that ignore robots.txt, and will survive subsequent builds, where a Disallow: /*? rule achieves neither.",
            "detail": "Derived from robots-query-string-block-compliance-split-2026-10-02 crossed with generated-robots-txt-overwrites-crawl-directives-2026-10-02 and canonical-does-not-control-crawling-2025-12-18. The robots.txt rule had two failure modes measured on our own infrastructure: it did nothing to meta-externalagent, which went 937 to 10,566 filter requests with the rule live, and it was deleted by the build after three weeks. A fragment is never transmitted to the server, so there is no URL for any crawler to request regardless of its robots.txt handling, nothing to canonicalise, and nothing blocked, removed or hidden, which satisfies the portfolio additive never-block standard. Google's faceted navigation guidance lists fragments as one of its two primary recommendations.",
            "test": "In the generator that builds these subdomains, filter and sort controls write state into a URL fragment instead of a query string, the server renders the unfiltered listing for the bare URL, and filter combinations matching no venues return HTTP 404. robots.txt stays permissive on all three subdomains. bars.bellyupjax.com is treated; bars.bellyuporlando.com and bars.bellyupstaugustine.com are untouched controls. Measure filter-URL requests per crawler per day, clean-URL requests over the same days, and bytes served, for 14 days after against the 14 before, with the controls over identical dates. Check the fragment behaviour is still present in the served page after the next two nightly builds.",
            "metric": "Filter-URL requests per day per crawler for GPTBot, meta-externalagent and Amazonbot, with clean-URL requests and bytes served, plus a build-durability check on the served page",
            "properties": [
                "bars.bellyupjax.com",
                "bars.bellyuporlando.com",
                "bars.bellyupstaugustine.com"
            ],
            "window": "14 days after the change against the 14 days before it, plus durability checks on day two and day three",
            "status": "open",
            "falsified_if": "Filter-URL requests to the treated subdomain do not fall by at least 90 percent while the controls stay within their prior range, which would mean the grid was not what the crawlers were following. Falsified more interestingly if meta-externalagent keeps enumerating the treated subdomain, since that would mean it constructs query strings from something other than the links we serve. Also falsified if the fragment behaviour is absent from the served page after the next nightly build, which would make the generator change no more durable than the robots.txt edit. A fourth outcome is inconclusive rather than falsified: if no enumerating crawler visits any of the three subdomains inside the window, the single-day session pattern simply missed it.",
            "warehouse_support": "Over 2026-09-27 to 2026-10-02 the three subdomains logged 206,380, 63,995 and 48,534 GPTBot requests at 99.99, 99.98 and 99.98 percent query-string share, against 22, 19 and 17 declared sitemap URLs, with clean-URL requests of 24, 12 and 9. The treated baseline is large enough to detect a 90 percent reduction. The clean-URL counts are far too small to support a significance claim and must be read as direction only. The Atlanta result supplies the number to beat on the Meta side: 937 to 10,566 under a robots.txt rule.",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "robots-query-string-block-compliance-split-2026-10-02",
                "generated-robots-txt-overwrites-crawl-directives-2026-10-02",
                "canonical-does-not-control-crawling-2025-12-18"
            ]
        },
        {
            "id": "vendor-not-purpose-predicts-parameter-enumeration-hypothesis",
            "level": "hypothesis",
            "kind": "experiment-candidate",
            "claim": "Whether an AI crawler enumerates query-parameter combinations, and whether it honours a wildcard robots.txt query-string rule, are both stable per-vendor policies, and our crawler_purpose classification adds no predictive power over vendor identity for either.",
            "detail": "Two independent windows point the same way. Six crawlers split two to one on enumeration under identical permissive robots.txt across fifteen subdomains, and the Atlanta before-and-after produced the same vendor ordering through a rule change rather than a cross-crawler comparison. In both, ClaudeBot sits on the non-enumerating, compliant side despite being classified crawler_purpose training in our own warehouse alongside GPTBot, Amazonbot and meta-externalagent. If purpose were the operative variable, ClaudeBot and GPTBot would have behaved alike on the same URLs in the same week. They did not.",
            "test": "Add two stored weekly fields to the crawler rollup: a parameter-enumeration ratio per fingerprint per site, computed only on sites exposing at least 500 reachable query-string combinations, and a flag recording whether that site served a query-string Disallow in that week. Then compare the enumeration ratio grouped by vendor against the same ratio grouped by crawler_purpose, across four consecutive weeks and every portfolio property exposing a filter grid.",
            "metric": "Query-string share of requests per fingerprint per site per week, reported beside fingerprint crawler_purpose, vendor, and the served-robots-rule flag",
            "properties": [
                "portfolio-wide, restricted to properties exposing a filter or sort control"
            ],
            "window": "Four consecutive weeks once both fields exist",
            "status": "open",
            "falsified_if": "The enumeration ratio varies more within a vendor across properties than between vendors on the same property, which would make it a site-side or window-side effect rather than a vendor policy. Also falsified if the ratio groups more cleanly by crawler_purpose than by vendor, which would restore our existing classification as the better predictor. Partial results are expected and useful, since a vendor may enumerate on one site and not another where internal linking differs, which is why the metric is restricted to sites that actually expose a grid.",
            "warehouse_support": "Enumeration ratios of 99.9, 94.6 and 99.0 percent for GPTBot, meta-externalagent and Amazonbot against 0.0 percent for ClaudeBot, OAI-SearchBot and PerplexityBot on fifteen unprotected subdomains over 2026-09-27 to 2026-10-02, plus the Atlanta compliance split. That is one property family and cannot establish stability across unrelated sites. Neither stored field exists. The robots flag in particular cannot be backfilled, because generated-robots-txt-overwrites-crawl-directives-2026-10-02 establishes that overwritten robots.txt files are recoverable only from weekly backup archives.",
            "first_recorded": "2026-10-02",
            "last_reviewed": "2026-10-02",
            "related": [
                "six-crawler-parameter-enumeration-split-2026-10-02",
                "robots-query-string-block-compliance-split-2026-10-02",
                "crawler-mode-split-hypothesis"
            ]
        }
    ],
    "external_findings_note": "Deliberately empty at seed time. This ledger starts only with claims measured from our own warehouse. External findings are added by the weekly routine, each with a named primary source and the date it was verified, so that nothing enters the record on the strength of a summary alone.",
    "history": [
        {
            "date": "2026-09-08",
            "brief": "seed",
            "changes": "Ledger created with two measured observations from the Digital Karma warehouse and the two hypotheses derived from them. Levels 1 and 2 intentionally left empty for the routine to fill from verified primary sources."
        },
        {
            "date": "2026-09-08",
            "brief": "2026-09-08",
            "changes": "Added three established entries from primary vendor documentation fetched this run (Google's user-triggered fetcher class and the Google-NotebookLM retirement, Cloudflare's crawler purpose defaults effective 2026-09-15, and Google's EEA aggregator and supplier unit documentation), one external-finding entry for the Counter-GEO-Bench preprint which we have not reproduced, and one open hypothesis, agentic-fetcher-target-selection-hypothesis. Re-reviewed all four seed entries and updated last_reviewed on each. crawler-mode-split-hypothesis was re-checked against this week's vendor documentation and deliberately left at hypothesis with status open: Google and Cloudflare now describe the same training versus retrieval split we observed, which corroborates the framing but does not test the claim, so no promotion is warranted. A second hypothesis about robots.txt permissiveness was considered and not opened, because the snapshot carries no per-property robots.txt field to test it against; that gap is recorded inside the hypothesis that was opened. Nothing removed, renamed, or promoted."
        },
        {
            "date": "2026-09-12",
            "brief": "2026-09-12",
            "changes": "Added one demonstrated entry, crawler-rollup-not-reproducible-2026-09-11, recording that the 2026-09-11 warehouse extract disagrees with the 2026-09-08 extract by up to a factor of ten on AI crawler counts over windows sharing 25 of 28 days, while the Google Search Console series in the same two files reproduces to the digit. Added three external findings fetched this run: the Q2D-Web agentic retrieval benchmark, the SearchAtlas evidential query graph framework, and John Mueller's 2026-09-11 statement that the June gap in the Search Console page indexing report is permanent. Added one open hypothesis, crawler-fingerprint-classification-drift-hypothesis, which is a prerequisite for the three already open rather than a fourth independent question. No promotions. Re-reviewed five existing entries against the new snapshot and updated last_reviewed on each. Two of them now stand without supporting numbers: crawler-mode-split-hypothesis rested on Claude-User multiplying nearly twelvefold while ClaudeBot stayed flat, and the new extract puts Claude-User at 1,149 recent against 1,514 prior; agentic-fetcher-target-selection-hypothesis rested on the contrast between narrow-footprint fetchers such as Google-Agent on 5 properties and broad-footprint crawlers on 127 to 129, and Google-Agent has no rows in the new extract at all. Both are deliberately left at hypothesis with status open and nothing rewritten, because a working theory losing its evidence is a fact about the record that the record should keep. The demonstrated entry ai-crawler-vendor-expansion-2026-08 is likewise left exactly as published, with the new entry filed beside it and pointing at it; Krisada may want to decide by hand whether it should be downgraded. Nothing removed, renamed, or promoted. Deliberately not added: a second hypothesis about which AI crawlers favour which property types, because every number that would motivate it comes from the same fingerprint rollup this week's finding puts in question."
        },
        {
            "date": "2026-09-23",
            "brief": "2026-09-23",
            "changes": "Added seven entries and re-reviewed eight. The run's own finding is crawler-rollup-reproduces-2026-09-18, a demonstrated entry recording that the 2026-09-18 extract reproduces the 2026-09-11 extract on nine of the eleven fingerprints this ledger recorded verbatim, all within 0.85x to 1.29x, with Claude-User four requests apart across a week, and that on every fingerprint where 2026-09-08 disagreed with 2026-09-11 the third extract sided against 2026-09-08. The two fingerprints outside the band, Applebot-Extended and Claude-SearchBot, are accounted for by the newer file's own prior-window counts and read as a real fall-off at a window boundary. Google-Agent is now absent for a second consecutive extract. Three established entries added from primary documentation fetched this run: google-search-profiles-badge-2026-09-16 and google-aggregator-supplier-units-local-business-2026-09-18, the latter filed beside google-eea-aggregator-supplier-units-2026-09 rather than written into it. Two external findings added, conversational-platform-domain-preference-2026-09 and terms-txt-agentic-access-protocol-2026-09, plus gsc-crawl-stats-gap-2026-09-15. One hypothesis opened, fetcher-property-preference-stability-hypothesis, which carries the full per-property fingerprint baseline verbatim so that next week's extract has a fixed reference to fail against; it is written as a stability test before a preference test because crawler-fingerprint-classification-drift-hypothesis remains its prerequisite. No promotions. crawler-fingerprint-classification-drift-hypothesis was re-checked and deliberately left at hypothesis with status open even though this week's evidence runs against continuous drift, because its own falsification clause requires two extracts whose recent windows overlap by at least 25 of 28 days and these were seven days apart giving 21, asks for agreement within 10 percent where this is within 30, and the extract still carries no snapshot id, no unclassified residual and no fingerprint dictionary version, so the fault still cannot be located in classification rather than ingestion. ai-crawler-vendor-expansion-2026-08 is again left exactly as published, now contradicted by two later extracts rather than one; the decision to downgrade it remains Krisada's by hand. realseolife-publish-burst-impressions-2026-08 and publish-burst-surface-hypothesis were advanced with four more weeks of data, impressions 1,936, 2,561, 2,669 and 2,084 against average positions 56.97, 54.79, 55.63 and 56.71, and 28-day impressions of 8,493 against 1,532 prior while clicks went 2 against 3; the hypothesis stays open because its test needs a second property and the snapshot carries no publishing-volume field to identify one. Deliberately not opened: a hypothesis on whether agentic fetch volume follows structured data coverage, blocked for the third week running by the absence of per-property robots.txt, CDN and schema coverage fields, which is recorded here as a standing instrumentation request rather than a fifth open question. Nothing removed, renamed, or promoted."
        },
        {
            "date": "2026-09-26",
            "brief": "2026-09-26",
            "changes": "Nine entries added and four re-reviewed. The run's own work is three demonstrated entries and one falsification. multi-vendor-identity-spoof-non-get-2026-09-25 records that 902 of the 1,065 non-GET requests classified as declared AI crawlers over the 28 days ending 2026-09-25 came from fifteen addresses that each presented nine to eleven vendors' crawler identities and ten to fifteen fingerprint names on one day, all in Google Cloud ranges, and that 91 of those requests used DELETE, which no crawler has reason to send to a content site. It is filed with its own limit stated: 0.33 percent of declared AI crawler volume, and no evidence whatever about the remaining 317,280 GET requests. rank-order-crawler-mix-test-measures-margin-2026-09-25 and rolling-window-turns-single-week-burst-into-trend-2026-09-25 together dismantle last week's own instrument. On the strength of them, fetcher-property-preference-stability-hypothesis is closed as falsified at test two of three on its own stated criterion, with four of seven properties reshuffling their top three; its claim, detail, test and frozen baseline are preserved verbatim and a status_history records the prior value rather than discarding it. The falsification is the less useful half. The useful half is that every property whose order held had its leading crawler ahead by at least 3.64 times and every property that reshuffled had its leaders within 1.71 times, so the criterion was measuring margin width, and that the hypothesis's headline example, Claude-User 587 against 1, was 586 requests inside the single week of 2026-08-22 to 2026-08-28 that a rolling 28-day window carried as a standing pattern for four weeks and then reported as a collapse. conversational-platform-domain-preference-2026-09 is explicitly held at external-finding and not downgraded, because our instrument failing says nothing about the arXiv result it came from. Two established entries added from primary sources fetched this run, webbotauth-httpsig-wg-draft-2026-09-01 and google-mediapartners-generalised-2026-09-17, plus google-videoobject-creator-interactionstatistic-2026-09-24. Two external findings added, akamai-ai-crawler-post-shift-2026-09-22 and datadome-bot-detection-failure-2026-09-23, both flagged as trade press summaries of vendor reports we could not fetch at source. One hypothesis opened, signed-agent-identity-separates-spoof-from-shift-hypothesis, which is the first open question in this ledger that proposes instrumentation we can actually build rather than waiting on a warehouse field that does not exist. crawler-fingerprint-classification-drift-hypothesis re-reviewed and left open: this week supplies a mechanism by which one actor can inflate many fingerprints at once, but at 0.33 percent of volume it is far too small to explain the swings the hypothesis was opened for, and its three requested instrumentation fields still do not exist. publish-burst-surface-hypothesis re-reviewed and left open with adjacent evidence only, since RealSEOLife.com's crawler volume rose sharply in the week it gained a large amount of new surface but that week also contains two portfolio-wide server changes and the hypothesis concerns impressions against position rather than crawler requests. No promotions. Nothing removed or renamed."
        },
        {
            "date": "2026-10-02",
            "run": "realseolife-weekly-brief",
            "brief": "https://www.realseolife.com/data/research/briefs/2026-10-02.json",
            "seed_article": "https://www.realseolife.com/article/gptbot-obeyed-robots-txt-meta-ignored-it",
            "experiment": "https://www.realseolife.com/experiments/robots-block-removed-does-the-crawler-return",
            "note": "Added 4 demonstrated entries from our own verified measurement (the Atlanta robots.txt compliance split across four crawlers, the generator overwriting robots.txt on 30 subdomains, the hour-by-hour GPTBot rate-limited session, and the six-crawler enumeration split), 3 established entries from primary documentation fetched this run, 1 external-finding, and 2 open hypotheses. Re-reviewed and added review notes to crawler-mode-split-hypothesis, crawler-fingerprint-classification-drift-hypothesis, agentic-fetcher-target-selection-hypothesis and publish-burst-surface-hypothesis. No promotions. Nothing removed. The redistribution question from bellyupatl-crawler-architecture-redistribution is explicitly recorded as still unanswered: real-page crawling on the Atlanta network roughly doubled after the block, but ambient portfolio GPTBot volume rose about twentyfold over the same window and the two candidate controls had no before-window, so the rise cannot be attributed to the block. A new experiment record was opened because the build removing the block on 2026-10-02 supplies the control that was missing."
        }
    ]
}
