{
  "version": "2026-09-13-r2",
  "title": "Secondary extraction: repeated-run stability in AI Search",
  "checked_at": "2026-09-13T00:00:00+07:00",
  "primary_source": {
    "title": "Don't Measure Once: Measuring Visibility in AI Search (GEO)",
    "url": "https://arxiv.org/abs/2604.07585",
    "version": "arXiv:2604.07585v1",
    "published": "2026-04-08",
    "publication_status": "preprint; not peer reviewed"
  },
  "supporting_source": {
    "title": "Quantifying Uncertainty in AI Visibility",
    "url": "https://arxiv.org/abs/2603.08924",
    "publication_status": "preprint; not peer reviewed"
  },
  "scope": "LUMESEO transcribed selected aggregate values from Tables 5 and 6 of the paper and calculated simple complements/differences for Vietnamese readers. This file is not the original response-level dataset and is not a new ChatGPT experiment by LUMESEO.",
  "study_context": {
    "engines": ["ChatGPT", "Perplexity", "Gemini", "Google AI Mode"],
    "server_location": "Switzerland",
    "maximum_simultaneous_repetitions": 10,
    "pair_window": "<=24 hours",
    "source_pairs_rule": "Only runs with at least one extracted citation; both-empty pairs excluded",
    "brand_pairs_rule": "Campaigns with mean brand detection rate >=70%; both-empty pairs excluded",
    "source_pairs_all_engines_after_filter": 3409,
    "brand_pairs_table_5_total": 3495,
    "chatgpt_specific_pair_count_table_6": null,
    "denominator_note": "The paper reports 3,409 source pairs after filtering across four engines. Table 5 reports 1,235 + 1,027 + 1,233 = 3,495 brand pairs across three campaigns. Table 6 does not expose a ChatGPT-specific pair count, so the ChatGPT source and brand means do not have one shared public denominator."
  },
  "table_5_brand_similarity_by_campaign": [
    {"campaign": "Consumer Electronics", "pairs": 1235, "max_runs": 10, "jaccard_mean": 0.477, "jaccard_sd": 0.298, "rbo_mean": 0.229},
    {"campaign": "Sporting Goods", "pairs": 1027, "max_runs": 10, "jaccard_mean": 0.327, "jaccard_sd": 0.298, "rbo_mean": 0.175},
    {"campaign": "Telecommunications", "pairs": 1233, "max_runs": 10, "jaccard_mean": 0.463, "jaccard_sd": 0.301, "rbo_mean": 0.220}
  ],
  "table_6_similarity_by_engine": [
    {"engine": "ChatGPT", "source_jaccard_mean": 0.233, "source_rbo_mean": 0.088, "brand_jaccard_mean": 0.437, "brand_rbo_mean": 0.192},
    {"engine": "Perplexity", "source_jaccard_mean": 0.282, "source_rbo_mean": 0.102, "brand_jaccard_mean": 0.492, "brand_rbo_mean": 0.202},
    {"engine": "Gemini", "source_jaccard_mean": 0.505, "source_rbo_mean": 0.230, "brand_jaccard_mean": 0.409, "brand_rbo_mean": 0.196},
    {"engine": "Google AI Mode", "source_jaccard_mean": 0.318, "source_rbo_mean": 0.254, "brand_jaccard_mean": 0.375, "brand_rbo_mean": 0.238}
  ],
  "derived_values": {
    "chatgpt_source_set_non_overlap": 0.767,
    "chatgpt_brand_set_non_overlap": 0.563,
    "chatgpt_brand_minus_source_jaccard": 0.204,
    "difference_label": "descriptive difference across differently filtered cohorts; not a matched effect",
    "derivation": "non_overlap = 1 - Jaccard mean; descriptive difference = brand Jaccard mean - source Jaccard mean",
    "warning": "Source and brand cohorts use different eligibility filters and Table 6 does not publish the ChatGPT-specific pair counts. The 0.204 difference is descriptive only, not a paired effect or causal estimate. The complements are not the probability that one named brand or URL disappears on the next run."
  },
  "availability_note": "The paper states that code and data are available at its linked GitHub repository. At LUMESEO's 2026-09-13 check, the repository cloned as empty, so LUMESEO did not claim a response-level recomputation. Values here are table-level transcription plus disclosed arithmetic.",
  "limitations": [
    "The source study used Swiss-server context, not Vietnamese users or Vietnamese prompts.",
    "Campaigns were consumer categories; findings may not transfer to B2B supplier discovery.",
    "The two cited arXiv studies are preprints and had not been peer reviewed at the snapshot date.",
    "The 3,409 source pairs and 3,495 Table 5 brand pairs are different public denominators; Table 6 does not disclose a ChatGPT-specific pair count.",
    "Jaccard measures set overlap and RBO measures ranked-list overlap; neither is recommendation quality.",
    "ChatGPT web search activation varied, and source similarity used cited runs only.",
    "Platform behavior and model versions may have changed after the collection window.",
    "LUMESEO did not independently recompute the response-level dataset because the linked repository was empty at check time."
  ]
}
