{
  "id": 18734,
  "url": "https://arxiv.org/abs/2510.10315",
  "title": "Is Misinformation More Open? A Study of robots.txt Gatekeeping on the Web",
  "summary": "arXiv:2510.10315v4 Announce Type: replace Abstract: Large Language Models (LLMs) are increasingly relying on web crawling to stay up to date and accurately answer user queries. These crawlers are expected to honor robots.txt files, which govern automated access. In this study, for the first time, we investigate whether reputable news websites and misinformation sites differ in how they configure these files, particularly in relation to AI crawlers. Analyzing a curated dataset, we find a stark co",
  "authors": "Nicolas Steinacker-Olsztyn, Devashish Gosain, Ha Dao",
  "category": "research",
  "topics": "misinformation,agents-autonomy,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-13T04:00:00.000Z",
  "fetched_at": "2026-08-13T05:10:37.786Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18734",
  "original_url": "https://arxiv.org/abs/2510.10315",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}