{
  "id": 1243,
  "url": "https://arxiv.org/abs/2606.10064v1",
  "title": "Bittensor Agent Arenas as a Trajectory Primitive: Distilling a Shopping Agent from ShoppingBench Subnet Traces",
  "summary": "Small-model agentic post-training is bottlenecked less by the algorithm than by the trajectory substrate it consumes. Leading recipes (RLVR, group-relative RL, rejection-sampled re-SFT) all need multi-turn traces carrying per-trajectory supervision, and the two existing sources fall short: frontier-synthesised data inherits the synthesizer's biases and collapses the long tail, while unfiltered production logs are unjudged and contaminated by shortcut behaviour. We argue that an incentive-aligned",
  "authors": "Shardul Bansal, Seth Schilbe, Jarrod Barnes",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-08T18:39:15.000Z",
  "fetched_at": "2026-07-14T14:15:07.845Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1243",
  "original_url": "https://arxiv.org/abs/2606.10064v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}