{
  "id": 758,
  "url": "https://arxiv.org/abs/2606.21627v1",
  "title": "Counsel: A Meta-Evaluation Dataset for Agentic Tasks",
  "summary": "As agentic systems tackle increasingly complex multi-step tasks, evaluating their trajectories presents a major bottleneck - human annotation of a single trajectory on popular agentic benchmarks can take hours, making it difficult to scale evaluations for measuring performance or curating training data. This has driven widespread reliance on automated approaches such as LLM-as-a-judge (LLMJ) to critique agents at the process and outcome-levels at scale, however, the soundness of LLMJ critiques o",
  "authors": "Sashank Pisupati, Henry Broomfield, Eujeong Choi, Antonia Calvi, Charlie Wang, Roman Engeler et al.",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-06-19T17:30:56.000Z",
  "fetched_at": "2026-07-14T14:14:46.035Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/758",
  "original_url": "https://arxiv.org/abs/2606.21627v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}