{
  "count": 50,
  "items": [
    {
      "id": 19138,
      "url": "https://arxiv.org/abs/2608.13049",
      "title": "H2R-Bench: Benchmarking Human-to-Robot Manipulation Video Generation in World Models",
      "summary": "Large-scale manipulation data is essential for robot learning, yet collecting robot demonstrations remains expensive and difficult to scale. Meanwhile, abundant egocentric human manipulation videos provide rich behavioral experiences, but transferring them across embodiments remains challenging due to differences between human hands and robotic end-effectors. Recent advances in video world models offer a promising pathway to synthesize robot-centric manipulation videos from human observations, w",
      "authors": "Dingyi Rong, Yue Shi, Chaofan Ma, Jiezhang Cao, Zongrui Wang, Zeyu Zhang",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19138"
    },
    {
      "id": 19139,
      "url": "https://arxiv.org/abs/2608.13489",
      "title": "DreamX-Phi 1.0: Action-Conditioned Video World Model for Robotic Manipulation",
      "summary": "We present DreamX-Phi 1.0, an action-conditioned video world model for robotic manipulation that, given an observed frame, a language instruction, and a prescribed action sequence comprising end-effector poses and gripper states, predicts the resulting future observations. Yet realism alone does not guarantee faithfulness: a convincing rollout can still move the wrong arm or lose the manipulated object. To ensure the prediction respects each arm's commanded path, we inject per-arm SE(3) transfor",
      "authors": "DreamX Team, Rui Chen, Xiangxiang Chu, Geng Li, Jifan Li, Qingfeng Shi",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19139"
    },
    {
      "id": 19140,
      "url": "https://arxiv.org/abs/2608.12743",
      "title": "Spatial Memory Agent: Experience-Grounded Procedure Memory for Spatial Intelligence",
      "summary": "Spatial intelligence is becoming a foundation for embodied agents, robotic planning, and multimodal assistants. To improve the spatial reasoning ability of VLM agents, existing work has mainly followed two lines. One line uses post-training methods, such as supervised fine-tuning and reinforcement learning. Another line adopts an agentic paradigm in which the model calls external spatial tools, such as depth estimation and 3D reconstruction tools, to gather intermediate spatial evidence. We stud",
      "authors": "Haokai Zhang, Yuhang Ding, Yunshu Zhou, Xinze Du, Shengtao Zhang, Zhiyue Zhao",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19140"
    },
    {
      "id": 19142,
      "url": "https://arxiv.org/abs/2608.13560",
      "title": "AutoDesign: Meta-Harness Optimization for Long-Horizon Agentic Design",
      "summary": "Transforming multimodal sources into condensed and structured media outputs can be fundamentally conceptualized as a long-horizon agentic process centered on a model-harness system. While an ideal harness system should align with human design priors and accumulate reusable experience through empirical exploration to drive recursive self-improvement, existing paradigms remain static and fall short of this capability. In this paper, we present AutoDesign, a framework that aligns with human design",
      "authors": "Yaxin Luo, Haobin Jiang, Jialv Zou, Xu Huang, Wenhao Yan, Haodong Li",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19142"
    },
    {
      "id": 19143,
      "url": "https://arxiv.org/abs/2608.12990",
      "title": "LycheeMemory V2: Efficient Long-Term Memory for LLM Agents via Semantic Segment-Level Consolidation",
      "summary": "Long-horizon LLM agents must preserve information from past interactions to support future tasks. Existing memory systems typically rely on eager consolidation, invoking LLMs after each interaction to extract, summarize, or update memories. This design makes memory construction increasingly costly as conversations grow. Coarse summarization can reduce construction cost but risks discarding fine-grained contextual evidence, whereas larger retrieval contexts or multi-hop LLM reasoning shift the ov",
      "authors": "Dongfang Li, Zixuan Liu, Junmai Wang, Jiahe Huang, Fuhao Li, Bonian Jia",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19143"
    },
    {
      "id": 19144,
      "url": "https://arxiv.org/abs/2608.13552",
      "title": "PlayWorld: Benchmarking World Models with Agent Players over Long-Horizon Objectives",
      "summary": "Video world models simulate future states conditioned on current observations and user actions. Recent systems have demonstrated impressive video consistency and action controllability over long sequences. However, fairly comparing these interactive models remains challenging. In practice, a human player typically evaluates a world model by pursuing long-horizon objectives through interaction. For example, a user may turn around 360 degrees to see whether the environment remains consistent, or w",
      "authors": "Kaixin Ding, Xi Chen, Minghong Cai, Zhiyuan Xu, Yiyang Wang, Yuxiang Lu",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19144"
    },
    {
      "id": 19146,
      "url": "https://arxiv.org/abs/2608.13505",
      "title": "Intern-S2-Preview: Scientific Agentic Foundation Model",
      "summary": "Scientific discovery increasingly requires AI systems that can reason over scientific evidence of heterogeneous modalities, interact with scientific tools and environments, and sustain progress across long task horizons. We present Intern-S2-Preview, a series of scientific agentic foundation models designed to support multimodal scientific understanding, reasoning, generation, and long-horizon tasks. The training pipeline begins with scientific multimodal pre-training over rendered scientific do",
      "authors": "Lei Bai, Jiaqi Cao, Chiyu Chen, Guanzhou Chen, Kai Chen, Guangran Cheng",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19146"
    },
    {
      "id": 19503,
      "url": "https://arxiv.org/abs/2608.13391",
      "title": "Context-Matched Distillation: Teacher Causality for Autoregressive Video Distillation",
      "summary": "Interactive autoregressive video generation demands both low-latency rollouts and precise online control. Few-step distillation accelerates generation by reducing denoising steps, while online control imposes a causal constraint: frames and blocks should depend on history and controls available during generation. Existing video distribution matching distillation (DMD) pipelines, however, often supervise causal few-step students using bidirectional teachers that score complete clips. The score fo",
      "authors": "Hmrishav Bandyopadhyay, Xuanchi Ren, Zijian Huang, Jay Zhangjie Wu, Tianshi Cao, Ruilong Li",
      "category": "research",
      "topics": "children-education",
      "published_at": "2026-08-12T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19503"
    },
    {
      "id": 18743,
      "url": "https://arxiv.org/abs/2608.10450",
      "title": "Persistent Recursive Worlds Enable Autonomous Software Evolution",
      "summary": "Complex software systems develop over timescales that exceed the lifespan of any individual coding agent. Most agentic software systems preserve continuity through persistent sessions, memories, managers or shared context. We introduce EvoX Genesis (hereafter, Genesis), which instead makes the software project persistent while allowing local agents to remain finite-lived. Genesis represents software as a persistent recursive world: each local world is situated by an accepted version and a reposi",
      "authors": "Beichen Huang, Zhenyu Liang, Bowen Zheng, Ran Cheng",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18743"
    },
    {
      "id": 18748,
      "url": "https://arxiv.org/abs/2608.12307",
      "title": "AI4AI at Test-Time: Strong-to-Weak Capability Transfer via Harnesses",
      "summary": "Recent work on distillation transfers the capabilities of large models to smaller ones often by updating the latter's parameters, through teacher forcing, on-policy distillation, and related training-time methods. In this paper, we ask whether such transfer can instead occur at test time. We study strong-to-weak scaffolding: whether a stronger builder model can construct inference-time harnesses that help a weaker target model solve tasks more reliably without any parameter updates. Using four r",
      "authors": "Cheng Qian, Wenting Zhao, Liangwei Yang, Heng Wang, Jielin Qiu, Heng Ji",
      "category": "research",
      "topics": "regulation",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18748"
    },
    {
      "id": 18749,
      "url": "https://arxiv.org/abs/2608.11616",
      "title": "MBA: Multimodal Benchmark and Agents for Real-World Business Ideation",
      "summary": "Agentic systems powered by large language models (LLMs) have opened new opportunities for business ideation. Yet existing approaches remain confined to a text-only paradigm, despite the inherently multimodal nature of real-world contexts. We thus introduce MBA-Bench, the first multimodal benchmark for training and evaluating business ideation agents, comprising 30K samples across six domains, each domain characterized by distinct visual cues not fully conveyed by text alone. Concretely, we autom",
      "authors": "Hojun Choi, Jaeyo Shin, Suin Lee, Hyunjung Shim",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18749"
    },
    {
      "id": 18750,
      "url": "https://arxiv.org/abs/2608.12313",
      "title": "AVA-Encoder: Towards Agent-Native Video Representation Learning",
      "summary": "Creative agents still lack an effective way to learn from high-quality human films, limiting their ability to produce cinematic-grade videos. A key challenge is the absence of a structured video representation that is both faithful to film content and directly usable for agentic reasoning and manipulation. To address the challenge, we propose the Agentic Video Auto-Encoder (AVA-Encoder), a framework for learning agent-native video representations via agentic auto-encoding. AVA-Encoder transforms",
      "authors": "Chuyue Li, Jinpeng Yu, Haozhe Wang, Tian Xueyun, Zhijing Zhang, Bingnan Li",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18750"
    },
    {
      "id": 18752,
      "url": "https://arxiv.org/abs/2608.11924",
      "title": "Spark-to-Paper: End-to-End Research Paper Generation as a Composable Skill",
      "summary": "Turning a research idea into a complete paper requires more than text generation: the system must retrieve literature, design and execute experiments, revise claims according to evidence, produce publication-ready figures, and maintain consistency across a long generation process. We present Spark-to-Paper, an end-to-end research paper generation system implemented as thirteen composable skills inside an existing coding assistant, without requiring a separate agent platform or orchestration serv",
      "authors": "Zhuoyang Qian, Biao Wu, Yiran Wang, Chris D Yan, Desan Dai, Liangwei Zheng",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18752"
    },
    {
      "id": 18753,
      "url": "https://arxiv.org/abs/2608.11878",
      "title": "ToolHazard: Scaling Adversarial Environments for Security Evaluation and Alignment of LLM-based Agents",
      "summary": "Large language model (LLM) agents integrated with external tools are vulnerable to indirect prompt injections embedded in environmental states. However, existing studies largely rely on manually implemented or reused environments, stochastic LLM-based tool simulation, and predefined injection locations, limiting scalable security research across broader domains. To bridge this gap, we propose **ToolHazard**, a scalable adversarial environment synthesis framework that reduces human engineering an",
      "authors": "Yutao Mou, Pengfei Yang, Zhe Yin, Zhangchi Xue, Xiaotian Luan, Dingyao Yu",
      "category": "research",
      "topics": "safety-alignment,agents-autonomy,environment",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18753"
    },
    {
      "id": 19148,
      "url": "https://arxiv.org/abs/2608.12036",
      "title": "Mechanist: AI as a Scientific Instrument for Discovering the Mechanisms of Intelligence",
      "summary": "AI models have achieved remarkable success across diverse domains, yet the mechanisms underlying their capabilities and the risks they may pose remain poorly understood. As AI development becomes faster and increasingly automated, mechanistic exploration remains largely manual, widening the gap between what models can do and our ability to understand and control them. To bridge this gap, we introduce Mechanist, an agentic system that uses AI as a scientific instrument for the autonomous discover",
      "authors": "Mengru Wang, Junfeng Fang, Shuofei Qiao, Zhenqian Xu, Haoming Xu, Haoxiong Wang",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19148"
    },
    {
      "id": 19149,
      "url": "https://arxiv.org/abs/2608.12123",
      "title": "Ready Cohorts: Bounding GPU Opportunity and Avoiding Host Round Trips in LLM-Agent Control",
      "summary": "LLM-agent services repeatedly execute small deterministic transitions between model and tool calls: route an outcome, update state, and emit the next effect. We ask when this control path exposes enough concurrent work for GPU execution, and what changes when a GPU-computed route decision remains on device. We formalize the ready-cohort boundary using fixed-partition share F, exact offline share P*, local upper bound U, and online achieved share A. Under zero service time, unlimited capacity, an",
      "authors": "Josef Liyanjun Chen",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19149"
    },
    {
      "id": 19151,
      "url": "https://arxiv.org/abs/2608.11574",
      "title": "Hand Visibility Detector: Per-Keypoint Visibility Estimation for Hands",
      "summary": "Hand Pose Estimation (HPE) is a fundamental technology for various applications such as AR/VR and robotics. In these applications, the visibility of each hand joint in the image is crucial for assessing the reliability of estimation results under occlusion. However, most existing HPE methods output joint positions without explicitly indicating their visibility. Although some methods account for occlusion or visibility, visibility estimation has mainly been used as an auxiliary signal for improvi",
      "authors": "Ryosei Hara, Masashi Hatano, Rintaro Yanagi, Atsushi Hashimoto, Takuma Yagi, Mariko Isogawa",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19151"
    },
    {
      "id": 19506,
      "url": "https://arxiv.org/abs/2608.11660",
      "title": "Hybrid-Policy Self-Editing for Composable Unstructured Knowledge Editing",
      "summary": "Large language models (LLMs) achieve remarkable performance across natural language tasks, yet they are trained on static corpora and their knowledge quickly becomes outdated in a fast-changing world. This motivates knowledge editing (KE), which updates specific knowledge in an LLM without changing unrelated others. Recent works move from structured knowledge triples toward unstructured KE (UKE), where the edit is a free-form passage that may state multiple facts at once. Nonetheless, existing e",
      "authors": "Tianci Liu, Zihan Dong, Tianchun Li, Yi-Chung Chen, Qiming Cao, Xingchen Wang",
      "category": "research",
      "topics": "regulation",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19506"
    },
    {
      "id": 19507,
      "url": "https://arxiv.org/abs/2608.12440",
      "title": "Specification-first convergence with an AI coding agent: a case study of dismantling a core architectural invariant across 189 files in a 717k-line codebase with no test oracle and no human code review",
      "summary": "This paper reports a single, fully instrumented case study of a large-scale architectural refactoring by an AI coding agent under a specification-first protocol, with no human review of the generated code and no pre-existing oracle to validate the target behaviour. The task, dismantling a central invariant across a large interdependent codebase, was assessed by the author as effectively infeasible through incremental refactoring, the kind of change that conventionally calls for a rewrite instead",
      "authors": "Joel Abenhaim",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-11T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19507"
    },
    {
      "id": 18376,
      "url": "https://arxiv.org/abs/2608.10812",
      "title": "Reference-Free Post-Training of Open Large Language Models for Multilingual Machine Translation",
      "summary": "We study reference-free post-training for multilingual machine translation with open large language models. Starting from the supervised-finetuned MiLMMT-46-v0.1 models, we apply Group Relative Policy Optimization (GRPO) with a reward that averages two reference-free quality estimation models and is gated by language identification. We then linearly interpolate the supervised fine-tuning (SFT) and reinforcement learning (RL) model checkpoints to obtain MiLMMT-46-v1.0. Across 46 languages, the re",
      "authors": "Chris Han, Pengzhi Gao, Pei Fu, Jian Luan",
      "category": "research",
      "topics": "regulation",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18376"
    },
    {
      "id": 18377,
      "url": "https://arxiv.org/abs/2608.10915",
      "title": "ComBodied Agents: a New Paradigm of Human-Centric Agentic AI",
      "summary": "After an older adult misses a medication dose, a software agent can send another reminder and an embodied agent can bring the medication. Yet neither explains whether the person forgot, is confused, has side effects, or deliberately refused, nor what support is appropriate. This reveals a structural gap in Agentic AI: Digital Agents primarily transform software states, while Embodied Agents transform physical states; neither makes a person's evolving state and agency the primary object of modeli",
      "authors": "Qianggang Ding, Xingyao Wang, Rui Feng, Zhibin Wang, Feixiang Wang, Kelong Mao",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18377"
    },
    {
      "id": 18378,
      "url": "https://arxiv.org/abs/2608.11079",
      "title": "SkillZip: Evaluation-Free Skill Compression for Self-Evolving Agents by Discovering Reusable Structure",
      "summary": "Self-evolving agents accumulate reusable skills by appending successful procedures and failure fixes. Over time, the same requirement is often restated in several branches, examples, and warnings, while common action sequences are copied rather than reused. The resulting skill becomes expensive to inject and difficult to maintain. Generic prompt compression is ill-suited to this setting because a skill is not a flat passage: its name and description define when it applies, its workflow controls",
      "authors": "Xiaofan Bai, Hongqiang Lin, Chao Liu, Yantao Zhang, Xuan Jin, Xipeng Cao",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18378"
    },
    {
      "id": 18380,
      "url": "https://arxiv.org/abs/2608.10366",
      "title": "DSAgentBench: Can Agents Automate End-to-End Data-Science Workflows in Real Computer Environments?",
      "summary": "Real-world data science involves long-horizon workflows that span data wrangling, exploration, modeling, visualization, and validation, and require coordinated use of tools such as notebooks, IDEs, terminals, browsers, and databases within real operating environments. Yet existing benchmarks lack real-computer interaction and do not evaluate whether agents can execute complete end-to-end data-science workflows in realistic computing environments, failing to capture the multi-stage, multi-tool na",
      "authors": "Mizanur Rahman, Mohammed Saidul Islam, Ridwan Mahbub, Md Tahmid Rahman Laskar, Shafiq Joty, Enamul Hoque Prince",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18380"
    },
    {
      "id": 18381,
      "url": "https://arxiv.org/abs/2608.10875",
      "title": "VibeLifeBench: Can Your Life Agent Be Proactive and Persistent in a Living World?",
      "summary": "Large language model (LLM) agents are increasingly deployed as personal assistants. Existing evaluations, however, mostly use short, self-contained requests in static environments. Everyday life assistance is different. A task runs for weeks rather than minutes. The world keeps changing while the agent is not being prompted. Many constraints are never stated outright. An agent that merely answers the request in front of it will fail at such a task. What is needed instead is an agent that stays p",
      "authors": "Xiaohongshu Inc",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18381"
    },
    {
      "id": 18751,
      "url": "https://arxiv.org/abs/2608.11274",
      "title": "Agent Safety Should Be a Runtime Contract",
      "summary": "The dominant paradigm treats AI safety as a property to be instilled during model training via RLHF, DPO, or Constitutional AI. We argue this is structurally insufficient for autonomous agents that execute code, mutate files, send messages, and modify databases. Agent safety should be a runtime contract enforced by the harness, and the contract has two complementary faces. The preventive face blocks dangerous actions before they happen via sandboxes, permission gates, output filters, and traject",
      "authors": "Albus W. Ng, Yi Han, Jusheng Zhang, Wenhao Wang",
      "category": "research",
      "topics": "safety-alignment,agents-autonomy",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18751"
    },
    {
      "id": 18754,
      "url": "https://arxiv.org/abs/2608.10628",
      "title": "InSight-doc: Agentic Visual Perception for Long-Document Understanding",
      "summary": "Long-document understanding often requires reasoning over many visually rich pages, making inference costly and prone to context rot. In this work, we propose InSight-doc, an agentic visual perception framework that treats visual resolution as an adaptive reasoning-time resource. InSight-doc starts from low resolution and selectively zooms into high-resolution regions for finer evidence, without relying on any external retriever. To train such an agent, we construct an active-perception corpus o",
      "authors": "Kaican Li, Weiyan Xie, Lewei Yao, Jiannan Wu, Lanqing Hong, Yongxiang Huang",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18754"
    },
    {
      "id": 18758,
      "url": "https://arxiv.org/abs/2608.10636",
      "title": "DistilVDR: A Compact End-to-End Visual Document Retriever via Dual-Student Distillation",
      "summary": "Visual document retrieval (VDR) is dominated by multi-billion-parameter models that are slow to index at full corpus scale and expensive to serve. Prior compression routes either train a smaller multi-vector encoder from scratch or distil only the query side; neither yields a compact single-vector retriever end-to-end. We present DistilVDR, a 524M end-to-end VDR system distilled bilaterally from a single 8B vision-language teacher under a pointwise cosine alignment loss. All supervision comes fr",
      "authors": "Zhuchenyang Liu, Ziyi Wang, Yao Zhang, Yu Xiao",
      "category": "research",
      "topics": "safety-alignment,children-education",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18758"
    },
    {
      "id": 18759,
      "url": "https://arxiv.org/abs/2608.11205",
      "title": "AdvFD: Boosting Visual Generation via Adversarial Fr'echet Distance Loss",
      "summary": "Fréchet distance has recently emerged as an effective distribution-level objective for generator post-training, complementing the conventional sample-level diffusion and flow-matching losses. However, directly optimizing Fréchet objectives can cause Fréchet hacking. The target metrics keep improving, but visual quality and Fréchet alignment in other feature spaces may stagnate or deteriorate. We attribute this failure to the static pretrained feature spaces used by existing Fréchet losses. These",
      "authors": "Mingju Gao, Jingkai Zhou, Kun Gai, Changqian Yu, Hao Tang",
      "category": "research",
      "topics": "safety-alignment",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18759"
    },
    {
      "id": 19150,
      "url": "https://arxiv.org/abs/2608.11350",
      "title": "Self-Evolving Embodied Agents via Skill-Harness Evolution",
      "summary": "Embodied agents are increasingly built as systems around foundation models, where performance depends not only on model weights but also on the skills, context, action interfaces, and execution harness surrounding the model. While supervised fine-tuning and reinforcement learning can adapt agents to new environments, they require additional data, rewards, and training runs; meanwhile, many train-free code-centric approaches rely on programmable robot APIs that may be unavailable in fixed-interfa",
      "authors": "Peidong Wang, Zhiming Ma, Ying Chang, Xufang Luo, Xiaocui Yang, Shi Feng",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19150"
    },
    {
      "id": 19509,
      "url": "https://arxiv.org/abs/2608.10538",
      "title": "SKILLER: Language-Level Reinforcement Learning for Reusable Skill Extraction in Small Language Models",
      "summary": "Agent skills represent a standardized format for packaging procedural knowledge and domain expertise, serving within agent harness systems as an essential mechanism to continually constrain a language model's behavior space for repeatable, high-quality task execution. However, because strong closed-source models entail high inference costs, current popular agent harnesses, such as Codex and OpenClaw, remain prohibitively expensive when deploying these skills to accomplish real-world tasks. The r",
      "authors": "Chenhao Dang, Siyuan Xiong, Conghui He, Weijia Li",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-10T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19509"
    },
    {
      "id": 17968,
      "url": "https://arxiv.org/abs/2608.09853",
      "title": "RynnValue: Scaling Robotic Value Foundation Models with Temporal Distance",
      "summary": "General-purpose reward models are increasingly the bottleneck for scaling robot learning, yet the recipe for learning value-related capabilities from large-scale heterogeneous corpora remains underexplored. Existing approaches tie supervision to task-internal anchors such as preferences or normalized progress, none of which transfer cleanly across embodiments and data sources. We introduce RynnValue, an open-source value foundation model for robotic manipulation that replaces these anchors with",
      "authors": "Dongchi Huang, Hongyin Zhang, Bohan Hou, Siteng Huang, Zhian Su, Hang Guo",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17968"
    },
    {
      "id": 17970,
      "url": "https://arxiv.org/abs/2608.09096",
      "title": "Evo-Bench: Can Language Models Improve Agent Harness?",
      "summary": "Large Language Models (LLMs) have driven rapid progress in autonomous agents, yet standard evaluations remain confined to static task solving. An emerging frontier is harness evolution---the agent's capacity to autonomously optimize its own operating harness. However, systematically benchmarking this capability remains challenging, as existing evaluations fail to isolate harness improvements from base model strength, prevent task-specific overfitting, or capture long-horizon iterative research.",
      "authors": "Lisheng Huang, Chen Yang, Hao Zhou, Huatong Song, Zongchao Chen, Ran Le",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17970"
    },
    {
      "id": 17971,
      "url": "https://arxiv.org/abs/2608.09420",
      "title": "Intent Speaks Louder: Controllable User Simulation Beyond Response Imitation",
      "summary": "User simulators are widely used as scalable environments for training and evaluating interactive assistants. Generating the next user turn is inherently one-to-many: the same profile and dialogue context may support multiple plausible continuations with different local interaction intents. A fluent response may therefore advance the dialogue through an inappropriate intent, such as acceptance rather than repair. Our key insight is that controllable user simulation should separate which local int",
      "authors": "Bo Wang, Ruixing Zhang, Yunqi Liu, Yang Zhang, Liangzhe Han, Tongyu Zhu",
      "category": "research",
      "topics": "environment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17971"
    },
    {
      "id": 17972,
      "url": "https://arxiv.org/abs/2608.09819",
      "title": "Macaron-V1: Towards Open Continual Learning with Self-Improvement and Mixture-of-LoRA",
      "summary": "Macaron-V1 is an open agent-model family for experiential intelligence: learning from experience in real environments and continuing to learn after deployment. It is organized around two system goals. Adaptation is pursued through recursive improvement of versioned model-harness pairs, where experience from one configuration is evaluated under an external contract and used to construct its successor. Collaboration is pursued via the Mixture-of-LoRA (MoL) architecture that freezes a base model, c",
      "authors": "Mind Lab, Vin Bo, Asher Cai, Jingwei Cao, Song Cao, Vic Cao",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17972"
    },
    {
      "id": 17973,
      "url": "https://arxiv.org/abs/2608.09802",
      "title": "SWE-Bench ProMax: Benchmarking Agents on Large-Scale Multilingual Code Refactoring",
      "summary": "As AI coding agents take on increasingly complex, long-horizon software engineering tasks, existing benchmarks are rapidly saturating and their evaluation quality has come under serious scrutiny: a recent audit found that nearly 60% of unsolved SWE-bench Verified instances contain flawed tests -- either overly narrow tests that reject correct solutions or overly broad tests that check unstated requirements -- and that frontier models can verbatim reproduce gold patches from training data. Code r",
      "authors": "Yuling Shi, Jinghan Xu, Kelin Fu, Wenhao Zeng, Shilin He, Lei Zhang",
      "category": "research",
      "topics": "agents-autonomy,transparency",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17973"
    },
    {
      "id": 17974,
      "url": "https://arxiv.org/abs/2608.09873",
      "title": "Sci-VBench: Evaluating Knowledge- and Reasoning-Intensive Video Generation in Science Domains",
      "summary": "We introduce Sci-VBench, a comprehensive benchmark for evaluating knowledge- and reasoning-intensive video generation across scientific domains. It contains 1,253 expert-annotated examples spanning 60 subjects across four core disciplines: Natural Science, Healthcare, Humanities & Social Sciences, and Engineering. Each example requires models to generate temporally rich videos that demand scientific reasoning and knowledge-grounded synthesis, going beyond surface-level visual plausibility. We fu",
      "authors": "Diandian Zhang, Tingyu Song, Lin Fu, Zheyuan Yang, Yilun Zhao",
      "category": "research",
      "topics": "healthcare",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/17974"
    },
    {
      "id": 18379,
      "url": "https://arxiv.org/abs/2608.10299",
      "title": "Co-Evolution in Agentic Systems: Toward Self-Directed Evolution Beyond Human Design",
      "summary": "Agentic systems are increasingly expected to improve after deployment, yet single-entity self-evolution is often bounded by a static learning context, such as fixed tasks and feedback. This survey focuses on co-evolution in agentic systems, a multi-component form of self-evolution in which multiple agents and their environment impose adaptive pressure on one another. To organize existing papers, we propose a progressive three-stage taxonomy that traces how the system gradually sheds human-engine",
      "authors": "Qing Zong, Jiayu Liu, Junhao Shen, Zecong Tang, Linsi Wu, Yuxuan Liu",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18379"
    },
    {
      "id": 18386,
      "url": "https://arxiv.org/abs/2608.09766",
      "title": "Cultivar: A Contrastive and Locale-Oriented Translation Benchmark for Investigating Contamination and Localisation Robustness",
      "summary": "Multilingual translation benchmarks are typically sourced in English and translated into other languages, treating language pairs as the unit of evaluation---a design that is prone to contamination over time and overlooks locale and cultural considerations. We therefore advocate for source-contrastive evaluation and instantiate it with Cultivar, a localised subset of FLORES, which enables locale-specific translation evaluation. When paired with unlocalised counterparts, performance discrepancy a",
      "authors": "Pinzhen Chen, Koel Dutta Chowdhury, Xiaoya Xu, David Tan, Doreen Osmelak, Ona de Gibert",
      "category": "research",
      "topics": "finance-investment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18386"
    },
    {
      "id": 18392,
      "url": "https://arxiv.org/abs/2608.09848",
      "title": "CEAA: A Cognitive Embodied Agents Architecture for Interactive Computing Systems",
      "summary": "The development of embodied Intelligent Virtual Agents (IVAs) that have cognitive capabilities in real-time interactive virtual environments remains a challenge, even with today's advancements in technology. Existing architectures are often focused on either the implementation of low-level reactive control systems that are constrained by commercial game engines, or high-level representations of reasoning models that can be difficult to implement in virtual worlds. This paper builds on that notio",
      "authors": "Aimilios Hadjiliasi, Louis Nisiotis",
      "category": "research",
      "topics": "agents-autonomy,environment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18392"
    },
    {
      "id": 18393,
      "url": "https://arxiv.org/abs/2608.09867",
      "title": "Stealing Reasoning Traces from Proprietary LLM APIs",
      "summary": "Leading large language model providers now conceal their models' step-by-step reasoning, or chain-of-thought, to protect intellectual property and limit information leakage. Rather than storing these traces server-side, providers return them to the client as blocks of encrypted text, which the client passes back with each subsequent request. Building on prior research, we identify an architectural vulnerability: these encrypted blocks are fully compatible and interchangeable across different ses",
      "authors": "Alexander Panfilov, David Schmotz, Ilia Shumailov, Luca Beurer-Kellner, Joachim Schaeffer, Ameya Prabhu",
      "category": "research",
      "topics": "copyright-ip",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18393"
    },
    {
      "id": 18394,
      "url": "https://arxiv.org/abs/2608.07346",
      "title": "A^2E : An End-to-End Agent Auditing Engine",
      "summary": "With the rapid advancement of large language models (LLMs), harnesses have become essential infrastructure for deploying agents across a wide range of domains. The fast-evolving harness ecosystem has also made rigorous capability evaluation increasingly important. However, efficiently building an end-to-end, systematic, and comprehensive evaluation pipeline remains a significant challenge. To address this challenge, we introduce A^2E (Agent Auditing Engine), an end-to-end evaluation engine desig",
      "authors": "Haoning Wang, Mingxun Zhang, Chenyue Yu, Yingjun Shang, Xia Hu, Guanchu Wang",
      "category": "research",
      "topics": "agents-autonomy,transparency",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18394"
    },
    {
      "id": 18755,
      "url": "https://arxiv.org/abs/2608.10288",
      "title": "Power law graph attention: exact generalization of scaled dot-product attention, empirical collapse at inference",
      "summary": "The Large Language Model from Power Law Decoder Representations (PLDR-LLM) and its attention, Power Law Graph Attention (PLGA), replace the fixed bilinear form of scaled dot-product attention (SDPA) with a learned, input-generated bilinear operator G_{LM}, built from a positive tensor A_{LM} by elementwise power laws. The architecture is fully specified, verified against pinned reference releases; claims are labeled theorem, conditional theorem, measurement, or conjecture. Unconditionally: PLGA",
      "authors": "Burc Gokden",
      "category": "research",
      "topics": "regulation",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18755"
    },
    {
      "id": 18757,
      "url": "https://arxiv.org/abs/2608.09900",
      "title": "Decoding-Level Taboo: A Diagnostic Stress Test for LLM Robustness",
      "summary": "Large language model evaluations typically focus on performance under nominal conditions, creating an illusion of capability where models comfortably walk a narrow, highly optimized generation corridor. In real-world deployments, however, complex system prompts, safety guardrails, and structural constraints continuously force models off this nominal path, driving a divergence between benchmark scores and deployment performance. To address this issue, we introduce Decoding-Level Taboo, a zero-pro",
      "authors": "Tadanobu Chuyo Kamijo, Ori Rottenstreich, Javier Conde, Gonzalo Martínez, Pedro Reviriego",
      "category": "research",
      "topics": "healthcare",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18757"
    },
    {
      "id": 19145,
      "url": "https://arxiv.org/abs/2608.08975",
      "title": "How Can Rhetoric Reward-Hack AI Reviewers? Dissecting Rhetorical Sensitivity in AI-Based Peer Review",
      "summary": "As large language models increasingly participate in scientific evaluation, we investigate a potential form of reward hacking: how rhetorical choices shape AI-review judgments when reported scientific content is preserved and how these effects vary across evaluation conditions. We construct a controlled corpus of 4,200 full-paper manuscripts derived from 120 anonymized ICLR 2026 submissions. Two LLM rewriters transform six rhetorical dimensions in opposing directions, and five LLM reviewers eval",
      "authors": "Ming Li, Chenguang Wang, Xirui Li, Xinyue Zeng, Dianqi Li, Peng Shi",
      "category": "research",
      "topics": "finance-investment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19145"
    },
    {
      "id": 19502,
      "url": "https://arxiv.org/abs/2608.06914",
      "title": "RibAssist 3D: Biplanar Rib-Fracture Detection, Addressing, and Selective 3D Localization from CT-Derived Projections",
      "summary": "Rib fractures are common and time-consuming to localize on computed tomography (CT). We ask whether fractures detected independently in two orthogonal CT-derived projections (anteroposterior and lateral) can be paired across views and triangulated into reliable 3D points at a controlled rate of false outputs, and we answer it with a staged diagnostic study. The projection geometry is exact, and given correct correspondence, localization is accurate (median 4.0 mm, 88% within 10 mm, 93.6% rib-exa",
      "authors": "Kabila Haile Soboka",
      "category": "research",
      "topics": "healthcare",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19502"
    },
    {
      "id": 19504,
      "url": "https://arxiv.org/abs/2608.09158",
      "title": "From Inaudible Inputs to Model Failures: Low-Frequency Safety Risks in LALMs",
      "summary": "Large audio-language models (LALMs) have demonstrated strong capabilities in understanding diverse audio inputs. This diversity includes low-frequency signals that are inaudible to humans but can still enter the model and influence its generation. However, the practical impact of such low-frequency inputs on LALMs remains largely unexplored. In this paper, we propose Intermittent Low-Frequency Lockout (ILL), an inaudible red teaming method that evaluates this risk using a universal waveform temp",
      "authors": "Yuanhe Zhang, Weiliu Wang, Jie Ren, Liang Lin, Zhenhong Zhou, Haoran Gao",
      "category": "research",
      "topics": "safety-alignment",
      "published_at": "2026-08-09T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/19504"
    },
    {
      "id": 18375,
      "url": "https://arxiv.org/abs/2608.08389",
      "title": "Not Worth Another Token: Marginal Value Estimation for Efficient Deep Research Agents",
      "summary": "Long-horizon research agents solve open-ended tasks through iterative retrieval, aggregation, and synthesis, but context grows rapidly while the marginal value of additional evidence often declines. This leads to unnecessary token cost, higher latency, and noisier inputs for final report generation. We study marginal value estimation for context management in deep research agents and present the first systematic stage-aware comparison of pruning strategies across the pipeline. We evaluate lightw",
      "authors": "Harshitha Kolukuluru, Reshma Ashok, Kirat Arora, Evan William Ciccarelli, Nischal Ashok Kumar, Lunyiu Nie",
      "category": "research",
      "topics": "agents-autonomy",
      "published_at": "2026-08-08T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18375"
    },
    {
      "id": 18383,
      "url": "https://arxiv.org/abs/2608.08477",
      "title": "VectraYX-Vision-1B: A Sub-2B Spanish/LATAM Cybersecurity Vision-Language Model with Structured Visual Reasoning and Native Tool Use",
      "summary": "We present VectraYX-Vision-1B, a sub-2B vision-language model (VLM) for Spanish/LATAM cybersecurity imagery, coupling a frozen SigLIP-so400m encoder to a 1.04B Spanish/LATAM security decoder via an MLP. To our knowledge, it is the first sub-2B VLM specialized for cyber UI (IDA, Ghidra, Wireshark, Nmap, Metasploit, Volatility) that answers in Spanish, emits structured reasoning via native tokens, invokes tools via Model Context Protocol ( ), and exports to llama.cpp's LLaVA mmproj format for air-",
      "authors": "Juan S. Santillana",
      "category": "research",
      "topics": "military-security",
      "published_at": "2026-08-08T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18383"
    },
    {
      "id": 18385,
      "url": "https://arxiv.org/abs/2608.06296",
      "title": "On-Policy Self-Distillation without Any Supervision",
      "summary": "On-policy (Self-)Distillation (OPD / OPSD) has shown strong potential for post-training large language models (LLMs). However, existing methods still rely heavily on external supervision, including ground-truth signals, environmental feedback, or guidance from larger models, and therefore fall short of genuine \"self\"-distillation. In this study, we show that on-policy self-distillation can be achieved using only a model's own generations via internal consistency. We propose unsupervised on-polic",
      "authors": "Yijiang Li, Bingyang Wang, Yijun Liang, Yunjie Tian, Di Fu, Nuno Vasconcelos",
      "category": "research",
      "topics": "regulation,environment",
      "published_at": "2026-08-08T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18385"
    },
    {
      "id": 18387,
      "url": "https://arxiv.org/abs/2608.08621",
      "title": "Business Arena: Benchmarking LLM Agents in a Realistic Marketplace",
      "summary": "Running a business is a challenging form of intelligent work. Operators must infer opportunities from partial signals, commit capital under uncertainty, adapt to delayed outcomes in a changing market, and satisfy regulatory obligations before trading legally. Frontier LLM agents can increasingly complete complex workflows, yet business-related capabilities are rarely evaluated in existing agent benchmarks. We introduce Business Arena, a controlled environment where an AI agent runs a cross-borde",
      "authors": "Yijun Pan, Yukun Lian, Kunyu Shi, Junbo Li, Hongwei Xue, Sicong Xie",
      "category": "research",
      "topics": "regulation,agents-autonomy,environment",
      "published_at": "2026-08-08T20:00:00.000Z",
      "source": "HuggingFace Daily Papers",
      "ethics_ai_record_url": "https://ethics.ai/record/18387"
    }
  ],
  "attribution": "via ethics.ai"
}