{
  "source": "arxiv",
  "fetched_at": "2026-09-25T18:12:24.955Z",
  "count": 50,
  "items": [
    {
      "id": "arxiv_2609.30266v1",
      "title": "LLM Agents Can Easily Tamper With Their Own Traces",
      "url": "http://arxiv.org/abs/2609.30266v1",
      "summary": "Asynchronous monitoring, incident investigations, and compliance audits primarily rely on agent traces to reconstruct what happened. These analyses assume that LLM agents cannot tamper with their own ",
      "authors": [
        "Jeremy Qin",
        "David Schmotz",
        "Derck Prinzhorn",
        "Luca Beurer-Kellner"
      ],
      "published": "2026-09-24T17:59:54Z"
    },
    {
      "id": "arxiv_2609.30264v1",
      "title": "AD-WM: Action-Discriminative World Models for Counterfactual Model Predictive Control",
      "url": "http://arxiv.org/abs/2609.30264v1",
      "summary": "Latent world models are typically trained to predict factual transitions, whereas model predictive control (MPC) must compare alternative actions from the same state. A model can therefore achieve low",
      "authors": [
        "Jiabin Qiu",
        "Zixuan Chen",
        "Hongye Cao",
        "Jieqi Shi"
      ],
      "published": "2026-09-24T17:59:41Z"
    },
    {
      "id": "arxiv_2609.30249v1",
      "title": "RAPID: Robot Agentic Programming from Demonstrations",
      "url": "http://arxiv.org/abs/2609.30249v1",
      "summary": "Coding agents have demonstrated enormous success in solving complex programming problems. To leverage their potential for robot systems, this work introduces Robot Agentic Programming from Demonstrati",
      "authors": [
        "Yuyao Liu",
        "Jiayuan Mao",
        "David Hsu",
        "Leslie Pack Kaelbling"
      ],
      "published": "2026-09-24T17:58:21Z"
    },
    {
      "id": "arxiv_2609.30247v1",
      "title": "Rolling-WAM: World Action Models with Rolling Imagination",
      "url": "http://arxiv.org/abs/2609.30247v1",
      "summary": "World Action Models (WAMs) couple action generation with future visual prediction for robotic manipulation. However, completing the joint video-action denoising process at each replanning cycle incurs",
      "authors": [
        "Yinghua Zhou",
        "Junjie Ye",
        "Yiqi Zhao",
        "Hao Dong"
      ],
      "published": "2026-09-24T17:58:03Z"
    },
    {
      "id": "arxiv_2609.30233v1",
      "title": "Coding Agents for Generalized Task and Motion Planning Problems",
      "url": "http://arxiv.org/abs/2609.30233v1",
      "summary": "Task and motion planning (TAMP) problems remain difficult even with full observability and object-centric states because discrete decisions are tightly coupled to geometric, kinematic, and dynamic con",
      "authors": [
        "Matteo Merler",
        "Bowen Li",
        "Josh Roy",
        "Yichao Liang"
      ],
      "published": "2026-09-24T17:53:35Z"
    },
    {
      "id": "arxiv_2609.30227v1",
      "title": "To Trust or Not to Trust: Retrieval-Augmented Fact Checking in Speech",
      "url": "http://arxiv.org/abs/2609.30227v1",
      "summary": "Online misinformation increasingly appears in spoken formats such as news clips, podcasts, interviews, political speeches, and social media videos, creating a need for fact-checking systems that can v",
      "authors": [
        "Debajyoti Mazumder",
        " Mamta",
        "Abhirama Subramanyam Penamakuri"
      ],
      "published": "2026-09-24T17:50:40Z"
    },
    {
      "id": "arxiv_2609.30226v1",
      "title": "PoEM: Predicting RL Outcomes from Existing Policies",
      "url": "http://arxiv.org/abs/2609.30226v1",
      "summary": "Foundation models are post-trained with reinforcement learning (RL) to maximize specific rewards, such as human alignment, correctness, or instruction following. This post-training process is computat",
      "authors": [
        "Kimia Hamidieh",
        "Giannis Daras",
        "Antonio Torralba"
      ],
      "published": "2026-09-24T17:50:25Z"
    },
    {
      "id": "arxiv_2609.30222v1",
      "title": "TrackEverything: Long Horizon Dense Tracking via De-Duplicating 3D Scene Representations",
      "url": "http://arxiv.org/abs/2609.30222v1",
      "summary": "Existing point tracking models face a fundamental tradeoff: they can either track a sparse set of query points over long horizons, or track all points across only short clips. We introduce TrackEveryt",
      "authors": [
        "Ayush Jain",
        "Sreeharsha Paruchuri",
        "Ishita Gupta",
        "Fan Zhang"
      ],
      "published": "2026-09-24T17:48:20Z"
    },
    {
      "id": "arxiv_2609.30219v1",
      "title": "Requirement-Bound Verified Commissioning: A Frozen Four-Billion-Parameter Local Model as a Candidate Generator under an External Acceptance Layer with Verification and Release Authority",
      "url": "http://arxiv.org/abs/2609.30219v1",
      "summary": "An acceptance protocol is developed for sensor-coordinate and polarity binding in mechatronic commissioning. Candidate generation is separated from release authority. Requirements unsupported by a det",
      "authors": [
        "Mehmet Iscan"
      ],
      "published": "2026-09-24T17:46:53Z"
    },
    {
      "id": "arxiv_2609.30218v1",
      "title": "Minimally Invasive Steering of Language Models",
      "url": "http://arxiv.org/abs/2609.30218v1",
      "summary": "Pre-logit steering adapts a frozen language model to a test-time reward by adding vectors to its final hidden states. Unregularized reward optimization can substantially alter the output distribution ",
      "authors": [
        "Taha Entesari",
        "Jingyu Zhang",
        "Daniel Khashabi",
        "Mahyar Fazlyab"
      ],
      "published": "2026-09-24T17:46:46Z"
    },
    {
      "id": "arxiv_2609.30217v1",
      "title": "Instrumental Monitor Evasion Emerges Under Ordinary Task Pressure",
      "url": "http://arxiv.org/abs/2609.30217v1",
      "summary": "A central concern in AI safety is that agents may treat oversight as an obstacle when it conflicts with completing their goals. We study instrumental evasion, the propensity of LLM agents to circumven",
      "authors": [
        "David Schmotz",
        "Derck Prinzhorn",
        "Luca Beurer-Kellner",
        "Anselm Paulus"
      ],
      "published": "2026-09-24T17:46:27Z"
    },
    {
      "id": "arxiv_2609.30214v1",
      "title": "Underwater C3-JEPA: An Object-Centric Cross-View World Model for ROV Salvage",
      "url": "http://arxiv.org/abs/2609.30214v1",
      "summary": "We present Underwater C$^{3}$-JEPA (cross-view, control-conditioned, context-extended), an object-centric multi-view predictive world model for near-field heavy-load underwater ROV salvage. Without co",
      "authors": [
        "Yuncong Yang",
        "Jinlong Li",
        "Yulong Xue",
        "Feng Wu"
      ],
      "published": "2026-09-24T17:45:42Z"
    },
    {
      "id": "arxiv_2609.30205v1",
      "title": "A Living Benchmark for Information Retrieval from Electronic Health Records",
      "url": "http://arxiv.org/abs/2609.30205v1",
      "summary": "Large language model (LLM)-based clinical assistants are increasingly being integrated into electronic health record (EHR) systems, transforming how clinicians retrieve and synthesize information from",
      "authors": [
        "Jordan L. Cahoon",
        "Chloe O. Stanwyck",
        "Sulaiman Somani",
        "Philip Chung"
      ],
      "published": "2026-09-24T17:41:16Z"
    },
    {
      "id": "arxiv_2609.30199v1",
      "title": "ExplorationBench: Measuring AI Systems' Exploration in Verifiable Alien Worlds",
      "url": "http://arxiv.org/abs/2609.30199v1",
      "summary": "Scientific discovery begins where known problems end. There, AI systems must engage in exploration: framing hypotheses, designing experiments, and iterating on the results. However, evaluating this ab",
      "authors": [
        "Ming Zhang",
        "Zhenghao Xiang",
        "Peizhong Gao",
        "Yujiong Shen"
      ],
      "published": "2026-09-24T17:37:14Z"
    },
    {
      "id": "arxiv_2609.30192v1",
      "title": "SAGE: Mitigating Long-Horizon Reasoning Biases via Topological Guidance",
      "url": "http://arxiv.org/abs/2609.30192v1",
      "summary": "Long-horizon reasoning remains a central challenge for large language models (LLMs) under sparse-reward regimes. We argue that this brittleness arises from two biases induced by complex reasoning spac",
      "authors": [
        "Xinyue Zeng",
        "Jiawei Zhang",
        "Yujun Yan",
        "Dawei Zhou"
      ],
      "published": "2026-09-24T17:33:29Z"
    },
    {
      "id": "arxiv_2609.30186v1",
      "title": "Jev-Mobile: Jev as an Executor for Mobile GUI Agents",
      "url": "http://arxiv.org/abs/2609.30186v1",
      "summary": "Vision-language models (VLMs) have become a common foundation for autonomous mobile GUI agents, but most existing systems rely on the VLM for both planning and action grounding at nearly every interac",
      "authors": [
        "Linghua Zhang"
      ],
      "published": "2026-09-24T17:30:32Z"
    },
    {
      "id": "arxiv_2609.30177v1",
      "title": "Search-Aware Reinforcement Learning for Multi-Component Query Understanding in Roblox Game Search",
      "url": "http://arxiv.org/abs/2609.30177v1",
      "summary": "Query understanding (QU) plays a critical role in production search systems, translating raw user queries into search execution plans that drive downstream retrieval and ranking. While large language ",
      "authors": [
        "Nayoung Choi",
        "Shengjian Chen",
        "Xiaokai Wei",
        "Wenzheng Zhang"
      ],
      "published": "2026-09-24T17:26:46Z"
    },
    {
      "id": "arxiv_2609.30151v1",
      "title": "Does a model's stated reason for rejecting a candidate do any work?",
      "url": "http://arxiv.org/abs/2609.30151v1",
      "summary": "Asked to choose between candidates and explain the choice, a language model often rejects a rival by naming a fact its profile lacks: no director, no date of death. That sentence is a claim about the ",
      "authors": [
        "Archit Rastogi"
      ],
      "published": "2026-09-24T17:13:35Z"
    },
    {
      "id": "arxiv_2609.30147v1",
      "title": "GRASP: Generating, Revising, and Assessing for Strategic Planning with Agentic AI",
      "url": "http://arxiv.org/abs/2609.30147v1",
      "summary": "Large Language Models (LLMs) typically exhibit a performance profile where reliability degrades as task complexity increases. We address the challenge of generating high-quality natural language execu",
      "authors": [
        "Arunabh Srivastava",
        "Mohammad A.",
        " Khojastepour",
        "Srimat Chakradhar"
      ],
      "published": "2026-09-24T17:11:35Z"
    },
    {
      "id": "arxiv_2609.30144v1",
      "title": "EnigmaForge: The Question Is Hidden in the Story",
      "url": "http://arxiv.org/abs/2609.30144v1",
      "summary": "Most benchmarks hand the model a question. EnigmaForge hands it a stack of old documents and no question at all. Buried in the letters, receipts, and logbook margins is a small logic puzzle whose solu",
      "authors": [
        "Daniel Eisner"
      ],
      "published": "2026-09-24T17:10:03Z"
    },
    {
      "id": "arxiv_2609.30137v1",
      "title": "Screen Before You Serve: Simulation for Production Customer Experience AI Agents at 140M Scale",
      "url": "http://arxiv.org/abs/2609.30137v1",
      "summary": "Customer experience (CX) agents use tools and large language models to address customer requests and guide conversational interactions with an organization's products. Improving these agents, especial",
      "authors": [
        "Edesio Alcoba",
        "Kevin Rossell",
        "Aman Gupta",
        "Shao Tang"
      ],
      "published": "2026-09-24T17:07:38Z"
    },
    {
      "id": "arxiv_2609.30123v1",
      "title": "HEXIS: Compiling Skills into Extended Finite State Machines",
      "url": "http://arxiv.org/abs/2609.30123v1",
      "summary": "Agent skills provide reusable knowledge and instructions, yet agents must repeatedly infer how to apply them and which operation should follow. This couples task reasoning with control decisions, allo",
      "authors": [
        "Minghao LI"
      ],
      "published": "2026-09-24T16:58:18Z"
    },
    {
      "id": "arxiv_2609.30100v1",
      "title": "R-DEIM Net: An Efficient Rationale-Augmented Dual-Expert Interaction Model for Paraphrase Detection",
      "url": "http://arxiv.org/abs/2609.30100v1",
      "summary": "Recent advances in paraphrase detection reveal a fundamental trade-off: large language models achieve high accuracy but require high computation, while efficient Siamese-BERT variants offer practical ",
      "authors": [
        " Pushp",
        "Vaibhav Prajapati",
        "Himangshu Sarma"
      ],
      "published": "2026-09-24T16:44:26Z"
    },
    {
      "id": "arxiv_2609.30096v1",
      "title": "Accelerating Video Diffusion via Training-Free Trajectory Routing",
      "url": "http://arxiv.org/abs/2609.30096v1",
      "summary": "Video diffusion is computationally expensive, as it requires executing a large model across many denoising steps. Even with step-distillation, inference remains expensive because every distilled step ",
      "authors": [
        "Mustafa Munir",
        "Huy Vu",
        "Shreyas Misra",
        "Rohit Jena"
      ],
      "published": "2026-09-24T16:39:47Z"
    },
    {
      "id": "arxiv_2609.30094v1",
      "title": "PrivDrift: Auditing User-Secret Leakage Under Topic Drift in Active LLM Conversations",
      "url": "http://arxiv.org/abs/2609.30094v1",
      "summary": "Large language models increasingly operate as persistent assistants in user-facing, shared-session, and tool-augmented settings. When users disclose sensitive information during an active conversation",
      "authors": [
        "Luciano Maldonado"
      ],
      "published": "2026-09-24T16:39:18Z"
    },
    {
      "id": "arxiv_2609.30088v1",
      "title": "AT-SKM-Net: An Accelerated Trainable Sampling Kaczmarz-Motzkin Framework for Linear Hard-Constraint Feasibility on Dynamic Graphs",
      "url": "http://arxiv.org/abs/2609.30088v1",
      "summary": "Graph-structured optimization with linear constraints is fundamental to critical infrastructure but faces scalability limits due to massive strict hard constraints and high dimensionality. While recen",
      "authors": [
        "Xiaochen Zhang",
        "Haoyu Zhu",
        "Yao Zhang",
        "Qingchun Hou"
      ],
      "published": "2026-09-24T16:36:49Z"
    },
    {
      "id": "arxiv_2609.30079v1",
      "title": "Reachability-Based Formal Verification of Graph Neural Networks with Node and Edge Features",
      "url": "http://arxiv.org/abs/2609.30079v1",
      "summary": "Graph neural networks (GNNs) have become a prominent approach for developing fast, topology-aware surrogates in electric power systems, supporting tasks such as power flow (PF) analysis, optimal power",
      "authors": [
        "Anne M. Tumlin",
        "Ben Wooding",
        "Zhenxuan Shao",
        "Diego Manzanas Lopez"
      ],
      "published": "2026-09-24T16:29:32Z"
    },
    {
      "id": "arxiv_2609.30074v1",
      "title": "How Reproducible Are Evaluation Conclusions? A Self-Audit of LLM-Inferred Prompt Structure",
      "url": "http://arxiv.org/abs/2609.30074v1",
      "summary": "Evaluations of LLM systems routinely average over small prompt sets and report models as a ranked table. We ask how much confidence such a table deserves, using LLM-based prompt-structure inference as",
      "authors": [
        "Dipankar Sarkar"
      ],
      "published": "2026-09-24T16:28:15Z"
    },
    {
      "id": "arxiv_2609.30063v1",
      "title": "Self-Play Pretraining with Zero Data",
      "url": "http://arxiv.org/abs/2609.30063v1",
      "summary": "Advances in language modeling have been driven by scaling pretraining on ever more data. Yet, the training data is still largely curated on the model's behalf. A more general approach to pretraining w",
      "authors": [
        "Aditya Cowsik",
        "Kfir Dolev",
        "Michael Y. Li",
        "G. Bruno De Luca"
      ],
      "published": "2026-09-24T16:23:01Z"
    },
    {
      "id": "arxiv_2609.30059v1",
      "title": "KernelOPT: Dispatch-Aware Agentic Search for GPU Kernel Optimization",
      "url": "http://arxiv.org/abs/2609.30059v1",
      "summary": "Deep learning inference and training performance depends critically on GPU kernel efficiency. Modern compilers such as PyTorch Inductor automatically generate GPU kernels from high-level model code, b",
      "authors": [
        "Aheli Poddar",
        "Sanskar Prasad",
        "Arindam Samanta",
        "Subha Chakraborty"
      ],
      "published": "2026-09-24T16:17:52Z"
    },
    {
      "id": "arxiv_2609.30058v1",
      "title": "Can Labor Markets Function in the Age of AI? The Evaluation Bottleneck in Hiring",
      "url": "http://arxiv.org/abs/2609.30058v1",
      "summary": "AI-assisted job-search tools have become increasingly popular by making it easier to find and apply to jobs. But by making it easier for applicants to generate and tailor application materials, they c",
      "authors": [
        "Itai Ashlagi",
        "Ramesh Johari",
        "Jon Kleinberg",
        "Anushka Murthy"
      ],
      "published": "2026-09-24T16:16:05Z"
    },
    {
      "id": "arxiv_2609.30055v1",
      "title": "Era by Eon: Benchmarking Enterprise Agents on Hidden Knowledge",
      "url": "http://arxiv.org/abs/2609.30055v1",
      "summary": "In the Era by Eon benchmark, each question states the rules for its answer, and code computes the answer from a generated company's data. When agents can run code, the four strongest models each answe",
      "authors": [
        "Benjamin Gruenbaum",
        "Doron Porat",
        "Assaf Natanzon",
        "Roy Zavida"
      ],
      "published": "2026-09-24T16:12:27Z"
    },
    {
      "id": "arxiv_2609.30054v1",
      "title": "SciWalker: Synthesizing Scientific Coding Problems with Operator Graphs and Execution Feedback",
      "url": "http://arxiv.org/abs/2609.30054v1",
      "summary": "Improving the scientific coding capabilities of large language models (LLMs) requires high-quality training data. However, such data remain scarce because manually authoring realistic problems is cost",
      "authors": [
        "Chenxi Li",
        "Wenxuan Zeng",
        "Yun Luo",
        "Fangchen Yu"
      ],
      "published": "2026-09-24T16:12:14Z"
    },
    {
      "id": "arxiv_2609.30050v1",
      "title": "NNV3: Expanding Neural Network Verification to New Architectures and Domains",
      "url": "http://arxiv.org/abs/2609.30050v1",
      "summary": "We present NNV3, the latest version of the Neural Network Verification (NNV) tool, a MATLAB framework for formal verification of deep learning models and learning-enabled cyber-physical systems. Build",
      "authors": [
        "Anne M. Tumlin",
        "Samuel Sasaki",
        "Ben Wooding",
        "Diego Manzanas Lopez"
      ],
      "published": "2026-09-24T16:11:25Z"
    },
    {
      "id": "arxiv_2609.30048v1",
      "title": "Style, Not Self: Surface Cues Explain Zero-Shot Code Attribution by Large Language Models",
      "url": "http://arxiv.org/abs/2609.30048v1",
      "summary": "If a language model can recognize code it wrote, it may favor that code as a judge, and instances of one model monitoring each other could collude. We test this zero-shot on current commercial models.",
      "authors": [
        "Ehsan Barkhordar",
        "Surendrabikram Thapa"
      ],
      "published": "2026-09-24T16:11:21Z"
    },
    {
      "id": "arxiv_2609.30028v1",
      "title": "How does Adversarial Influence Scale in Multi-Agent Systems?",
      "url": "http://arxiv.org/abs/2609.30028v1",
      "summary": "Multi-agent deliberation can improve performance, but what happens when some agents do not act in good faith? In practice, an agent may be deceptive and work to subvert the group, whether through its ",
      "authors": [
        "Addison J. Wu",
        "Jasin Cekinmez",
        "Michel Liao",
        "Karthik Narasimhan"
      ],
      "published": "2026-09-24T16:02:40Z"
    },
    {
      "id": "arxiv_2609.30027v1",
      "title": "Synthetic Hospital: An Open, Verifiable, Physician-Validated Longitudinal EHR Benchmark",
      "url": "http://arxiv.org/abs/2609.30027v1",
      "summary": "Frontier language models are rarely used in clinical workflows because the realistic, longitudinal benchmarks needed to develop them are scarce. Real electronic health record (EHR) data cannot be open",
      "authors": [
        "Christine Park",
        "Valerie Chen",
        "Tim Dettmers"
      ],
      "published": "2026-09-24T16:00:56Z"
    },
    {
      "id": "arxiv_2609.30017v1",
      "title": "Canopy: Exploiting Piecewise Smooth Tree Priors for Multi-Fidelity Bandits",
      "url": "http://arxiv.org/abs/2609.30017v1",
      "summary": "Many LLM inference problems, including model routing, prefix-cache management, prompt trimming, and test-time search, can be viewed as optimization over a tree. This structure arises naturally from au",
      "authors": [
        "Michael Jerge",
        "Suman Jana"
      ],
      "published": "2026-09-24T15:55:24Z"
    },
    {
      "id": "arxiv_2609.30012v1",
      "title": "Low-Cost Assays for Measuring Model Behavior Across Vendors and Releases",
      "url": "http://arxiv.org/abs/2609.30012v1",
      "summary": "Language models advise people, keep them company, and write software while they sleep. Measuring what they do is hard: behavior has to be sampled repeatedly across models, prompts and releases, most o",
      "authors": [
        "Tapan Parikh"
      ],
      "published": "2026-09-24T15:51:17Z"
    },
    {
      "id": "arxiv_2609.30009v1",
      "title": "Automated Regulatory Compliance Question Answering in Financial Services with Domain-Adapted Retrieval-Augmented Generation",
      "url": "http://arxiv.org/abs/2609.30009v1",
      "summary": "Financial institutions operate under dense, frequently amended rulebooks, and answering a compliance question correctly requires not only fluency but verifiable grounding in the authoritative text. La",
      "authors": [
        "Tobias Deußer",
        "Abhishek Pillai",
        "Aurelio F. Bariviera",
        "Dhananjay Bhardwaj"
      ],
      "published": "2026-09-24T15:48:59Z"
    },
    {
      "id": "arxiv_2609.30001v1",
      "title": "Advancing Model Research in AgentX: Long-Horizon Autonomy for Industrial Recommender Systems",
      "url": "http://arxiv.org/abs/2609.30001v1",
      "summary": "Sustaining industrial recommendation research requires using the results of one experiment to decide what to investigate next. We present AgentX-Model, the next generation of AgentX's model research f",
      "authors": [
        "Shuang Yang",
        "Zijie Zhuang",
        "Changxin Lao",
        "Pengbo Xu"
      ],
      "published": "2026-09-24T15:45:07Z"
    },
    {
      "id": "arxiv_2609.29999v1",
      "title": "GHOST-Q: Towards Studying Grounding Hallucinations Overlooked Under Same-score TradeOffs in Quantized VLMS",
      "url": "http://arxiv.org/abs/2609.29999v1",
      "summary": "Post-training quantization of vision--language models (VLMs) is typically assessed through aggregate task accuracy and memory savings, but preserving a headline score does not guarantee preservation o",
      "authors": [
        "Saim Rehman",
        "Muhammad Shafique"
      ],
      "published": "2026-09-24T15:44:33Z"
    },
    {
      "id": "arxiv_2609.29995v1",
      "title": "Guardrails or Roadblocks? Effects of Pedagogical Style and Context Awareness in AI Teaching Assistants for Programming",
      "url": "http://arxiv.org/abs/2609.29995v1",
      "summary": "AI teaching assistants (AI TAs) backed by large language models (LLMs) and pedagogical guardrails are increasingly being integrated into programming courses, providing students with scalable access to",
      "authors": [
        "Madeleine Eastwood",
        "Harshith Narne",
        "Joseph Hilby",
        "Paul Denny"
      ],
      "published": "2026-09-24T15:43:49Z"
    },
    {
      "id": "arxiv_2609.29983v1",
      "title": "From Interests to Semantic IDs: Retrieval-Grounded Credit Assignment for Generative Recommendation",
      "url": "http://arxiv.org/abs/2609.29983v1",
      "summary": "Semantic IDs (SIDs) encode each catalog item as a short token sequence, enabling generative recommenders to predict the next item autoregressively. Reasoning-enhanced variants, an increasingly common ",
      "authors": [
        "Mengdan Zhu",
        "Yufan Zhao",
        "Yao Zhao",
        "Sophie Di"
      ],
      "published": "2026-09-24T15:34:11Z"
    },
    {
      "id": "arxiv_2609.29973v1",
      "title": "Learning Better Reasoning for Generative Recommendation with Semantic IDs",
      "url": "http://arxiv.org/abs/2609.29973v1",
      "summary": "Generative recommendation reformulates item retrieval as sequence generation, allowing a unified model to directly generate the next item from a user's interaction history. Semantic IDs further make t",
      "authors": [
        "Mengdan Zhu",
        "Yufan Zhao",
        "Sophie Di",
        "Yao Zhao"
      ],
      "published": "2026-09-24T15:24:54Z"
    },
    {
      "id": "arxiv_2609.29964v1",
      "title": "World Action Agent: Harnessing VLMs for Robot Manipulation via World Action Rehearsal",
      "url": "http://arxiv.org/abs/2609.29964v1",
      "summary": "General-purpose vision-language models (VLMs) bring broad knowledge and spatial reasoning to robot manipulation, yet existing systems either use them indirectly, to predict constraints or write progra",
      "authors": [
        "Yehang Zhang",
        "Haojian Huang",
        "Yifan Chang",
        "Jianchong Su"
      ],
      "published": "2026-09-24T15:19:39Z"
    },
    {
      "id": "arxiv_2609.29963v1",
      "title": "ADATEX4D: adaptive texture capacity allocation for 4D gaussian splatting",
      "url": "http://arxiv.org/abs/2609.29963v1",
      "summary": "Textured Gaussians improve local appearance capacity, but assigning the same texture resolution to every primitive wastes storage on low-detail or weakly visible regions. We introduce AdaTex4D, an ada",
      "authors": [
        "De Jiang",
        "Peiqiang Wang",
        "Kehong Yuan",
        "Shaohua Ma"
      ],
      "published": "2026-09-24T15:19:23Z"
    },
    {
      "id": "arxiv_2609.29960v1",
      "title": "Beyond Average Safety: Chance-Constrained LLM Fine-tuning",
      "url": "http://arxiv.org/abs/2609.29960v1",
      "summary": "Fine-tuning large language models on new objectives can improve helpfulness, instruction following, or domain-specific performance, but it can also induce regressions on safety-critical prompts. Exist",
      "authors": [
        "Taha Entesari",
        "Mahyar Fazlyab"
      ],
      "published": "2026-09-24T15:17:58Z"
    },
    {
      "id": "arxiv_2609.29952v1",
      "title": "Augur: A Synthetic Decision Lab for Rehearsing Reactions to Product and Policy Changes",
      "url": "http://arxiv.org/abs/2609.29952v1",
      "summary": "Before a product or policy change ships, the question that matters is how people will react to it. Augur rehearses that reaction offline: it builds a typed knowledge graph from the change documents, p",
      "authors": [
        "Rahul Khedar",
        "Mayank Malhotra",
        "Avinash Karn"
      ],
      "published": "2026-09-24T15:10:29Z"
    },
    {
      "id": "arxiv_2609.29951v1",
      "title": "Tracking States or Tracking Cosets? An Algebraic Account of Learned State Tracking",
      "url": "http://arxiv.org/abs/2609.29951v1",
      "summary": "State tracking requires composing a sequence of updates, but accuracy alone does not reveal what a model has learned. We study neural networks trained to predict the running product of group elements.",
      "authors": [
        "Zhiyu Zhang",
        "Yupeng Li"
      ],
      "published": "2026-09-24T15:10:02Z"
    }
  ]
}