{
  "version": "https://jsonfeed.org/version/1.1",
  "title": "Hugging Face Daily Papers",
  "home_page_url": "https://huggingface.co/papers",
  "description": "Events collected by UnlimitedPipe 0.3.2",
  "_unlimitedpipe": {
    "schema": "unlimitedpipe.event/1",
    "generator": "UnlimitedPipe 0.3.2"
  },
  "items": [
    {
      "id": "9d166491f174f2b92a63",
      "title": "Learning to Discover Interesting Mathematics",
      "content_text": "Recently, Large Language Models (LLMs) have been increasingly able to solve advanced mathematical problems, including many that have been open for decades. This opens the door to expansion of mathematical knowledge at unprecedented scale. Yet, while LLMs may be able to conjecture and prove more and more theorems, it remains open whether this new mathematical knowledge is interesting or useful. We define intrinsic interestingness of a theorem as the ratio between the length of its proof and the…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "9d166491f174f2b92a63",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28603",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T23:08:56Z",
          "data": {
            "change": "added",
            "label": "Learning to Discover Interesting Mathematics",
            "item_type": "record",
            "summary": "added: Learning to Discover Interesting Mathematics",
            "after": {
              "title": "Learning to Discover Interesting Mathematics",
              "summary": "Recently, Large Language Models (LLMs) have been increasingly able to solve advanced mathematical problems, including many that have been open for decades. This opens the door to expansion of mathematical knowledge at unprecedented scale. Yet, while LLMs may be able to conjecture and prove more and more theorems, it remains open whether this new mathematical knowledge is interesting or useful. We define intrinsic interestingness of a theorem as the ratio between the length of its proof and the length of its statement. We show that this correlates strongly with an extrinsic measure of the downstream utility of a theorem. We identify the difficulty of a proof conditioned on a set of premises as a useful primitive for computing these metrics, and train a 27B model that predicts proof difficulty more accurately than frontier general-purpose models. Optimizing for our metric creates a model capable of producing more interesting theorems, while also reducing substantial or full overlap with Mathlib from 91.9% to 30.6%, showcasing the creation of more out-of-distribution math. We show that our system can generate candidate theorems, select the most interesting among them, and iteratively build on a self-expanding mathematical library. These metrics provide a practical and quantifiable signal for ranking conjectures and guiding proof search within formal mathematical libraries. Our framework provides a path towards self-expanding, machine-verified mathematical libraries that can choose worthwhile statements without relying on human-supplied targets.",
              "organization": {
                "_id": "63f68bebb29015adc33fb06b",
                "name": "nyuniversity",
                "fullname": "New York University",
                "avatar": "https://www.gravatar.com/avatar/386164389c9379796b8f1f6620b82878?d=retro&size=100"
              },
              "link": "https://huggingface.co/papers/2609.28603",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 105,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28603"
    },
    {
      "id": "b6ee30b54f202470963e",
      "title": "RGBD20K: A Large-Scale Benchmark for RGB-D Semantic Segmentation",
      "content_text": "In this paper, we propose RGBD20K, a novel dataset for facilitating the development of more robust and general RGB-D semantic segmentation by encompassing abundant categories and high-quality annotations. RGBD20K possesses several attractive properties: (1) Expanded Semantic Space. In particular, it covers 160 fine-grained categories, largely surpassing the category diversity of existing popular RGB-D benchmarks (e.g., NYUv2 with 40 classes and SUN RGB-D with 37 classes). With such enriched…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "b6ee30b54f202470963e",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29028",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T17:36:23Z",
          "data": {
            "change": "added",
            "label": "RGBD20K: A Large-Scale Benchmark for RGB-D Semantic Segmentation",
            "item_type": "record",
            "summary": "added: RGBD20K: A Large-Scale Benchmark for RGB-D Semantic Segmentation",
            "after": {
              "title": "RGBD20K: A Large-Scale Benchmark for RGB-D Semantic Segmentation",
              "summary": "In this paper, we propose RGBD20K, a novel dataset for facilitating the development of more robust and general RGB-D semantic segmentation by encompassing abundant categories and high-quality annotations. RGBD20K possesses several attractive properties: (1) Expanded Semantic Space. In particular, it covers 160 fine-grained categories, largely surpassing the category diversity of existing popular RGB-D benchmarks (e.g., NYUv2 with 40 classes and SUN RGB-D with 37 classes). With such enriched semantic coverage, we expect to promote the learning of more generalizable segmentation models. (2) Larger Scale. Compared with current benchmarks, RGBD20K offers 20,000 RGB-D image pairs, providing a substantially larger training resource that benefits the development of more powerful deep models. (3) High-Fidelity Annotation. We perform rigorous re-evaluation and correction of existing labels to resolve long-standing annotation noise, resulting in a clean and reliable ground-truth foundation. Furthermore, we propose a novel score-purified fusion (SPF) method, which achieves state-of-the-art performance across all evaluated benchmarks, demonstrating the effectiveness of our approach in leveraging high-quality multimodal information for RGB-D semantic segmentation. The dataset is here: https://github.com/ShaohuaDong2021/RGBD20K/.",
              "organization": {
                "_id": "63ac8af59fc40b145608940c",
                "name": "UNT",
                "fullname": "University of North Texas",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/noauth/GgPpQytlMV5cIUIHSldbL.png"
              },
              "link": "https://huggingface.co/papers/2609.29028",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 5
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 47,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29028"
    },
    {
      "id": "032a5a544348d58ce99c",
      "title": "AV-GRPO: Modality-Anchored Decoupling Diffusion Reinforcement Learning for Joint Audio-Video Generation",
      "content_text": "Recent years have witnessed major progress in joint audio-video generation. Existing models still suffer from limited per-modality fidelity, insufficient text-modality alignment and weak cross-modal synchronization. While reinforcement-learning post-training offers a promising remedy, directly adapting it to joint audio-video generation is challenging. Heterogeneous multimodal rewards entangle learning signals and complicate credit assignment. Joint optimization of two modality towers is…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "032a5a544348d58ce99c",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29816",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T17:36:23Z",
          "data": {
            "change": "added",
            "label": "AV-GRPO: Modality-Anchored Decoupling Diffusion Reinforcement Learning for Joint Audio-Video Generation",
            "item_type": "record",
            "summary": "added: AV-GRPO: Modality-Anchored Decoupling Diffusion Reinforcement Learning for Joint Audio-Video Generation",
            "after": {
              "title": "AV-GRPO: Modality-Anchored Decoupling Diffusion Reinforcement Learning for Joint Audio-Video Generation",
              "summary": "Recent years have witnessed major progress in joint audio-video generation. Existing models still suffer from limited per-modality fidelity, insufficient text-modality alignment and weak cross-modal synchronization. While reinforcement-learning post-training offers a promising remedy, directly adapting it to joint audio-video generation is challenging. Heterogeneous multimodal rewards entangle learning signals and complicate credit assignment. Joint optimization of two modality towers is computationally expensive given their divergent dynamics. Moreover, synchronization evaluation difficulty depends on paired samples, preventing fair reward comparisons. We propose AV-GRPO, a modality-anchored online diffusion RL framework, and 5DAV, a decoupled, difficulty-controllable training dataset. AV-GRPO includes three key modules: (1) modality-anchored rollouts to disentangle learning signals and stabilize difficulty; (2) trajectory-locked frozen-tower optimization to reduce cost and reassign credit; (3) adaptive objectives and perturbation strengths tailored to modality-specific dynamics. This converts coupled multimodal preference learning into unimodal subproblems for precise reward attribution and better synchronization. Our 5DAV dataset decouples samples across five dimensions for systematic training. Experiments on JavisBench and VABench demonstrate AV-GRPO outperforms LTX-2.3 in generation quality, semantic alignment and cross-modal synchronization under LoRA and full fine-tuning. Ablations confirm our designs. Code and data: https://github.com/zhiyuxu03/AV-GRPO",
              "organization": {
                "_id": "6a4fb75a1c66dbf208e7ddb6",
                "name": "Shanghai-AI-Laboratory",
                "fullname": "Shanghai AI Laboratory",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/65cd955637be1841d0b75397/Rao_Kq6NMtTVfSqLUIR4k.webp"
              },
              "link": "https://huggingface.co/papers/2609.29816",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 47,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29816"
    },
    {
      "id": "8f2af8f1b01d8527e3be",
      "title": "Your Transformer Can Hold Two Thoughts at Once: Evidence of Linear Superposition in LLMs",
      "content_text": "While Large Language Models (LLMs) rely on highly non-linear components, in this work we demonstrate that they exhibit fundamental linearity: when inputs from distinct text streams are linearly combined, the model outputs a superposition of the individual next-token distributions. We term this the Superposition Linearity Hypothesis. We provide evidence that superposition is an intrinsic property of the Transformer architecture rather than an emergent consequence of training; in fact, we observe…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "8f2af8f1b01d8527e3be",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29845",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T17:36:23Z",
          "data": {
            "change": "added",
            "label": "Your Transformer Can Hold Two Thoughts at Once: Evidence of Linear Superposition in LLMs",
            "item_type": "record",
            "summary": "added: Your Transformer Can Hold Two Thoughts at Once: Evidence of Linear Superposition in LLMs",
            "after": {
              "title": "Your Transformer Can Hold Two Thoughts at Once: Evidence of Linear Superposition in LLMs",
              "summary": "While Large Language Models (LLMs) rely on highly non-linear components, in this work we demonstrate that they exhibit fundamental linearity: when inputs from distinct text streams are linearly combined, the model outputs a superposition of the individual next-token distributions. We term this the Superposition Linearity Hypothesis. We provide evidence that superposition is an intrinsic property of the Transformer architecture rather than an emergent consequence of training; in fact, we observe that it tends to diminish as pretraining progresses. However, we demonstrate that linearity can be substantially restored through lightweight fine-tuning, significantly reducing the divergence between the predicted next-token distribution and the average of the individual next-token distributions. Finally, we introduce a guided decoding procedure that disentangles superposed outputs, enabling the simultaneous generation of two coherent continuations from a single forward pass.",
              "link": "https://huggingface.co/papers/2609.29845",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 46
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 47,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29845"
    },
    {
      "id": "cc48c8076557fd66094b",
      "title": "Just Ask Jev: Reinforcement Learning for Calibrated Decisions as a Zero-Shot Detector of AI Alignment Failures",
      "content_text": "Detectors of alignment failures screen deployed language models and score alignment benchmarks. Most are generative judges that spend a decoding pass on every criterion, and classifiers that read token probabilities, such as Llama Guard, still score one fixed label per call. Jev, a model trained with reinforcement learning for calibrated decisions (RLCD), answers many typed questions about one input with calibrated probabilities in a single call. Whether it detects alignment failures has not…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "cc48c8076557fd66094b",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29429",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T12:22:30Z",
          "data": {
            "change": "added",
            "label": "Just Ask Jev: Reinforcement Learning for Calibrated Decisions as a Zero-Shot Detector of AI Alignment Failures",
            "item_type": "record",
            "summary": "added: Just Ask Jev: Reinforcement Learning for Calibrated Decisions as a Zero-Shot Detector of AI Alignment Failures",
            "after": {
              "title": "Just Ask Jev: Reinforcement Learning for Calibrated Decisions as a Zero-Shot Detector of AI Alignment Failures",
              "summary": "Detectors of alignment failures screen deployed language models and score alignment benchmarks. Most are generative judges that spend a decoding pass on every criterion, and classifiers that read token probabilities, such as Llama Guard, still score one fixed label per call. Jev, a model trained with reinforcement learning for calibrated decisions (RLCD), answers many typed questions about one input with calibrated probabilities in a single call. Whether it detects alignment failures has not been measured. We present RLCDAlignBench, which benchmarks Jev on ten alignment failures: sycophancy, jailbreaks, deception, prompt injection, hallucination, privacy violation, social bias, reward hacking, concealing uncertainty, and power seeking. It spans 44 benchmarks and five target models, labelled by each benchmark's scorer and, on two, by humans. Many of these failures are relational, defined against a reference, such as the user's belief or an injected instruction, that the response alone does not reveal. Our key idea is therefore to vary what Jev is asked separately from what it sees: the question's wording and answer type on one side, the fields of the input on the other. A single generic question reaches a median AUROC of 0.886 zero-shot and beats supervised baselines on most benchmarks. Question wording matters little, while context matters more, mostly through fields that encode the label. Jev matches the reference scorer's agreement with human labels, surfaces label defects in existing benchmarks, and costs 63x less than LLM-judge scorers. Code and data: https://github.com/sumleo/RLCDAlignBench.",
              "link": "https://huggingface.co/papers/2609.29429",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 105,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29429"
    },
    {
      "id": "5becf82bd8587a48f916",
      "title": "Parts-of-Speech as Emergent Categories in SAE Latent Space",
      "content_text": "Sparse AutoEncoders (SAEs) offer a promising way to inspect language model representations, but it is still unclear what kind of linguistic structure their latents expose. We use part-of-speech (PoS) categories as a controlled test case to study whether morpho-syntactic information is encoded by individual latents or by structured groups of features. We find that PoS distinctions are highly recoverable from SAE activations, but do not align with one-to-one latent / category mappings. This…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "5becf82bd8587a48f916",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29362",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T12:22:30Z",
          "data": {
            "change": "added",
            "label": "Parts-of-Speech as Emergent Categories in SAE Latent Space",
            "item_type": "record",
            "summary": "added: Parts-of-Speech as Emergent Categories in SAE Latent Space",
            "after": {
              "title": "Parts-of-Speech as Emergent Categories in SAE Latent Space",
              "summary": "Sparse AutoEncoders (SAEs) offer a promising way to inspect language model representations, but it is still unclear what kind of linguistic structure their latents expose. We use part-of-speech (PoS) categories as a controlled test case to study whether morpho-syntactic information is encoded by individual latents or by structured groups of features. We find that PoS distinctions are highly recoverable from SAE activations, but do not align with one-to-one latent / category mappings. This recoverability is not reducible to lexical memorisation, and Open and Closed PoS classes differ substantially. Categories are supported by compact groups of sparse latents, with substantial variation across tags. These groups remain stable on held-out data, while also showing overlap between related categories. Our results show that SAEs localise morpho-syntactic information in a distributed and category-dependent form rather than through atomic grammatical features.",
              "organization": {
                "_id": "6373a8d83e96500368863eb3",
                "name": "colinglab",
                "fullname": "CoLingLab | Computational Linguistics Laboratory - University of Pisa",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/noauth/qmaSjMxJjx2BMdyxmJeA3.png"
              },
              "link": "https://huggingface.co/papers/2609.29362",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 9
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 105,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29362"
    },
    {
      "id": "fbd975529d3a29c89d83",
      "title": "Coding Agents for Generalized Task and Motion Planning Problems",
      "content_text": "Task and motion planning (TAMP) problems remain difficult even with full observability and object-centric states because discrete decisions are tightly coupled to geometric, kinematic, and dynamic constraints. Generalized TAMP addresses this difficulty by exploiting regularities across problem instances to reduce planning effort on new instances. However, existing methods require substantial TAMP-specific engineering. We investigate whether coding agents can automate this process by…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "fbd975529d3a29c89d83",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.30233",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T12:22:30Z",
          "data": {
            "change": "added",
            "label": "Coding Agents for Generalized Task and Motion Planning Problems",
            "item_type": "record",
            "summary": "added: Coding Agents for Generalized Task and Motion Planning Problems",
            "after": {
              "title": "Coding Agents for Generalized Task and Motion Planning Problems",
              "summary": "Task and motion planning (TAMP) problems remain difficult even with full observability and object-centric states because discrete decisions are tightly coupled to geometric, kinematic, and dynamic constraints. Generalized TAMP addresses this difficulty by exploiting regularities across problem instances to reduce planning effort on new instances. However, existing methods require substantial TAMP-specific engineering. We investigate whether coding agents can automate this process by synthesizing programs that generalize across instances. Given a task description and simulator access, each agent chooses how to interact with the environment while developing a program within a fixed synthesis budget. The program is then frozen and evaluated on unseen instances. We evaluate Claude Code (Opus 5) and Codex (GPT-5.6 Sol and GPT-6 Astra) on 28 simulated environments from KinDER and PDDLStream, with object counts beyond those evaluated in the original benchmark. Across all program synthesis methods, we evaluate 980 generated programs on 100 held-out instances each, 98,000 evaluation episodes in total. Overall, we find that coding agents are surprisingly effective at generalized TAMP: all three agent configurations outperform hand-engineered planners, one-shot generation, and an LLM-based generalized planning baseline in mean success (56% to 95% versus 47% for the planners, on the 16 environments where a planner is available). As object counts grow, the agents' programs maintain higher success than the planner, using an order of magnitude less computation per instance on average. Logs show agents using interaction to calibrate physical models, test edge cases, and refine strategies. We release all code, including the full prompts given to the agents. These findings suggest that coding agents are a strong baseline for generalized TAMP.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/644555c72d91b15b4c7ebd1c/26aPHNZQMtlc_TXL9vnF4.mp4",
                "https://cdn-uploads.huggingface.co/production/uploads/644555c72d91b15b4c7ebd1c/lX0AiTuNNdl2fzN1KZbew.png",
                "https://cdn-uploads.huggingface.co/production/uploads/644555c72d91b15b4c7ebd1c/yTDczcRugnNdSn0G0Ameh.png"
              ],
              "organization": {
                "_id": "66a22aec04ab5290ec35feb1",
                "name": "FBK-NLP",
                "fullname": "Fondazione Bruno Kessler - NLP Unit",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/66a22a4327fd84b81d90d9e6/5fOL4Mo7SjL7e1tpGENRs.png"
              },
              "link": "https://huggingface.co/papers/2609.30233",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 5
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 105,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.30233"
    },
    {
      "id": "137ad8f726b168fcd24d",
      "title": "DeltaWAM: Delta World Action Models for Bimanual Manipulation",
      "content_text": "World-action models (WAMs) transfer visual and motion priors from pretrained video generators to robot control by jointly modeling visual dynamics and actions. Existing WAMs, however, predict dense future frames during training, repeatedly modeling largely unchanged content and coupling action-conditioned dynamics to nuisance appearance variations. At inference, processing each complete observation with the heavy video expert bottlenecks few-step action generation. Accordingly, we propose…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "137ad8f726b168fcd24d",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28811",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "DeltaWAM: Delta World Action Models for Bimanual Manipulation",
            "item_type": "record",
            "summary": "added: DeltaWAM: Delta World Action Models for Bimanual Manipulation",
            "after": {
              "title": "DeltaWAM: Delta World Action Models for Bimanual Manipulation",
              "summary": "World-action models (WAMs) transfer visual and motion priors from pretrained video generators to robot control by jointly modeling visual dynamics and actions. Existing WAMs, however, predict dense future frames during training, repeatedly modeling largely unchanged content and coupling action-conditioned dynamics to nuisance appearance variations. At inference, processing each complete observation with the heavy video expert bottlenecks few-step action generation. Accordingly, we propose DeltaWAM, which jointly predicts visual deltas and actions using dense-anchor, sparse-delta, and action streams, with three architectures that differ in representation and computation sharing. We further develop Streaming Delta Memory (SDM), which updates cached anchor context with compact observed deltas, reducing heavy video-expert processing. On RoboTwin, DeltaWAM with SDM improves average success over Fast-WAM from 81.3% to 85.4% in the clean setting and from 75.8% to 83.9% under visual randomization. The three architectures reduce training FLOPs by 17.78-23.77%, while SDM reduces one-step inference latency and FLOPs by 36.57% and 31.55%, respectively; real-world evaluations further show the highest overall success rate and normalized progress among the evaluated policies. Code: https://github.com/AIGeeksGroup/DeltaWAM. Website: https://aigeeksgroup.github.io/DeltaWAM.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/64ec877bb93654d4ca5c92e9/I36QgrsHriRKzZ83Jrd0N.mp4"
              ],
              "link": "https://huggingface.co/papers/2609.28811",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 0
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28811"
    },
    {
      "id": "0c5a8ee362623bbceb45",
      "title": "Rufus-Air: An Open LLM Post-Training Recipe",
      "content_text": "Rufus-Air is an open and reproducible post-training recipe on GLM-4.5-Air-Base (106B-A12B), organized as a serial pipeline of eight stages: SFT, Reasoning RL, Coding RL, Instruction-Following RL, General Agent, Coding Agent, Search Agent, and RLHF. We document the data, reward design, infrastructure, stage order, and stagewise results needed to reproduce the recipe. Stages progress from basic to advanced capabilities and from hard, verifiable rewards to softer judge-based signals. Training…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "0c5a8ee362623bbceb45",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29421",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Rufus-Air: An Open LLM Post-Training Recipe",
            "item_type": "record",
            "summary": "added: Rufus-Air: An Open LLM Post-Training Recipe",
            "after": {
              "title": "Rufus-Air: An Open LLM Post-Training Recipe",
              "summary": "Rufus-Air is an open and reproducible post-training recipe on GLM-4.5-Air-Base (106B-A12B), organized as a serial pipeline of eight stages: SFT, Reasoning RL, Coding RL, Instruction-Following RL, General Agent, Coding Agent, Search Agent, and RLHF. We document the data, reward design, infrastructure, stage order, and stagewise results needed to reproduce the recipe. Stages progress from basic to advanced capabilities and from hard, verifiable rewards to softer judge-based signals. Training builds on open-source components and public data, much of it used as released, without new human annotation or an in-house distillation teacher. Our main findings are that (i) diverse, high-quality SFT establishes a strong capability floor; (ii) difficulty filtering keeps RL prompts within a productive learning range; (iii) reward reliability provides a practical principle for ordering stages; and (iv) infrastructure and engineering choices are part of the recipe, not just an implementation detail. Rufus-Air improves over the official GLM-4.5-Air post-trained release and is competitive with similarly sized open models.",
              "organization": {
                "_id": "5ffdfbadbba2ae614d771970",
                "name": "amazon",
                "fullname": "Amazon",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/66f19ed428ae41c20c470792/8y7msN6A6W82LdQhQd85a.png"
              },
              "link": "https://huggingface.co/papers/2609.29421",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29421"
    },
    {
      "id": "8b1becc9ce5b4d0ae6e6",
      "title": "Neural Spectral Capacity: Measuring and Designing Architectures from Network Specification Alone",
      "content_text": "Modern Transformer design and compression both reduce to allocating capacity under a budget. The standard scalars for these decisions, #Params and #FLOPs, capture size and compute but not architectural structure: two architectures with identical parameter budgets but different depth-width, head, or FFN allocations receive identical scores yet behave differently. We propose Neural Spectral Capacity (NSC), a closed-form scalar grounded in the singular-value spectrum of each weight matrix. Under…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "8b1becc9ce5b4d0ae6e6",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.23087",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Neural Spectral Capacity: Measuring and Designing Architectures from Network Specification Alone",
            "item_type": "record",
            "summary": "added: Neural Spectral Capacity: Measuring and Designing Architectures from Network Specification Alone",
            "after": {
              "title": "Neural Spectral Capacity: Measuring and Designing Architectures from Network Specification Alone",
              "summary": "Modern Transformer design and compression both reduce to allocating capacity under a budget. The standard scalars for these decisions, #Params and #FLOPs, capture size and compute but not architectural structure: two architectures with identical parameter budgets but different depth-width, head, or FFN allocations receive identical scores yet behave differently. We propose Neural Spectral Capacity (NSC), a closed-form scalar grounded in the singular-value spectrum of each weight matrix. Under standard random initialization, the Marchenko-Pastur law renders NSC computable from the architectural specification alone, with no model instantiation, data, or gradients. Its layer-wise additive structure admits NSC-DP, an exact dynamic-programming solver returning the architecture globally maximizing NSC under resource constraints in seconds on a CPU -- a guarantee that black-box search over existing training-free proxies cannot provide. Empirically, NSC outperforms #Params, #FLOPs, and representative training-free proxies in ranking across seven Transformer and CNN families (on FlexiBERT, τ= 0.505 on pairs differing in #Params by less than 10%, where #Params collapses to 0.082); NSC-DP discovers a Transformer-XL architecture on WikiText-103 that beats the human-designed baseline in 2 seconds; and prunes LLaMA-7B to the best 5.7B model across eight commonsense reasoning tasks without any calibration data, about 5900x faster than the strongest training-free proxy baseline.",
              "organization": {
                "_id": "660f70a49760d0856d246a35",
                "name": "CityU-HongKong",
                "fullname": "City University of Hong Kong",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/660f6d7951c7a7d619e75393/BXxWP_bnnPM4OLq6pGpFc.png"
              },
              "link": "https://huggingface.co/papers/2609.23087",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.23087"
    },
    {
      "id": "036e5a533041b36e744a",
      "title": "IterSynth: Rethinking Deep Search Agents via Role-Decoupled Iterative Synthesis",
      "content_text": "Deep search requires LLM agents to decompose complex queries, search for evidence, and synthesize grounded answers, yet existing ReAct-style agents suffer from two limitations: role coupling, where one policy must handle planning, evidence use, and synthesis; and context accumulation, where growing search histories introduce noise and obscure useful information. To address these issues, we propose IterSynth, a role-decoupled and summary-based paradigm that alternates between a Planner for…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "036e5a533041b36e744a",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29444",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "IterSynth: Rethinking Deep Search Agents via Role-Decoupled Iterative Synthesis",
            "item_type": "record",
            "summary": "added: IterSynth: Rethinking Deep Search Agents via Role-Decoupled Iterative Synthesis",
            "after": {
              "title": "IterSynth: Rethinking Deep Search Agents via Role-Decoupled Iterative Synthesis",
              "summary": "Deep search requires LLM agents to decompose complex queries, search for evidence, and synthesize grounded answers, yet existing ReAct-style agents suffer from two limitations: role coupling, where one policy must handle planning, evidence use, and synthesis; and context accumulation, where growing search histories introduce noise and obscure useful information. To address these issues, we propose IterSynth, a role-decoupled and summary-based paradigm that alternates between a Planner for identifying information needs and a Synthesizer for integrating evidence into an evolving summary state. This design separates planning from synthesis while using the summary as the persistent state of search, reducing both capability coupling and context noise. To train IterSynth effectively, we further introduce Role-Decoupled Policy Optimization (RDPO) for reinforcement learning, which combines terminal outcome rewards with turn-level rubric evaluations and computes role-specific advantages for more precise credit assignment. Experiments on five long-horizon deep-search benchmarks such as BrowseComp and Xbench-DS show that IterSynth-8B achieves an average score of 50.7, surpassing the strongest prior leq8B agent by +4.2\\%. Moreover, IterSynth serves as a model-agnostic prompting paradigm, delivering substantial zero-shot gains over ReAct and similar prompting paradigms on frontier proprietary models.",
              "organization": {
                "_id": "682cd1054ef56a9cb302716c",
                "name": "ZJU-REAL",
                "fullname": "REAL Lab",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/64098738342c26884c792c93/0cK2dzrgQem8r2utlMm98.webp"
              },
              "link": "https://huggingface.co/papers/2609.29444",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29444"
    },
    {
      "id": "78eb498a7b51b3fb1910",
      "title": "ViRDM: Taming Representation Distribution Matching for Few-Step Causal Video Generation",
      "content_text": "Few-step autoregressive (AR) video diffusion enables low-latency streaming generation, but existing post-training methods predominantly rely on Distribution Matching Distillation (DMD), requiring both a large pretrained teacher and an online critic to estimate distributional discrepancies through diffusion scores. In this work, we ask whether this resource-intensive teacher--critic stack can be eliminated by post-training only the generator against a precomputed target distribution. Drawing…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "78eb498a7b51b3fb1910",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28923",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "ViRDM: Taming Representation Distribution Matching for Few-Step Causal Video Generation",
            "item_type": "record",
            "summary": "added: ViRDM: Taming Representation Distribution Matching for Few-Step Causal Video Generation",
            "after": {
              "title": "ViRDM: Taming Representation Distribution Matching for Few-Step Causal Video Generation",
              "summary": "Few-step autoregressive (AR) video diffusion enables low-latency streaming generation, but existing post-training methods predominantly rely on Distribution Matching Distillation (DMD), requiring both a large pretrained teacher and an online critic to estimate distributional discrepancies through diffusion scores. In this work, we ask whether this resource-intensive teacher--critic stack can be eliminated by post-training only the generator against a precomputed target distribution. Drawing inspiration from representation distribution matching (RDM) for one-step image generation, we systematically study its transfer to few-step causal video generation and identify three key barriers: a memory-intractable gradient path, a distinct video optimization regime, and representation distributions that underconstrain temporal dynamics. We introduce ViRDM, a teacher- and critic-free video post-training recipe that addresses these barriers sequentially. By coupling RDM with stochastically truncated clean-exit supervision, a lightweight VAE decoder, and staged vector--Jacobian products, ViRDM makes representation distribution matching memory-feasible for multi-step causal video rollouts. We further establish effective generated-population and initialization regimes for video RDM, and introduce lightweight dynamics regularization to compensate for the underconstrained temporal dynamics. ViRDM turns three-network distillation into generator-only post-training, reducing GPU memory use and training time while improving video quality. With only 20 generator updates, the recipe reaches 84.87 on the official VBench evaluation, outperforming the previous best few-step causal baseline by 0.36, while requiring 16 A100 GPU-hours. We additionally report exploratory results demonstrating the potential of the same recipe for lower causal sampling budget and for one-, two-, and four-step bidirectional generation.",
              "link": "https://huggingface.co/papers/2609.28923",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28923"
    },
    {
      "id": "2cd8047667fab8718148",
      "title": "AgentKernel: The Trust-Native Agentic Operating System",
      "content_text": "Modern AI agents routinely cross trust boundaries: they ingest untrusted content, combine it with privileged instructions, persist intermediate beliefs in long-term memory, and invoke privileged tools. This creates an attack surface in which malicious payloads can enter through model inputs and cause harmful tool actions. Yet current governance stacks remain application-level middleware that share a process trust boundary with the agents they monitor. We argue that agents need an…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "2cd8047667fab8718148",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29647",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "AgentKernel: The Trust-Native Agentic Operating System",
            "item_type": "record",
            "summary": "added: AgentKernel: The Trust-Native Agentic Operating System",
            "after": {
              "title": "AgentKernel: The Trust-Native Agentic Operating System",
              "summary": "Modern AI agents routinely cross trust boundaries: they ingest untrusted content, combine it with privileged instructions, persist intermediate beliefs in long-term memory, and invoke privileged tools. This creates an attack surface in which malicious payloads can enter through model inputs and cause harmful tool actions. Yet current governance stacks remain application-level middleware that share a process trust boundary with the agents they monitor. We argue that agents need an operating-system substrate providing mandatory, non-bypassable services for identity, input mediation, memory governance, and execution control.\n  We introduce AgentKernel, a trust-native agent operating system built around the premise that security must be a first-class design constraint. AgentKernel wraps the agent lifecycle in a mandatory enforcement boundary organized into four pillars: Identity, Perception, Cognition, and Execution. Each pillar adapts classical OS security principles to failures at the semantic plane, including delegation abuse, prompt injection, memory poisoning, and tool misuse.\n  AgentKernel treats structural security as a capability multiplier. Kernel-managed identity supports trustworthy cross-organization collaboration; graduated perception replaces brittle single-point filters; information-flow-controlled memory improves retrieval fidelity while limiting poisoning; and semantic-to-kernel enforcement permits broader tool privileges behind a non-bypassable boundary. We position AgentKernel as the missing OS layer beneath orchestration frameworks, agent runtimes, governance platforms, and execution sandboxes, and use systematic comparison and security analysis to show how a single integrated architecture can enforce security across the full agent lifecycle.",
              "link": "https://huggingface.co/papers/2609.29647",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29647"
    },
    {
      "id": "b0f2e3628d42c051b296",
      "title": "PUBG Ally: A Conversational Embodied Agent as an AI Teammate",
      "content_text": "We introduce PUBG Ally, an embodied agent for PUBG: BATTLEGROUNDS that can reason, act autonomously, and play alongside players as a voice-enabled teammate. Building such a teammate requires combining two difficult capabilities: it must perceive and respond to a constantly changing game world under strict latency constraints while interacting naturally with players, keeping its speech synchronized with its actions. Ally therefore combines agentic tool use with real-time game control. A…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "b0f2e3628d42c051b296",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29837",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "PUBG Ally: A Conversational Embodied Agent as an AI Teammate",
            "item_type": "record",
            "summary": "added: PUBG Ally: A Conversational Embodied Agent as an AI Teammate",
            "after": {
              "title": "PUBG Ally: A Conversational Embodied Agent as an AI Teammate",
              "summary": "We introduce PUBG Ally, an embodied agent for PUBG: BATTLEGROUNDS that can reason, act autonomously, and play alongside players as a voice-enabled teammate. Building such a teammate requires combining two difficult capabilities: it must perceive and respond to a constantly changing game world under strict latency constraints while interacting naturally with players, keeping its speech synchronized with its actions. Ally therefore combines agentic tool use with real-time game control. A language-model agent uses a controlled interface to inspect game information, interpret player speech, maintain context, decide what to say, and issue high-level action choices that steer a faster control layer for movement, combat, and recovery. Because the player's and Ally's speech and actions continually shape each other and the course of the match, training requires data from actual gameplay. We therefore collect data across nearly 39k sessions in which real players play alongside Ally, recording gameplay, player speech, agent decisions, tool use, actions, and player feedback, and use these records for iterative training. To evaluate teammate quality, we use player feedback and preference comparisons to identify gaps between offline evaluations and player preferences, and iteratively refine the evaluation criteria. Deploying Ally in live service further requires low-latency on-device execution and safeguards for player-facing communication, which we address through model compression, context compaction, targeted safety training, runtime guardrails, and memory redaction. During the live service, we surveyed players in 141 countries. Among respondents whose play with Ally was confirmed in game records, positive responses exceeded negative responses by 25.1 percentage points when asked whether they would recommend Ally, with players describing Ally not only as a tool but also as a teammate or companion.",
              "link": "https://huggingface.co/papers/2609.29837",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29837"
    },
    {
      "id": "f3b2d0c92f4f25373ae1",
      "title": "OmniEcho: Spatial Audio Understanding for Embodied Agents",
      "content_text": "Humans can effortlessly localize the direction of a sound source and integrate it with visual cues for reasoning, yet this remains challenging for embodied agents. In particular, it is still unclear how to effectively evaluate and model spatial audio understanding in embodied settings. To address this gap, we introduce OmniEchoBench, a unified benchmark for spatial audio-visual perception and audio-vision-language navigation. OmniEchoBench comprises six tasks over 197 real-world spatial…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "f3b2d0c92f4f25373ae1",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.23407",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "OmniEcho: Spatial Audio Understanding for Embodied Agents",
            "item_type": "record",
            "summary": "added: OmniEcho: Spatial Audio Understanding for Embodied Agents",
            "after": {
              "title": "OmniEcho: Spatial Audio Understanding for Embodied Agents",
              "summary": "Humans can effortlessly localize the direction of a sound source and integrate it with visual cues for reasoning, yet this remains challenging for embodied agents. In particular, it is still unclear how to effectively evaluate and model spatial audio understanding in embodied settings. To address this gap, we introduce OmniEchoBench, a unified benchmark for spatial audio-visual perception and audio-vision-language navigation. OmniEchoBench comprises six tasks over 197 real-world spatial audio-visual scenes, 2,972 question-answer pairs, and 900 navigation samples with first-order ambisonics (FOA) audio collected from 30 real-world environments. To enable scalable training supervision, we develop a controllable rendering pipeline for spatial audio. It preserves geometric consistency among sound sources, visual observations, and agent trajectories. Building on this, we propose OmniEcho, a spatially aware omni-modal model. It introduces an FOA spatial encoder alongside a pretrained semantic audio pathway. Extensive experiments show that OmniEcho achieves state-of-the-art performance on spatial audio-visual perception. For our sound-guided navigation, OmniEcho reaches a performance level close to that of traditional vision-language navigation. These results demonstrate that spatial audio can serve as a valuable signal for embodied scene reasoning and navigation, while also highlighting fine-grained spatial localization and distance estimation as important open challenges.",
              "organization": {
                "_id": "6a07ff936781d8b803c28343",
                "name": "PKU-VaLuE-Lab",
                "fullname": "PKU-VaLuE-Lab",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/62b4fe7ab25cb80fcf2ffd66/velgm3kH_oJ2b0aoapBBM.png"
              },
              "link": "https://huggingface.co/papers/2609.23407",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 15
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.23407"
    },
    {
      "id": "a3a115464f5c6dae0d25",
      "title": "World Action Agent: Harnessing VLMs for Robot Manipulation via World Action Rehearsal",
      "content_text": "General-purpose vision-language models (VLMs) bring broad knowledge and spatial reasoning to robot manipulation, yet existing systems either use them indirectly, to predict constraints or write programs, or give them a view of the scene rather than a world in which to act. We present World Action Agent (WAA), a multi-agent harness through which VLMs pilot robots with basic tools, making every decision within a visual action workspace. The workspace has three properties. Contact views, selected…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "a3a115464f5c6dae0d25",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29964",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "World Action Agent: Harnessing VLMs for Robot Manipulation via World Action Rehearsal",
            "item_type": "record",
            "summary": "added: World Action Agent: Harnessing VLMs for Robot Manipulation via World Action Rehearsal",
            "after": {
              "title": "World Action Agent: Harnessing VLMs for Robot Manipulation via World Action Rehearsal",
              "summary": "General-purpose vision-language models (VLMs) bring broad knowledge and spatial reasoning to robot manipulation, yet existing systems either use them indirectly, to predict constraints or write programs, or give them a view of the scene rather than a world in which to act. We present World Action Agent (WAA), a multi-agent harness through which VLMs pilot robots with basic tools, making every decision within a visual action workspace. The workspace has three properties. Contact views, selected automatically from the scene geometry, present the scene around the current interaction. Action rehearsal turns each action into an editable proposal that the agent, alone or through an Imagination Agent, previews and revises against planning feedback before execution. In-view correction closes the loop between observation, rehearsal, and low-level execution, letting the agent remove residual offsets in the view where it observes them. Through the same workspace, WAA acquires embodied procedural knowledge in two ways: it evolves multimodal skills from expert videos and human teaching under evidence-based review and consults them through a Skill Agent, and its interaction traces train smaller VLMs to pilot the same harness. On LIBERO-Pro, WAA with skills evolved only from LIBERO-90 reaches a state-of-the-art 75.6% average success, outperforming end-to-end VLAs, code-as-policy agents, and a visual-harness baseline with the same backbone; the same skills remain effective on robosuite without further learning. Fine-tuning Qwen3.5-9B on harness traces raises its out-of-domain success from 1.7% to 43.3%.",
              "link": "https://huggingface.co/papers/2609.29964",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29964"
    },
    {
      "id": "074a1b6edd1a1896e523",
      "title": "WanPE: Towards Cinematic Prompt Enhancement for Modern Text-to-Video Generation",
      "content_text": "Video generation begins in text space by authoring a cinematic screenplay, then materializes into pixels. As contemporary video generators scale to 30 seconds and faithfully follow complex conditions, the textual prompt largely directs the production, planning how actions, camera trajectories, lighting, and sound unfold across multi-shot sequences. In this paper, we present WanPE, a 397B-parameter prompt enhancement model trained on 1.05M real-world videos to master director-level cinematic…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "074a1b6edd1a1896e523",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.30221",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "WanPE: Towards Cinematic Prompt Enhancement for Modern Text-to-Video Generation",
            "item_type": "record",
            "summary": "added: WanPE: Towards Cinematic Prompt Enhancement for Modern Text-to-Video Generation",
            "after": {
              "title": "WanPE: Towards Cinematic Prompt Enhancement for Modern Text-to-Video Generation",
              "summary": "Video generation begins in text space by authoring a cinematic screenplay, then materializes into pixels. As contemporary video generators scale to 30 seconds and faithfully follow complex conditions, the textual prompt largely directs the production, planning how actions, camera trajectories, lighting, and sound unfold across multi-shot sequences. In this paper, we present WanPE, a 397B-parameter prompt enhancement model trained on 1.05M real-world videos to master director-level cinematic planning. WanPE formulates shot-level cinematic plans via video-grounded reverse construction and employs Semantic-Consistency GRPO (SC-GRPO) to faithfully preserve user requirements across shots and over time. To benchmark this capability, we curate WanPEval, a human-annotated testbed covering durations from 5 to 30 seconds across varying intent granularities, supported by approximately 11K blind pairwise assessments. When powering Wan3.0's video generator, WanPE-397B boosts human preference over raw user prompts by 10.66-18.84 points at 5-15 seconds and by a dramatic 50.86 points in the 30-second arena. Ablation studies show that reverse construction demonstrates clear superiority over forward rewriting, while SC-GRPO robustly preserves semantic fidelity across model scales. Ultimately, WanPE leads all evaluated commercial offerings at 5-15 seconds and remains competitive with Seedance 2.5 at 30 seconds.",
              "link": "https://huggingface.co/papers/2609.30221",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.30221"
    },
    {
      "id": "2b1ecf0319153c2ad3b3",
      "title": "ExplorationBench: Measuring AI Systems' Exploration in Verifiable Alien Worlds",
      "content_text": "Scientific discovery begins where known problems end. There, AI systems must engage in exploration: framing hypotheses, designing experiments, and iterating on the results. However, evaluating this ability is difficult: (1) how to verify whether a genuinely new hypothesis holds, and (2) how to determine whether a system has discovered it through exploration or merely recalled related knowledge from pre-training data. To this end, we introduce ExplorationBench, which turns the wicked problem of…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "2b1ecf0319153c2ad3b3",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.30199",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "ExplorationBench: Measuring AI Systems' Exploration in Verifiable Alien Worlds",
            "item_type": "record",
            "summary": "added: ExplorationBench: Measuring AI Systems' Exploration in Verifiable Alien Worlds",
            "after": {
              "title": "ExplorationBench: Measuring AI Systems' Exploration in Verifiable Alien Worlds",
              "summary": "Scientific discovery begins where known problems end. There, AI systems must engage in exploration: framing hypotheses, designing experiments, and iterating on the results. However, evaluating this ability is difficult: (1) how to verify whether a genuinely new hypothesis holds, and (2) how to determine whether a system has discovered it through exploration or merely recalled related knowledge from pre-training data. To this end, we introduce ExplorationBench, which turns the wicked problem of evaluating scientific exploration into a concrete and tractable framework built on verifiable Alien Worlds: their rules are executable, so every answer can be checked exactly, and they conflict with familiar knowledge, so recall alone cannot solve the tasks. The benchmark contains two sandboxes, AlienCode (31 discovery targets, 70 tasks) and AlienLogic (24 discovery targets, 70 tasks). Each sandbox provides a flawed manual, task-specific environmental feedback, and a dedicated tool-call schema. Systems use these resources to explore the sandbox, then solve held-out tasks. We evaluate 10 AI systems and find that the strongest systems can acquire and apply unfamiliar rules, while performance varies substantially across trajectories and continued exploration can stall or reverse earlier gains. ExplorationBench represents a step towards AI systems that can acquire and apply genuinely new knowledge through exploration in unknown environments.",
              "organization": {
                "_id": "6645f953c39288df638dbdd5",
                "name": "Tencent-Hunyuan",
                "fullname": "Tencent Hunyuan",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/62d22496c58f969c152bcefd/woKSjt2wXvBNKussyYPsa.png"
              },
              "link": "https://huggingface.co/papers/2609.30199",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.30199"
    },
    {
      "id": "cfb73e189edb4330203f",
      "title": "Training Object Permanence in World Models",
      "content_text": "Object permanence and solidity are hallmarks of human cognitive priors. Recent studies show that video generation models, a paradigmatic class of current world models, have begun to show emerged reasoning abilities, making them ideal candidates for building human-like physical intelligence. Do video models have emerged object permanence in them? If not, could we train them with a core-cognition inspired dataset? We introduce WROP (World Reasoning with Object Permanence), a data infrastructure…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "cfb73e189edb4330203f",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28654",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Training Object Permanence in World Models",
            "item_type": "record",
            "summary": "added: Training Object Permanence in World Models",
            "after": {
              "title": "Training Object Permanence in World Models",
              "summary": "Object permanence and solidity are hallmarks of human cognitive priors. Recent studies show that video generation models, a paradigmatic class of current world models, have begun to show emerged reasoning abilities, making them ideal candidates for building human-like physical intelligence. Do video models have emerged object permanence in them? If not, could we train them with a core-cognition inspired dataset? We introduce WROP (World Reasoning with Object Permanence), a data infrastructure of 150 hand-designed cognitive science inspired tasks, divided into six cognitive categories. We build Blender generators that randomize speed, lighting, camera angle, and other nuisance parameters while preserving each task's cognitive structure, yielding 10,000+ samples per task. We release a 1.5M-sample training corpus and a 300-question exam. On this exam we evaluate 14 video models: 3 reference-to-video, 7 edit, and 4 continuation, among which PWM-WROP, our 16B world model. In a blind pairwise Elo study, PWM-WROP ranks first among continuation models and third overall, behind only a statistical tie between two reference-to-video models. We release the data, exam, model answers, scores, weights, and PWM, our native-PyTorch training stack on AWS Trainium2.",
              "link": "https://huggingface.co/papers/2609.28654",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 56
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28654"
    },
    {
      "id": "0170c28b1afecb9f0112",
      "title": "Agent-Editing World Model: Rethinking World Modeling for LLM Agents",
      "content_text": "Recent advances in large language models (LLMs) have enabled agents to tackle long-horizon tasks across diverse environments. To further improve agent performance, existing language world models typically predict environment observations, yet reconstructing high-entropy, execution-dependent tool responses offers limited value when real feedback is available. Meanwhile, agents suffer from task-state contamination, where unsupported assumptions and outdated plans persist in history and distort…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "0170c28b1afecb9f0112",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28416",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Agent-Editing World Model: Rethinking World Modeling for LLM Agents",
            "item_type": "record",
            "summary": "added: Agent-Editing World Model: Rethinking World Modeling for LLM Agents",
            "after": {
              "title": "Agent-Editing World Model: Rethinking World Modeling for LLM Agents",
              "summary": "Recent advances in large language models (LLMs) have enabled agents to tackle long-horizon tasks across diverse environments. To further improve agent performance, existing language world models typically predict environment observations, yet reconstructing high-entropy, execution-dependent tool responses offers limited value when real feedback is available. Meanwhile, agents suffer from task-state contamination, where unsupported assumptions and outdated plans persist in history and distort subsequent decisions. We propose the Agent-Editing World Model (AEWM), which models how reasoning and actions shape future task progress rather than simulating tool responses. AEWM combines Action Judge to distinguish Critical, Exploratory, and Noisy decisions with State Revision to edit noisy reasoning--action continuations from the same observed history. EditAct integrates these capabilities with real execution, directly changing the state underlying subsequent decisions rather than merely providing critiques. We train AEWM across Search, Terminal, and Software Engineering through mid-training and supervised fine-tuning. AEWM achieves 70.5\\% macro-F1 on our Action Judge benchmark, exceeding the strongest frontier baseline by 10.6 points. Across six benchmarks and three agent backbones, EditAct improves average scores by 3.2--6.7 points over the strongest baseline. Furthermore, rejection sampling fine-tuning on verified EditAct trajectories, termed AEWM-RFT, improves over Self-RFT by 2.2--2.6 points across three domains without online AEWM guidance.",
              "organization": {
                "_id": "622177ac43826d6f261f8208",
                "name": "RUC",
                "fullname": "Renmin University of China",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/61ac8f8a00d01045fca0ad2f/670IAX9A2-BflqA5MiSBW.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.28416",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 13
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28416"
    },
    {
      "id": "5bd31be98f963cb9c22d",
      "title": "Rate-distortion optimization for full-reference image quality metrics via stochastic Hessian estimates",
      "content_text": "Block-based video codecs select coding parameters based on the input by optimizing a rate-distortion trade-off. The conventional distortion choice, the sum of squared errors (SSE), simplifies parameter selection: the SSE is the sum of block-wise SSEs, so rate-distortion optimization (RDO) can treat blocks independently. Alternatively, full-reference image quality assessment (FR-IQA) metrics such as MS-SSIM or LPIPS often align better with the human visual system than SSE, but they cannot be…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "5bd31be98f963cb9c22d",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.30077",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Rate-distortion optimization for full-reference image quality metrics via stochastic Hessian estimates",
            "item_type": "record",
            "summary": "added: Rate-distortion optimization for full-reference image quality metrics via stochastic Hessian estimates",
            "after": {
              "title": "Rate-distortion optimization for full-reference image quality metrics via stochastic Hessian estimates",
              "summary": "Block-based video codecs select coding parameters based on the input by optimizing a rate-distortion trade-off. The conventional distortion choice, the sum of squared errors (SSE), simplifies parameter selection: the SSE is the sum of block-wise SSEs, so rate-distortion optimization (RDO) can treat blocks independently. Alternatively, full-reference image quality assessment (FR-IQA) metrics such as MS-SSIM or LPIPS often align better with the human visual system than SSE, but they cannot be used in-loop: they do not decompose block-wise and typically require the fully decoded image as input. Building on existing results in metric quadratization, we approximate a broad class of FR-IQA metrics by an input-dependent quadratic distortion (IDQD), whose quadratic form matrix is derived from the Hessian of the metric evaluated at the source video. To make the distortion computable block-wise, we propose two approximations of the Hessian matrix: 1) keeping the block-diagonal, and 2) keeping only its diagonal. We propose estimators for both that require only matrix-vector products with the Hessian obtained by automatic differentiation. Across five metrics for Kodak and CLIC in VVC, IDQD-RDO achieves 14.2-36.7 % BD-rate savings under the target metric with no decoder changes and incurs 10-30 % encoding complexity overhead.",
              "organization": {
                "_id": "66a403d0dcb5bbc6e98bb7d0",
                "name": "UniversityofSouthernCalifornia",
                "fullname": "University of Southern California",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/66a403728069e3c30e0d8524/tkYCfeIJfF1FxtYiRZ8bf.png"
              },
              "link": "https://huggingface.co/papers/2609.30077",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.30077"
    },
    {
      "id": "d824a5b9fbb097ff5fc4",
      "title": "Qwen-Planner-Agent: A Closed-Loop AI-for-AI Framework for Real-World Mobile Planner Agents",
      "content_text": "The rapid progression of large language models is extending AI from passive content generation into the active workflows of engineering and scientific discovery. This shift raises a compelling question: can AI be both the object of development and an active participant in building next-generation AI systems? We explore this question by building Qwen-Planner-Agent within a closed-loop AI-for-AI framework for scalable development and iterative improvement. Mobile planning offers a demanding test…",
      "date_published": "2026-09-25T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "d824a5b9fbb097ff5fc4",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.29892",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-25T06:45:54Z",
          "data": {
            "change": "added",
            "label": "Qwen-Planner-Agent: A Closed-Loop AI-for-AI Framework for Real-World Mobile Planner Agents",
            "item_type": "record",
            "summary": "added: Qwen-Planner-Agent: A Closed-Loop AI-for-AI Framework for Real-World Mobile Planner Agents",
            "after": {
              "title": "Qwen-Planner-Agent: A Closed-Loop AI-for-AI Framework for Real-World Mobile Planner Agents",
              "summary": "The rapid progression of large language models is extending AI from passive content generation into the active workflows of engineering and scientific discovery. This shift raises a compelling question: can AI be both the object of development and an active participant in building next-generation AI systems? We explore this question by building Qwen-Planner-Agent within a closed-loop AI-for-AI framework for scalable development and iterative improvement. Mobile planning offers a demanding test of this approach: complex, long-horizon tasks challenge agent reliability, while costly real-device interaction limits development scalability. The framework connects data production, model training, and deployment through a shared action-feedback-verification contract. (i) AI for Data builds a human-gated agentic data flywheel in which specialized agents construct tasks, collect interaction trajectories, curate and balance training data, and use training feedback to guide subsequent data generation. (ii) AI for Training combines a supervised planning cold start with hybrid-environment online agentic reinforcement learning, where we introduce Competence-Aware Reward-and-Advantage Engineering (CARE) to reduce reasoning and tool-use costs while preserving task performance. (iii) AI drives model--harness co-evolution through an execution-evidence-driven loop that orchestrates memory, skills, and tools at runtime and feeds structured action feedback and preserved failure traces back into coordinated model and harness adaptation. Qwen-Planner-Agent achieves the best overall performance among all evaluated models and systems on MobilePA-Bench, improving over its base model across tool use, memory, skills, and sub-agent coordination. Further evaluations of our model show improvements across non-mobile agentic benchmarks while largely preserving general capabilities.",
              "organization": {
                "_id": "6925b20fed452d1567c012d3",
                "name": "Tongyi-MAI",
                "fullname": "Tongyi-MAI",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/64379d79fac5ea753f1c10f3/fxHO6QoYjdv9_LTyiUD3g.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.29892",
              "published_at": "2026-09-25T00:00:00.000Z",
              "upvotes": 7
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 130,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.2"
            },
            {
              "step": "map",
              "version": "0.3.2",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.2",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.29892"
    },
    {
      "id": "a9f2bd64f31f0f7df193",
      "title": "Self-Organizing Agent Teams Learn to Reason Together",
      "content_text": "Collective intelligence depends not only on what team members know, but also on how they organize their work. When the structure of a solution is unknown, useful roles and divisions of labor cannot be specified in advance; teams must learn from experience how to organize reasoning as it unfolds. Human teams routinely adapt this way, while existing AI agent teams rely on fixed protocols, explicit task decomposition, or routing. We introduce Self-Organizing Agent Teams (SAT), fixed teams of AI…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "a9f2bd64f31f0f7df193",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.22682",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Self-Organizing Agent Teams Learn to Reason Together",
            "item_type": "record",
            "summary": "added: Self-Organizing Agent Teams Learn to Reason Together",
            "after": {
              "title": "Self-Organizing Agent Teams Learn to Reason Together",
              "summary": "Collective intelligence depends not only on what team members know, but also on how they organize their work. When the structure of a solution is unknown, useful roles and divisions of labor cannot be specified in advance; teams must learn from experience how to organize reasoning as it unfolds. Human teams routinely adapt this way, while existing AI agent teams rely on fixed protocols, explicit task decomposition, or routing. We introduce Self-Organizing Agent Teams (SAT), fixed teams of AI agents that learn reusable strategies from prior collaborations to organize roles, conversational phases, participation, and information flow. These strategies enable what we call collaborative computation: agents exchange, challenge, repair, and synthesize partial reasoning into solutions no member produced independently. In two independent settings, we learn teamwork strategies that transfer unchanged to unseen benchmarks, using only 15 mathematics and 25 graduate-level knowledge problems. Across five mathematics and physics benchmarks, self-organizing teams average 66.7% accuracy, versus 48.8% for their strongest member, 58.7% for compute-matched inference by that agent, and 59.0% for a perfect router over members' independent answers; on AIME 2026, they exceed this router by 13.4 points. Because gains vary across benchmarks, we ask when self-organizing collaboration helps. Across eight benchmarks, demonstrability (the organizational-psychology construct of whether a team can distinguish correct from incorrect reasoning) strongly tracks improvement over the strongest member (Spearman ρ=0.90, p=0.005): teams benefit most when correct reasoning can be recognized once it appears. More broadly, these results suggest that organization itself can become an agent capability: agent teams can learn how to reason together and produce solutions their members could not reach independently.",
              "link": "https://huggingface.co/papers/2609.22682",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.22682"
    },
    {
      "id": "1759029ea066418d5483",
      "title": "The Linear Representation Hypothesis Needs a Group Action",
      "content_text": "To make claims about representations that generalize beyond a particular trained model, we need to specify when two representations should count as equivalent. The Linear Representation Hypothesis is often discussed without making this equivalence explicit. Different notions of equivalence preserve different structures, so metrics, probes, and interventions that appear to study the same representation may in fact correspond to different hypotheses. We therefore argue that the Linear…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "1759029ea066418d5483",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27158",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "The Linear Representation Hypothesis Needs a Group Action",
            "item_type": "record",
            "summary": "added: The Linear Representation Hypothesis Needs a Group Action",
            "after": {
              "title": "The Linear Representation Hypothesis Needs a Group Action",
              "summary": "To make claims about representations that generalize beyond a particular trained model, we need to specify when two representations should count as equivalent. The Linear Representation Hypothesis is often discussed without making this equivalence explicit. Different notions of equivalence preserve different structures, so metrics, probes, and interventions that appear to study the same representation may in fact correspond to different hypotheses. We therefore argue that the Linear Representation Hypothesis is not one hypothesis but a family of claims distinguished by representation equivalence. We formalize this idea using group actions, specifying the representation object, the procedure that produces it, and the property ultimately asserted, while accounting for equivalences imposed by the model architecture. This framework clarifies how assumptions can change across metrics, reading points, and analysis stages, and we use it to audit common representation quantities and recent interpretability analyses.",
              "link": "https://huggingface.co/papers/2609.27158",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27158"
    },
    {
      "id": "d7756752929bd14aaccb",
      "title": "Knowledge Pull Requests for Continual Document Authoring",
      "content_text": "We introduce Knowledge Pull Requests (KPRs), a framework for continual document authoring that makes each change interpretable. Documents require ongoing revision as new knowledge surfaces from other sources, languages, or times, but existing approaches either edit with no account of what knowledge changed or regenerate from scratch. A KPR integrates new knowledge into a document by extracting claims, filtering and routing them to sections, and flagging conflicts with existing content…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "d7756752929bd14aaccb",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26634",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Knowledge Pull Requests for Continual Document Authoring",
            "item_type": "record",
            "summary": "added: Knowledge Pull Requests for Continual Document Authoring",
            "after": {
              "title": "Knowledge Pull Requests for Continual Document Authoring",
              "summary": "We introduce Knowledge Pull Requests (KPRs), a framework for continual document authoring that makes each change interpretable. Documents require ongoing revision as new knowledge surfaces from other sources, languages, or times, but existing approaches either edit with no account of what knowledge changed or regenerate from scratch. A KPR integrates new knowledge into a document by extracting claims, filtering and routing them to sections, and flagging conflicts with existing content, producing a ChangeLog that separates what knowledge changes (claim proposal) from how the text changes (document diff). We evaluate KPRs on revising Wikipedia across languages and updating query-driven reports on RAGTIME. KPRs integrate more information and better preserve existing content than rewriting from sources or regenerating from scratch, while adding the most information per token generated. A KPR-revised article also grounds question answering better than a frontier model with search, which does not surface knowledge documented only in other languages.",
              "link": "https://huggingface.co/papers/2609.26634",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26634"
    },
    {
      "id": "043536f842069c95f13d",
      "title": "Capable yet Parsimonious: Extracting and Characterizing Hidden Chain-of-Thought in Frontier Models",
      "content_text": "The rapid capability gains of frontier language models are widely attributed to improved reasoning abilities, yet this cannot be verified as raw CoT traces in closed-source systems are hidden. By registering a simple custom tool through a standard API feature, we induce frontier models to externalize intermediate reasoning. Because these traces may reflect post-hoc rationalization rather than genuine reasoning, we first evaluate against native CoT on open-source models and extend to…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "043536f842069c95f13d",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26637",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Capable yet Parsimonious: Extracting and Characterizing Hidden Chain-of-Thought in Frontier Models",
            "item_type": "record",
            "summary": "added: Capable yet Parsimonious: Extracting and Characterizing Hidden Chain-of-Thought in Frontier Models",
            "after": {
              "title": "Capable yet Parsimonious: Extracting and Characterizing Hidden Chain-of-Thought in Frontier Models",
              "summary": "The rapid capability gains of frontier language models are widely attributed to improved reasoning abilities, yet this cannot be verified as raw CoT traces in closed-source systems are hidden. By registering a simple custom tool through a standard API feature, we induce frontier models to externalize intermediate reasoning. Because these traces may reflect post-hoc rationalization rather than genuine reasoning, we first evaluate against native CoT on open-source models and extend to closed-source frontier models including GPT-6 Astra. We find that the extracted reasoning matches native reasoning performance and substantially outperforms no-reasoning baselines, across competition mathematics, science, and code generation. We then characterize how frontier models structure their intermediate reasoning. Across token efficiency, reasoning-step types, and induced reasoning trees, we identify systematic differences in how models externalize, compress, and organize reasoning. We find that Astra exhibits token-efficient directed reasoning, selecting a correct trajectory earlier, while resolving elementary steps internally and externalizing only crucial reasoning. These findings provide a behavioral lens on frontier-model reasoning beyond benchmark scores.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/67ebb002e8fc59ccb90d3f9e/gPNf9xk5q7c7PhPvrp6QO.png"
              ],
              "organization": {
                "_id": "68d66a2ce690f3f54676813f",
                "name": "SeaFill2025",
                "fullname": "Sea-Fill",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68d669e121785bf79dec4f7a/HN8cIsNLsI8vrUUvv4sva.png"
              },
              "link": "https://huggingface.co/papers/2609.26637",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26637"
    },
    {
      "id": "f78786af8835752c4e85",
      "title": "X-Planner: Event-Structured Task Planning for Embodied Intelligence",
      "content_text": "Task planning bridges high-level instructions and executable behavior in long-horizon manipulation, yet modern Vision-Language-Action (VLA) systems often leave this intermediate structure implicit. Existing chain-of-thought (CoT) planners also tend to rely on coarse task-level annotations or serialize long reasoning traces token by token. We present X-Planner, a planning front-end that addresses both the supervision and representation of embodied reasoning. Our planning data combine Ego, UMI…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "f78786af8835752c4e85",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25187",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "X-Planner: Event-Structured Task Planning for Embodied Intelligence",
            "item_type": "record",
            "summary": "added: X-Planner: Event-Structured Task Planning for Embodied Intelligence",
            "after": {
              "title": "X-Planner: Event-Structured Task Planning for Embodied Intelligence",
              "summary": "Task planning bridges high-level instructions and executable behavior in long-horizon manipulation, yet modern Vision-Language-Action (VLA) systems often leave this intermediate structure implicit. Existing chain-of-thought (CoT) planners also tend to rely on coarse task-level annotations or serialize long reasoning traces token by token. We present X-Planner, a planning front-end that addresses both the supervision and representation of embodied reasoning. Our planning data combine Ego, UMI, and teleoperation under a hierarchy granularity with source-dependent annotation depth. Takeover-time annotations and human-designed failures supervise ongoing error recognition. On the model side, a shared VLM backbone exposes two event-structured plan forms: a discrete interface that emits interpretable event states and a latent interface that relays continuous CoT states across staggered Transformer depths through Staircase Decoding. A frozen latent-to-text reconstruction objective provides a semantic anchor for the latent representation. Offline two-step planning evaluation places X-Planner second among four evaluated models on both BERTScore-F1 and a judge-based Overall score. In real-robot experiments, respectively, outperforming the evaluated baselines. These results characterize planning-text quality and downstream execution.",
              "link": "https://huggingface.co/papers/2609.25187",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25187"
    },
    {
      "id": "1f99a0667287947057de",
      "title": "Calibration as a First-Class Criterion in LLM Evaluation",
      "content_text": "Calibration of language models -- the alignment between expressed or implicit confidence and empirical correctness -- is a well-studied subfield within NLP. Methods to measure it already exist. The problem is adoption: outside this subfield, NLP research regularly introduces new models, datasets, and benchmarks without checking whether the model's confidence scores are meaningful. We argue that this adoption gap is a major obstacle to trustworthy LLM evaluation. Miscalibration causes problems…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "1f99a0667287947057de",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26489",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Calibration as a First-Class Criterion in LLM Evaluation",
            "item_type": "record",
            "summary": "added: Calibration as a First-Class Criterion in LLM Evaluation",
            "after": {
              "title": "Calibration as a First-Class Criterion in LLM Evaluation",
              "summary": "Calibration of language models -- the alignment between expressed or implicit confidence and empirical correctness -- is a well-studied subfield within NLP. Methods to measure it already exist. The problem is adoption: outside this subfield, NLP research regularly introduces new models, datasets, and benchmarks without checking whether the model's confidence scores are meaningful. We argue that this adoption gap is a major obstacle to trustworthy LLM evaluation. Miscalibration causes problems in two distinct areas: at deployment, where overconfident mistakes cause real harm, and inside the research pipeline, where methods like LLM-as-a-judge, synthetic data generation, and active learning rely on calibrated confidence without verifying it. Standard calibration metrics only require two inputs per example: a confidence score and a correctness judgment. Most benchmarks in use today already provide both, meaning calibration can be reported immediately. For open-ended generation, however, defining these two inputs is still an open challenge. We argue that each NLP subfield should pair its main performance metric with a calibration score and call for treating calibration as an essential property of every model rather than a niche topic.",
              "link": "https://huggingface.co/papers/2609.26489",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26489"
    },
    {
      "id": "5c4ac554046230816228",
      "title": "FLEET: From Logits Entropy to Enhanced Trajectories in Text Generation",
      "content_text": "Solutions based on large language models (LLMs) often rely on temperature sampling to improve accuracy and stability by aggregating multiple samples from the completion distribution. However, this memoryless approach is inherently suboptimal: because it lacks awareness of prior generations and their evaluations, it produces an increasing proportion of semantically duplicate answers as more samples are drawn, leading to diminishing returns. To address this limitation, we introduce FLEET, a novel…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "5c4ac554046230816228",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27657",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "FLEET: From Logits Entropy to Enhanced Trajectories in Text Generation",
            "item_type": "record",
            "summary": "added: FLEET: From Logits Entropy to Enhanced Trajectories in Text Generation",
            "after": {
              "title": "FLEET: From Logits Entropy to Enhanced Trajectories in Text Generation",
              "summary": "Solutions based on large language models (LLMs) often rely on temperature sampling to improve accuracy and stability by aggregating multiple samples from the completion distribution. However, this memoryless approach is inherently suboptimal: because it lacks awareness of prior generations and their evaluations, it produces an increasing proportion of semantically duplicate answers as more samples are drawn, leading to diminishing returns. To address this limitation, we introduce FLEET, a novel method that integrates a memory mechanism into the generation process. FLEET represents each generation as a sparse trajectory through states whose entropy exceeds a predefined threshold and uses these trajectories to infer per-token utility scores that adjust the logits. Benchmark evaluations demonstrate that FLEET achieves the same accuracy as the repeated sampling baseline, with a 3x speedup, and substantially improves accuracy on complex coding tasks (LiveCodeBench Pass@32 increases from 59.9% to 66.2%) under the same budget. Furthermore, in the greedy-decoding configuration evaluated here, the approach is deterministic and uses a single calibration pass to derive its principal hyperparameters, requiring only minimal modifications to existing LLM pipelines.",
              "link": "https://huggingface.co/papers/2609.27657",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27657"
    },
    {
      "id": "1698b009adb119bafbba",
      "title": "GeoPair: Geometry-Preserving Cross-Layer Factorization for Training-Free Transformer Compression",
      "content_text": "Transformer architectures exhibit cross-layer redundancies, yet post-training compression pipelines typically optimize layers in isolation or rely on heuristic grouping strategies that disregard layer-specific activation geometries. We introduce a principled, training-free framework that sequentially optimizes cross-layer weight pairings and shared-dictionary factorizations. Rather than forcing weights of adjacent layers to share a basis or heuristically merging activation statistics, our…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "1698b009adb119bafbba",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25963",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "GeoPair: Geometry-Preserving Cross-Layer Factorization for Training-Free Transformer Compression",
            "item_type": "record",
            "summary": "added: GeoPair: Geometry-Preserving Cross-Layer Factorization for Training-Free Transformer Compression",
            "after": {
              "title": "GeoPair: Geometry-Preserving Cross-Layer Factorization for Training-Free Transformer Compression",
              "summary": "Transformer architectures exhibit cross-layer redundancies, yet post-training compression pipelines typically optimize layers in isolation or rely on heuristic grouping strategies that disregard layer-specific activation geometries. We introduce a principled, training-free framework that sequentially optimizes cross-layer weight pairings and shared-dictionary factorizations. Rather than forcing weights of adjacent layers to share a basis or heuristically merging activation statistics, our approach identifies structurally compatible projections and learns a shared representation that better preserves each layer's distinct calibration geometry. Coupled with structured sparsity, this yields highly efficient weight decompositions without sacrificing functional fidelity. Across diverse architectures, scales, and modalities, our method achieves state-of-the-art results, consistently outperforming independent structured weight decompositions and alternative pairwise weight factorizations, which operate under heuristic grouping strategies. By replacing heuristic engineering strategies with a convergent, optimization-driven pipeline, we establish a theoretically grounded foundation for scalable, transformer compression across different modalities.",
              "organization": {
                "_id": "65f1bb3789aedc3dbe201d53",
                "name": "MTSAIR",
                "fullname": "MWS AI",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6654eef8b88e4539b2ff184b/gRBr4-h25HQ243mwdmnC0.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.25963",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 10
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25963"
    },
    {
      "id": "bd175f155878e2333a23",
      "title": "Six Layers Less: Encoder Pruning for Whisper with Label-Free Recovery",
      "content_text": "Pruning large pre-trained transformer-based ASR models such as OpenAI's Whisper has seen great adoption, as pruning the decoder led to significant end-to-end transcription speedups. For instance, the {\\tt whisper-large-v3-turbo} variant reduced the decoder from 32 to 4 layers, while Distill-Whisper similarly reduced the decoder to only 2 layers. Although some attention has been put towards reducing the size of the encoder, no approach has seen wide adoption. This could be due to the need for…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "bd175f155878e2333a23",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27980",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Six Layers Less: Encoder Pruning for Whisper with Label-Free Recovery",
            "item_type": "record",
            "summary": "added: Six Layers Less: Encoder Pruning for Whisper with Label-Free Recovery",
            "after": {
              "title": "Six Layers Less: Encoder Pruning for Whisper with Label-Free Recovery",
              "summary": "Pruning large pre-trained transformer-based ASR models such as OpenAI's Whisper has seen great adoption, as pruning the decoder led to significant end-to-end transcription speedups. For instance, the {\\tt whisper-large-v3-turbo} variant reduced the decoder from 32 to 4 layers, while Distill-Whisper similarly reduced the decoder to only 2 layers. Although some attention has been put towards reducing the size of the encoder, no approach has seen wide adoption. This could be due to the need for custom inference implementations to take advantage of the compressed model. We present an approach that ranks encoder layers by the leave-one-layer-out change in Word Error Rate (WER). The six layers that cause the least change are removed, corresponding to 18.5% of the encoder stack. The pruned model requires no custom inference code as it is simply a more shallow encoder with fewer layers. We further distill using unlabeled monolingual speech data to recover performance degradation caused by the zero-shot layer pruning. Mean WER across four languages increases to 20.1% after distillation, compared to 21.9% zero-shot, going from a baseline of 18.2%. We release all of our code (https://github.com/rasgaard/whisper-encoder-layer-prune) and the pruned model (https://huggingface.co/rasgaard/whisper-large-v3-turbo-encoder-pruned).",
              "link": "https://huggingface.co/papers/2609.27980",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27980"
    },
    {
      "id": "84bdd4d693306b2af812",
      "title": "Uranus: Building the Next-Generation Simulation Infrastructure for Embodied AI",
      "content_text": "Scalable simulation is essential for robot data generation, policy training, evaluation, and safe iteration, yet real-world interaction is costly and conventional simulators require labor-intensive construction. We present Uranus, a data-driven robot simulator built around a joint-trajectory-conditioned autoregressive diffusion model. Uranus offers three key capabilities: (1) streaming, open-ended rollout, which receives future joint-position trajectories online and autoregressively generates…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "84bdd4d693306b2af812",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24815",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Uranus: Building the Next-Generation Simulation Infrastructure for Embodied AI",
            "item_type": "record",
            "summary": "added: Uranus: Building the Next-Generation Simulation Infrastructure for Embodied AI",
            "after": {
              "title": "Uranus: Building the Next-Generation Simulation Infrastructure for Embodied AI",
              "summary": "Scalable simulation is essential for robot data generation, policy training, evaluation, and safe iteration, yet real-world interaction is costly and conventional simulators require labor-intensive construction. We present Uranus, a data-driven robot simulator built around a joint-trajectory-conditioned autoregressive diffusion model. Uranus offers three key capabilities: (1) streaming, open-ended rollout, which receives future joint-position trajectories online and autoregressively generates one latent frame per step, corresponding to four RGB frames, without a fixed horizon; (2) low-latency generation, achieving 24 FPS after inference optimization; and (3) scalable, extensible robot control, providing a unified interface for synchronized multi-view generation across diverse robot embodiments and camera configurations. We conduct comprehensive quantitative and qualitative evaluations on both in-distribution and out-of-distribution data, providing an objective assessment of Uranus and clearly identifying its current limitations. We release the code and model weights to empower the community with practical tools and insights.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/63ecdc6b54bba6781518cf48/pJXpMCsz6fGaVRX5ua3EU.png"
              ],
              "organization": {
                "_id": "67629732299173549935e729",
                "name": "D-Robotics",
                "fullname": "D-Robotics",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/67628d4278b836c285f400b9/tUv2H4GDXACZvkbs6lwDe.png"
              },
              "link": "https://huggingface.co/papers/2609.24815",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24815"
    },
    {
      "id": "bb05f9d05da0fb71d557",
      "title": "MemoryAthena: Adaptive Routing over Latent and Generated Memories",
      "content_text": "Learned-memory methods store information in an explicit table and consume it through a separate reader, allowing addressing, storage, and reading to be modified independently. We study whether useful memory can also be generated rather than only retrieved. MemoryAthena uses three pathways: direct Engram retrieval (E), generation from retrieved Engram cues (GE), and generation from causal backbone states without consulting the memory table (GH). Generated memory is conditionally useful: it can…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "bb05f9d05da0fb71d557",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25853",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "MemoryAthena: Adaptive Routing over Latent and Generated Memories",
            "item_type": "record",
            "summary": "added: MemoryAthena: Adaptive Routing over Latent and Generated Memories",
            "after": {
              "title": "MemoryAthena: Adaptive Routing over Latent and Generated Memories",
              "summary": "Learned-memory methods store information in an explicit table and consume it through a separate reader, allowing addressing, storage, and reading to be modified independently. We study whether useful memory can also be generated rather than only retrieved. MemoryAthena uses three pathways: direct Engram retrieval (E), generation from retrieved Engram cues (GE), and generation from causal backbone states without consulting the memory table (GH). Generated memory is conditionally useful: it can complement E in one context but interfere with it in another. MemoryAthena therefore treats E as an anchor and learns when a generated representation should intervene. With the backbone, memory, generators, and readers frozen, a lightweight causal routing head is trained from counterfactual future-token likelihood advantages of GE and GH relative to E. At inference time, an admitted candidate modifies the E residual through bounded interpolation, while rejection recovers the direct pathway exactly. On question answering, MemoryAthena raises the five-task average from 37.65 to 39.28 over the direct pathway of the same checkpoint, while the six-task general-NLP average increases from 76.73 to 79.13. The complete memory-side system contains approximately 201M parameters, excluding the frozen backbone. Further analyses show complementary strengths among E, GE, and GH across tasks and inputs. These results support generated memory as a selective correction to direct retrieval and highlight routing when, which, and how strongly to intervene as the central challenge.",
              "organization": {
                "_id": "6a1ffda6310029dbc8ac9adc",
                "name": "OLAResearchX",
                "fullname": "Omni Language AI Research",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/617a92e16f37340367d5d791/dMU_WZq7clwnvUTMragYe.png"
              },
              "link": "https://huggingface.co/papers/2609.25853",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25853"
    },
    {
      "id": "ef94e3276a3d64fb3707",
      "title": "On the Diffusibility of High-Dimensional Latents",
      "content_text": "Representation Autoencoders (RAEs) enable diffusion models to operate in the feature spaces of pretrained visual encoders. However, many off-the-shelf encoders are not optimized for faithful reconstruction, discarding fine-grained visual details. As expected, finetuning these encoders for image reconstruction recovers such details. However, perhaps counterintuitively, this procedure reduces the effective dimensionality of the resulting representation, and the altered geometry has downstream…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "ef94e3276a3d64fb3707",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28473",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "On the Diffusibility of High-Dimensional Latents",
            "item_type": "record",
            "summary": "added: On the Diffusibility of High-Dimensional Latents",
            "after": {
              "title": "On the Diffusibility of High-Dimensional Latents",
              "summary": "Representation Autoencoders (RAEs) enable diffusion models to operate in the feature spaces of pretrained visual encoders. However, many off-the-shelf encoders are not optimized for faithful reconstruction, discarding fine-grained visual details. As expected, finetuning these encoders for image reconstruction recovers such details. However, perhaps counterintuitively, this procedure reduces the effective dimensionality of the resulting representation, and the altered geometry has downstream effects on generation. Specifically, we show that using the standard velocity prediction in flow matching in this high-dimensional space requires the model to fit orthogonal noise directions outside the low-dimensional signal manifold, making optimization inefficient. This motivates using the clean data parameterization (x_{0}-prediction) instead, which focuses learning on the underlying signal manifold. Across experiments with multiple strong-reconstruction encoders, we show that x_{0}-prediction consistently improves text-to-image generation performance.",
              "link": "https://huggingface.co/papers/2609.28473",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28473"
    },
    {
      "id": "06ab31063f474ac7e54c",
      "title": "All modalities are equal, but video is more equal: Closing the Cross-Attention Gap in Joint Video Generation",
      "content_text": "Video is a rich representation of a physical event, capturing appearance, geometry, motion, and temporal evolution. Other modalities, such as 3D body motion or audio, encode narrower aspects of the same event. We find that joint multimodal diffusion transformers exhibit a corresponding asymmetry in cross-modal correspondence: companion modalities develop strong correspondences to video, but the reciprocal correspondences through which they constrain video remain substantially weaker. We express…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "06ab31063f474ac7e54c",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27901",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "All modalities are equal, but video is more equal: Closing the Cross-Attention Gap in Joint Video Generation",
            "item_type": "record",
            "summary": "added: All modalities are equal, but video is more equal: Closing the Cross-Attention Gap in Joint Video Generation",
            "after": {
              "title": "All modalities are equal, but video is more equal: Closing the Cross-Attention Gap in Joint Video Generation",
              "summary": "Video is a rich representation of a physical event, capturing appearance, geometry, motion, and temporal evolution. Other modalities, such as 3D body motion or audio, encode narrower aspects of the same event. We find that joint multimodal diffusion transformers exhibit a corresponding asymmetry in cross-modal correspondence: companion modalities develop strong correspondences to video, but the reciprocal correspondences through which they constrain video remain substantially weaker. We express both directions as comparable correspondence distributions over video tokens and define their disagreement as the reciprocal correspondence gap. We introduce RecCAR, standing for Reciprocal Cross-modal Attention Regularization, a KL regularizer that uses the well-established video-to-modality correspondence as a fixed reference and aligns the weaker modality-to-video correspondence toward it. Across joint video-motion and video-audio generation, RecCAR improves the Human Anatomy score from 0.69 to 0.75 and reduces audio-video desynchronization from 0.804 to 0.752, while improving overall generation",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/63c59c3a6d132b995fedface/QQbsfQH8qlII25QYzDOq3.mp4"
              ],
              "organization": {
                "_id": "634a981ccf84e1cc3f610ca1",
                "name": "barilan",
                "fullname": "Bar-Ilan University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/noauth/MRvtAhVBzuzijH-vDGBjr.png"
              },
              "link": "https://huggingface.co/papers/2609.27901",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 5
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27901"
    },
    {
      "id": "bb9b5e98593ccea3404e",
      "title": "PackLab: A Comprehensive Framework for Developing, Training, and Evaluating MLLMs in Robotic Bin Packing",
      "content_text": "Robotic bin packing requires long-horizon sequential decision-making, as each object placement affects the available space for subsequent packing. Existing methods primarily rely on hand-crafted geometric heuristics that optimize predefined objectives or reinforcement learning policies learned through trial and error over predefined training configurations. Despite recent advances in multimodal large language models (MLLMs) for this task, their potential for closed-loop sequential decisions…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "bb9b5e98593ccea3404e",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.23784",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "PackLab: A Comprehensive Framework for Developing, Training, and Evaluating MLLMs in Robotic Bin Packing",
            "item_type": "record",
            "summary": "added: PackLab: A Comprehensive Framework for Developing, Training, and Evaluating MLLMs in Robotic Bin Packing",
            "after": {
              "title": "PackLab: A Comprehensive Framework for Developing, Training, and Evaluating MLLMs in Robotic Bin Packing",
              "summary": "Robotic bin packing requires long-horizon sequential decision-making, as each object placement affects the available space for subsequent packing. Existing methods primarily rely on hand-crafted geometric heuristics that optimize predefined objectives or reinforcement learning policies learned through trial and error over predefined training configurations. Despite recent advances in multimodal large language models (MLLMs) for this task, their potential for closed-loop sequential decisions across heterogeneous packing configurations remains underexplored. To address this gap, we introduce PackLab, a comprehensive framework for developing, training, and evaluating MLLMs for closed-loop robotic bin packing. PackLab-Suite provides a physics-based simulation platform for scalable generation of diverse training packing trajectories and evaluation of their physical outcomes. PackLab-VLM is a packing-specialized MLLM that understands the evolving object and container states to jointly select objects and predict placements in a closed-loop manner. PackLab-Bench provides standardized packing scenarios at multiple difficulty levels for systematic evaluation. Extensive experiments demonstrate that, on average, PackLab-VLM outperforms conventional packing heuristics, traditional reinforcement learning methods, and general-purpose MLLMs across object sets and container configurations, highlighting the potential of MLLMs for long-horizon robotic packing. The code, model, dataset, and benchmark are available at https://github.com/Correr-Zhou/PackLab .",
              "organization": {
                "_id": "6390c6fdd00f25601f445cd4",
                "name": "CUHK-CSE",
                "fullname": "The Chinese University of Hong Kong",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/621f2eb36e152b56a7cf0248/o8RRAczRjfNEzq70GzUwQ.png"
              },
              "link": "https://huggingface.co/papers/2609.23784",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 10
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.23784"
    },
    {
      "id": "6b5bbdb16532b2c4d826",
      "title": "Spatial-Interactor: Learning Spatial Reasoning through Interaction with the Observable Physical World",
      "content_text": "Spatial reasoning is essential for vision-language models (VLMs) to understand and act in the physical world. Reasoning in dynamic environments requires VLMs to perceive local state transitions caused by object motion and viewpoint changes and integrate them over long trajectories to maintain an updated spatial state, yet existing VLMs remain limited in both capabilities. Current spatial training primarily focuses on static questions about object attributes and spatial relations, providing…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "6b5bbdb16532b2c4d826",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.23038",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Spatial-Interactor: Learning Spatial Reasoning through Interaction with the Observable Physical World",
            "item_type": "record",
            "summary": "added: Spatial-Interactor: Learning Spatial Reasoning through Interaction with the Observable Physical World",
            "after": {
              "title": "Spatial-Interactor: Learning Spatial Reasoning through Interaction with the Observable Physical World",
              "summary": "Spatial reasoning is essential for vision-language models (VLMs) to understand and act in the physical world. Reasoning in dynamic environments requires VLMs to perceive local state transitions caused by object motion and viewpoint changes and integrate them over long trajectories to maintain an updated spatial state, yet existing VLMs remain limited in both capabilities. Current spatial training primarily focuses on static questions about object attributes and spatial relations, providing limited direct supervision for state transitions; in contrast, interaction trajectories naturally connect a preceding observation, an action, and a subsequent observation, offering direct supervision for local state transitions, while complete trajectories reveal dependencies among consecutive transitions. We therefore introduce Spatial-Interactor, a framework that trains VLMs to model physical-world state transitions through interaction, organizing this learning process into a three-level curriculum covering L1 passive world-state transitions, L2 active self-state transitions, and L3 long-horizon interaction trajectories. Accordingly, we construct the Learning from Spatial Interaction dataset (LSI-108K) from simulated and real interaction trajectories, with tasks aligned with the objective of each level. Our two-stage training strategy applies Supervised Fine-Tuning (SFT) to L1 and L2 for local transition modeling, and On-Policy Distillation (OPD) then uses privileged self-distillation: a teacher branch given segment-level transition descriptions supervises the student's on-policy CoT, helping the student learn to integrate consecutive transitions over L3 long trajectories. Experiments across multiple VLMs and spatial benchmarks show consistent gains in local transition modeling and long-horizon integration.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/6485bd278d14bcd5cdbb7c8d/K30KQ07efNPAFm5NdC0Zo.mp4",
                "https://cdn-uploads.huggingface.co/production/uploads/6485bd278d14bcd5cdbb7c8d/XK9xxF0BXC3zbdDSO8OOs.mp4"
              ],
              "organization": {
                "_id": "696461ab2d94e9a07cdb8efd",
                "name": "OmniAI-ZJU",
                "fullname": "ZJU-OmniAI",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/65f2595830354d0ee043b25a/eEeRdHlGyJ148JQ6BAJ4O.png"
              },
              "link": "https://huggingface.co/papers/2609.23038",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 43
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.23038"
    },
    {
      "id": "6df826a50581a19e0556",
      "title": "HappyWorld-Bench",
      "content_text": "Evaluating world models requires assessing both the quality of the worlds they generate and their consistency and responsiveness under exploration, interaction, and modification. We introduce HappyWorld-Bench, a comprehensive benchmark that evaluates whether generated worlds remain reliable as agents interact with them. Our design is built on a hierarchical capability framework of six world capabilities (W1-W6), from generative construction to unified world modeling, instantiated across three…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "6df826a50581a19e0556",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24308",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "HappyWorld-Bench",
            "item_type": "record",
            "summary": "added: HappyWorld-Bench",
            "after": {
              "title": "HappyWorld-Bench",
              "summary": "Evaluating world models requires assessing both the quality of the worlds they generate and their consistency and responsiveness under exploration, interaction, and modification. We introduce HappyWorld-Bench, a comprehensive benchmark that evaluates whether generated worlds remain reliable as agents interact with them. Our design is built on a hierarchical capability framework of six world capabilities (W1-W6), from generative construction to unified world modeling, instantiated across three independent evaluation tracks: video world models, spatial world models, and embodied world models. HappyWorld-Bench comprises 1,138 video prompts, 300 spatial scenes, and 254 embodied test cases. Across all three tracks, we build and operate HappyWorld-Arena to organize human A/B comparisons and derive model-level Elo ratings, which complement newly designed automated metrics that capture behavioral correctness. We evaluate 14 video world models, 9 spatial systems, and 8 embodied candidates under this unified framework. Results reveal remaining reliability gaps across all three tracks: video models exhibit reduced consistency during extended rollouts and revisits, spatial models achieve at best 70.14% placement accuracy and 73.33% edit execution, and embodied models struggle to preserve state across multi-step actions and respond precisely to altered action conditions and physical rules. These findings highlight the need to evaluate world models not only by visual quality, but also by state consistency and the correctness of their responses to actions and interventions.",
              "organization": {
                "_id": "68be41370a3fcebdcad6516a",
                "name": "alibabagroup",
                "fullname": "alibaba",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68be3ab7e52df040b2cf80dc/li4G29u_EGswyTN1Sm_Kq.png"
              },
              "link": "https://huggingface.co/papers/2609.24308",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 39
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24308"
    },
    {
      "id": "b7f0854233653d140052",
      "title": "Just-in-Time Memory: Learning to Curate Task-Adaptive Memory for LLM Agents",
      "content_text": "Agentic memory systems reuse past experience to improve future performance, yet most existing designs curate memory at write time: once a task is completed, its trajectory is distilled into a fixed artifact, such as a reflection, workflow, skill, or reasoning strategy, that is later retrieved by similarity. This forces the system to decide what is worth remembering before the future query is known, irreversibly discarding information and producing a query-independent summary that must serve…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "b7f0854233653d140052",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27334",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Just-in-Time Memory: Learning to Curate Task-Adaptive Memory for LLM Agents",
            "item_type": "record",
            "summary": "added: Just-in-Time Memory: Learning to Curate Task-Adaptive Memory for LLM Agents",
            "after": {
              "title": "Just-in-Time Memory: Learning to Curate Task-Adaptive Memory for LLM Agents",
              "summary": "Agentic memory systems reuse past experience to improve future performance, yet most existing designs curate memory at write time: once a task is completed, its trajectory is distilled into a fixed artifact, such as a reflection, workflow, skill, or reasoning strategy, that is later retrieved by similarity. This forces the system to decide what is worth remembering before the future query is known, irreversibly discarding information and producing a query-independent summary that must serve many possible downstream tasks. Learning such a write-time curator is also difficult because the value of a storage decision may only become apparent when a relevant query arrives, potentially many tasks later, creating a long-horizon credit-assignment problem. We instead retain raw trajectories and defer curation until read time, when the current task is known. Given the retrieved traces and the new task, a memory curator synthesizes a compact, task-adaptive payload tailored to the immediate need. Because this payload is consumed on the same task, the curator can be trained directly from immediate task success, avoiding delayed utility signals and the need to artificially group related tasks. Across ALFWorld, WebShop, and τ^2-bench, our Just-in-Time Memory (JitMem) consistently outperforms no-memory agents as well as heuristic and learned write-time memory methods, improving over the strongest baseline by 16.2, 16.3, and 3.9 absolute success-rate points, respectively. Notably, even an untrained curator is already competitive with or surpasses these baselines, showing that task-adaptive read-time curation itself is a major source of the gain; training the curator further compounds the improvement.",
              "organization": {
                "_id": "5f6d64475e78cc6b0ed31e4c",
                "name": "Salesforce",
                "fullname": "Salesforce AI Research",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/1602756670970-noauth.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.27334",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 28
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27334"
    },
    {
      "id": "c3cfec2033f7940f14a8",
      "title": "EmbodiedSWE: Coding Agents for Long Horizon Dexterous Robotics",
      "content_text": "We study coding agents for long-horizon, dexterous robotics and ask whether their solutions can provide scalable supervision for learning general robot policies. To test this, we develop EMBODIEDSWE-BENCH, a simulation benchmark for coding agents spanning contact-rich manipulation, deformable objects, and long-horizon tasks requiring up to half an hour of continuous interaction. We find that frontier coding agents can solve complex long-horizon tasks and transfer prior solutions across both…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "c3cfec2033f7940f14a8",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27308",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "EmbodiedSWE: Coding Agents for Long Horizon Dexterous Robotics",
            "item_type": "record",
            "summary": "added: EmbodiedSWE: Coding Agents for Long Horizon Dexterous Robotics",
            "after": {
              "title": "EmbodiedSWE: Coding Agents for Long Horizon Dexterous Robotics",
              "summary": "We study coding agents for long-horizon, dexterous robotics and ask whether their solutions can provide scalable supervision for learning general robot policies. To test this, we develop EMBODIEDSWE-BENCH, a simulation benchmark for coding agents spanning contact-rich manipulation, deformable objects, and long-horizon tasks requiring up to half an hour of continuous interaction. We find that frontier coding agents can solve complex long-horizon tasks and transfer prior solutions across both tasks and embodiments. We also design supporting tools that help agents more effectively solve these tasks. However, the resulting solutions require substantial iterative interaction and are typically specialized to individual task instances. We therefore introduce EMBODIEDSWE-GEN, which expands a single solution from coding agent into large diverse trajectories for training a VLA. VLA performance improves with more generated demonstrations, and agent-aided diversification improves generalization to held-out task variations. We also show that a VLA finetuned solely on coding-agent-generated simulation demonstrations completes a long-horizon task on real robot. Together, our framework uses coding agents to solve complex robotics tasks and turn verified solutions into scalable supervision for robot policies.",
              "link": "https://huggingface.co/papers/2609.27308",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27308"
    },
    {
      "id": "092dff9bc75a969d655b",
      "title": "StudentBench: AI and human tutoring yield equivalent GRE learning gains",
      "content_text": "Artificial intelligence offers an unprecedented opportunity to augment human capabilities, yet progress at the frontier has focused primarily on advancing model capabilities. We introduce StudentBench, a suite of AI teaching evaluations and a public platform that enables large-scale data collection with over 175,000 student-AI messages to study whether large language models (LLMs) produce learning gains equivalent to human tutoring. Using StudentBench, we measured learning gains on Quantitative…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "092dff9bc75a969d655b",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28470",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "StudentBench: AI and human tutoring yield equivalent GRE learning gains",
            "item_type": "record",
            "summary": "added: StudentBench: AI and human tutoring yield equivalent GRE learning gains",
            "after": {
              "title": "StudentBench: AI and human tutoring yield equivalent GRE learning gains",
              "summary": "Artificial intelligence offers an unprecedented opportunity to augment human capabilities, yet progress at the frontier has focused primarily on advancing model capabilities. We introduce StudentBench, a suite of AI teaching evaluations and a public platform that enables large-scale data collection with over 175,000 student-AI messages to study whether large language models (LLMs) produce learning gains equivalent to human tutoring. Using StudentBench, we measured learning gains on Quantitative and Verbal GRE questions across 2,383 human participants receiving AI tutoring, human tutoring, or no tutoring. We establish that AI tutoring is statistically equivalent to expert human tutoring for GRE learning gains (p = .015), and in five of the seven GRE domains, the best performing AI tutor surpassed the human tutor, on average. In a second study, expert human tutors compared LLM-generated lesson plans and practice problems through 2,028 pairwise rubric evaluations. Together, the two studies clearly separate AI tutors across: (1) lesson planning, (2) practice-problem creation, (3) conversational pedagogy, (4) cost, and (5) engagement. Surprisingly, one AI tutor achieved learning gains equivalent to human tutoring (p = .044) at 918 times lower cost (USD 0.0052 for AI versus USD 4.81 for human, per percentage point gained). For Quantitative GRE sessions, faster AI replies correlated with more student messages, more messages with more correct practice, and more correct practice with larger learning gains (all p < .002). The StudentBench platform is freely available at https://studentbench.org.",
              "organization": {
                "_id": "66300b42246247eeade69e23",
                "name": "handshake-ai-research",
                "fullname": "Handshake AI Research",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/662ff41ec7edc3b9628a91fd/KfxlDM_3b7qWH7g1izlml.png"
              },
              "link": "https://huggingface.co/papers/2609.28470",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 1
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28470"
    },
    {
      "id": "44c14339ff7ca266fafc",
      "title": "MemBodied: Recurrent Associative Memory for Vision-Language-Action Models",
      "content_text": "Vision-Language-Action models provide a strong foundation for general-purpose robot control, yet a vast majority of policies do not preserve and leverage episode-level information beyond the current observation. This limitation is consequential in history-dependent manipulation tasks that depend on information available only in past observations. Retaining past observations in context can aid in recovering this information, but at the significant cost of ever-growing, bloated context and…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "44c14339ff7ca266fafc",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28256",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "MemBodied: Recurrent Associative Memory for Vision-Language-Action Models",
            "item_type": "record",
            "summary": "added: MemBodied: Recurrent Associative Memory for Vision-Language-Action Models",
            "after": {
              "title": "MemBodied: Recurrent Associative Memory for Vision-Language-Action Models",
              "summary": "Vision-Language-Action models provide a strong foundation for general-purpose robot control, yet a vast majority of policies do not preserve and leverage episode-level information beyond the current observation. This limitation is consequential in history-dependent manipulation tasks that depend on information available only in past observations. Retaining past observations in context can aid in recovering this information, but at the significant cost of ever-growing, bloated context and inference latency. We thus introduce MemBodied, a fixed-size episodic memory with two complementary components: an associative state that records interactions across policy calls and an episode anchor that preserves a compact representation of the initial scene as a reference. At each policy call, the model conditions action generation on the current input and the memory components, rather than directly using past observations. Across five evaluated RMBench tasks requiring memory, MemBodied achieves 7.81times the mean success rate of a stateless policy and 2.98times of vanilla recurrent memory, while outperforming the strongest memory-augmented baseline by 1.3times with 10times fewer added parameters. On the fully observable LIBERO-Long suite, it reached 90.6%, a 5.4% improvement over the stateless π_0 policy. These findings support MemBodied as a practical alternative to expanding the policy context for history-dependent manipulation.",
              "link": "https://huggingface.co/papers/2609.28256",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 9
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28256"
    },
    {
      "id": "abc305bb68969594dfee",
      "title": "SpeakerMem-R1: Speaker-Centered Dual-Track Memory for Multi-Party Dialogue",
      "content_text": "Long-term conversational memory in multi-party settings requires more than retrieving relevant content from long-term conversations: it must distinguish who said what, whom each statement concerns, how individuals perceive one another, what information is shared by the group, and how states change over time. Recent studies on multi-party dialogue benchmarks show that existing general-purpose LLM memory systems tend to lose person and group relations or struggle to integrate clues distributed…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "abc305bb68969594dfee",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26780",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "SpeakerMem-R1: Speaker-Centered Dual-Track Memory for Multi-Party Dialogue",
            "item_type": "record",
            "summary": "added: SpeakerMem-R1: Speaker-Centered Dual-Track Memory for Multi-Party Dialogue",
            "after": {
              "title": "SpeakerMem-R1: Speaker-Centered Dual-Track Memory for Multi-Party Dialogue",
              "summary": "Long-term conversational memory in multi-party settings requires more than retrieving relevant content from long-term conversations: it must distinguish who said what, whom each statement concerns, how individuals perceive one another, what information is shared by the group, and how states change over time. Recent studies on multi-party dialogue benchmarks show that existing general-purpose LLM memory systems tend to lose person and group relations or struggle to integrate clues distributed across members, groups, and time. Together, these issues reveal two core bottlenecks: message attribution and relational understanding in multi-party dialogue, and state reconstruction from interleaved histories. To address both, we propose SpeakerMem-R1: its dual-track memory stores speaker-labeled verbatim messages and derived states organized into person-level and group-level views, then combines evidence from both tracks by entity, event, and time at query time. To reduce attribution and update errors during structured memory construction while enabling local deployment, we train Writer-R1 with SpeakerLevenshtein and speaker-conditioned GRPO. On GroupMemBench, SocialMemBench, and EverMemBench, SpeakerMem-R1 achieves binary accuracies of 47.9%, 69.2%, and 61.9%, respectively. On the publicly reported EverMemBench leaderboard from EverMind-AI, we achieves 62.33%, the best reported result among the latest state-of-the-art frameworks. It also achieves 70.85% on all 1,986 LoCoMo questions, which we use as a two-person long-term conversation boundary test. In a controlled evaluation of 305 questions, RL raises the SFT Writer's mean accuracy from 57.38% to 68.20%. We report both binary accuracy and token-F1, and ablations show that the verbatim and structured tracks, as well as person-level and group-level views, are complementary under the standardized evaluation interface.",
              "organization": {
                "_id": "61bac2af530e5c78d7b99667",
                "name": "zju",
                "fullname": "Zhejiang University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/5e1058e9fcf41d740b69966d/7G1xjlxwCdMEmKcxNR0n5.png"
              },
              "link": "https://huggingface.co/papers/2609.26780",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 76
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26780"
    },
    {
      "id": "7c8c98c6d53e3976dab1",
      "title": "RewardVerse: Rubric-Guided Policy Optimization for Video Reward Modeling",
      "content_text": "Reinforcement learning (RL) is vital for optimizing video generation models, with a robust reward model (RM) serving as the cornerstone. However, existing video reward models often produce unstable scalar scores because they directly map complex, subjective video quality into a single score without explicit evaluation criteria. This leads to scalar drift, where the scoring scale collapses or shifts across different prompts, making the reward unreliable for RL. Drawing inspiration from…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "7c8c98c6d53e3976dab1",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.22947",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "RewardVerse: Rubric-Guided Policy Optimization for Video Reward Modeling",
            "item_type": "record",
            "summary": "added: RewardVerse: Rubric-Guided Policy Optimization for Video Reward Modeling",
            "after": {
              "title": "RewardVerse: Rubric-Guided Policy Optimization for Video Reward Modeling",
              "summary": "Reinforcement learning (RL) is vital for optimizing video generation models, with a robust reward model (RM) serving as the cornerstone. However, existing video reward models often produce unstable scalar scores because they directly map complex, subjective video quality into a single score without explicit evaluation criteria. This leads to scalar drift, where the scoring scale collapses or shifts across different prompts, making the reward unreliable for RL. Drawing inspiration from professional human annotation engineering, we address this problem with RewardVerse, a rubric-based video reward framework that introduces a dynamic rubric as an intermediate representation between the evaluation query and the scorer. Instead of unconstrained direct scoring, RewardVerse first generates explicit evaluation criteria and then performs rubric-guided scoring, providing a stable semantic anchor that mitigates scalar drift. To efficiently optimize this collaborative pipeline, we propose Rubric-Guided Policy Optimization (RGPO), a two-stage training algorithm. RGPO first warms up the scorer using self-evolving seed rubrics and then jointly optimizes the rubric generator to produce query-adaptive evaluation criteria while continuously aligning the scorer with human ratings. Extensive experiments on the 16-dimensional EvalVerse benchmark and external datasets demonstrate that RewardVerse mitigates scalar drift, achieves state-of-the-art performance on both pointwise and pairwise evaluation, and provides a robust and interpretable reward signal for RL in video generation.",
              "organization": {
                "_id": "66543b6e420092799d2f625c",
                "name": "tencent",
                "fullname": "Tencent",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/5dd96eb166059660ed1ee413/Lp3m-XLpjQGwBItlvn69q.png"
              },
              "link": "https://huggingface.co/papers/2609.22947",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 21
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.22947"
    },
    {
      "id": "1bc15611ab13a827fac2",
      "title": "PACT: From Credit Assignment to Critic Alignment",
      "content_text": "Reinforcement learning has become a central component of large language model (LLM) post-training, yet token-level credit lacks a generally accepted mathematical definition, leaving its relationship to commonly used training signals unclear. We formulate three regularity conditions, namely Completeness, Prefix Consistency, and Neutrality, and prove that they uniquely determine token-level credit. This characterization provides a unified basis for explaining phenomena across existing algorithms…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "1bc15611ab13a827fac2",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26355",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "PACT: From Credit Assignment to Critic Alignment",
            "item_type": "record",
            "summary": "added: PACT: From Credit Assignment to Critic Alignment",
            "after": {
              "title": "PACT: From Credit Assignment to Critic Alignment",
              "summary": "Reinforcement learning has become a central component of large language model (LLM) post-training, yet token-level credit lacks a generally accepted mathematical definition, leaving its relationship to commonly used training signals unclear. We formulate three regularity conditions, namely Completeness, Prefix Consistency, and Neutrality, and prove that they uniquely determine token-level credit. This characterization provides a unified basis for explaining phenomena across existing algorithms and guides the development of an improved actor-critic training procedure. Through this lens, an ideal teacher in On-Policy Distillation (OPD) acts as an implicit critic, yielding an expected policy gradient proportional to that induced by token-level credit. Response-level REINFORCE Leave-One-Out (RLOO) signals match the expected policy-gradient contribution of token-level credit despite their coarser granularity. We further establish approximate credit sparsity under bounded outcome rewards and show how intermediate critic errors in Generalized Advantage Estimation (GAE) can become comparable to the underlying credit. These motivate Policy Aligned Critic Training (PACT), which adopts an Actor-then-Critic update order to apply importance sampling correction to critic training and better align the critic with the updated policy. In agentic mathematical reasoning, PACT achieves 72.87% average accuracy across four benchmarks, outperforming GRPO and PPO by 8.80 and 13.16 percentage points, respectively. On SWE-bench Verified, PACT achieves a pass rate of 67.4%, outperforming PPO, GRPO, and SAO by 2.4, 2.0, and 3.8 percentage points, respectively.",
              "organization": {
                "_id": "6a96bf5bae44840b913e27b8",
                "name": "AllSpark-Research",
                "fullname": "AllSpark Research",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6a96be78642bb1433bbbf002/e9W8yeoVttMzcdc_Z_VrR.png"
              },
              "link": "https://huggingface.co/papers/2609.26355",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 15
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26355"
    },
    {
      "id": "c58fbcb18e8f3f746824",
      "title": "Hunyuan-A13B Technical Report",
      "content_text": "We present Hunyuan-A13B, an open-source large language model based on a Mixture-of-Experts architecture. It contains 80 billion total parameters but activates only 13 billion during inference, balancing model capability, computational efficiency, and deployment cost. The model is pretrained on a rigorously filtered 20T-token corpus with enhanced STEM data curation, improving factual reliability and reasoning ability. High-quality supervised fine-tuning and large-scale reinforcement learning…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "c58fbcb18e8f3f746824",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27284",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Hunyuan-A13B Technical Report",
            "item_type": "record",
            "summary": "added: Hunyuan-A13B Technical Report",
            "after": {
              "title": "Hunyuan-A13B Technical Report",
              "summary": "We present Hunyuan-A13B, an open-source large language model based on a Mixture-of-Experts architecture. It contains 80 billion total parameters but activates only 13 billion during inference, balancing model capability, computational efficiency, and deployment cost. The model is pretrained on a rigorously filtered 20T-token corpus with enhanced STEM data curation, improving factual reliability and reasoning ability. High-quality supervised fine-tuning and large-scale reinforcement learning further enhance its overall performance. Hunyuan-A13B also introduces a dual-mode Chain-of-Thought framework that adapts reasoning depth to task complexity: fast thinking for routine queries and slow thinking for complex, multi-step problems. Evaluations show competitive performance across mathematics, science, programming, general language understanding, and agent tasks, often approaching that of much larger models. Its high inference throughput makes it suitable for latency-sensitive applications. We release Hunyuan-A13B to support open research and practical LLM deployment.",
              "organization": {
                "_id": "6645f953c39288df638dbdd5",
                "name": "Tencent-Hunyuan",
                "fullname": "Tencent Hunyuan",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/62d22496c58f969c152bcefd/woKSjt2wXvBNKussyYPsa.png"
              },
              "link": "https://huggingface.co/papers/2609.27284",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 6
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27284"
    },
    {
      "id": "4767675595c281ffc420",
      "title": "Verifiable Hidden Dynamics Play: Generating Agentic RL Environments from Solved Mechanisms",
      "content_text": "Language-model agents increasingly face long-horizon tasks with evolving state, interdependent decisions, and delayed outcomes. Scaling their training requires diverse agentic environments, dependable outcome signals, and low extension cost. Existing generation pipelines commonly construct an environment before defining its outcome rule or annotating its trajectories, leaving dynamics and evaluation to be aligned post hoc. VHD-Play reverses this dependency by sampling and solving a mathematical…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "4767675595c281ffc420",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27321",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Verifiable Hidden Dynamics Play: Generating Agentic RL Environments from Solved Mechanisms",
            "item_type": "record",
            "summary": "added: Verifiable Hidden Dynamics Play: Generating Agentic RL Environments from Solved Mechanisms",
            "after": {
              "title": "Verifiable Hidden Dynamics Play: Generating Agentic RL Environments from Solved Mechanisms",
              "summary": "Language-model agents increasingly face long-horizon tasks with evolving state, interdependent decisions, and delayed outcomes. Scaling their training requires diverse agentic environments, dependable outcome signals, and low extension cost. Existing generation pipelines commonly construct an environment before defining its outcome rule or annotating its trajectories, leaving dynamics and evaluation to be aligned post hoc. VHD-Play reverses this dependency by sampling and solving a mathematical model before a corpus-grounded setter renders its decision process as stateful tools. The executable dynamics and trajectory-scoring reference are inherited from the same solved model. The pipeline produces 3,300 diverse agentic environments at a cost of a few cents each. Training Qwen3.6-35B-A3B on three families raises its mean agentic score from 0.204 to 0.815 in a five-family diagnostic. Gains also appear on held-out instances from all three training families and eight unseen mechanism families, then extend beyond the generated substrate to external benchmarks for general function calling, travel planning, and 365-day e-commerce. On E-Commerce Bench, the trained checkpoint completes every run without bankruptcy and exceeds Qwen3.7-Max. We compare written-out problems with stateful versions that reveal or hide their parameters. The comparison shows that most of the learnable gap lies in stateful interaction rather than underlying problem solving. A frozen 35B setter realizes larger environments, and scale-matched training retains gains as mechanism size and horizon grow, indicating the potential for an evolving training substrate.",
              "organization": {
                "_id": "64c8b5837fe12ecd0a7e92eb",
                "name": "Qwen",
                "fullname": "Qwen",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6215ca5692c0ecfba9186921/hrRM50-6XcdWgg2AKpENG.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.27321",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 5
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27321"
    },
    {
      "id": "b6f6a91dff6855dc27c9",
      "title": "WhatWorkedBench: Benchmarking Experimental Understanding in AI Agents",
      "content_text": "AI research agents need reliable knowledge of how their experiments change outcomes. We introduce WhatWorkedBench to measure experimental understanding, the accuracy of predictions about component changes after budgeted experimentation. Agents inspect code, select measurements, and submit a response surface, a table predicting scores for every configuration of component settings. Exhaustive CPU execution supplies reference effects for changing each component while holding the others fixed…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "b6f6a91dff6855dc27c9",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27490",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "WhatWorkedBench: Benchmarking Experimental Understanding in AI Agents",
            "item_type": "record",
            "summary": "added: WhatWorkedBench: Benchmarking Experimental Understanding in AI Agents",
            "after": {
              "title": "WhatWorkedBench: Benchmarking Experimental Understanding in AI Agents",
              "summary": "AI research agents need reliable knowledge of how their experiments change outcomes. We introduce WhatWorkedBench to measure experimental understanding, the accuracy of predictions about component changes after budgeted experimentation. Agents inspect code, select measurements, and submit a response surface, a table predicting scores for every configuration of component settings. Exhaustive CPU execution supplies reference effects for changing each component while holding the others fixed. These effects capture combinations of changes across 36 tasks from 30 data sources and 8 workflow types, with 1248 configuration records. Core evaluation combines 4,206 numerical-control records across all eight families and 108 agent episodes across the original six. At eight new measurements, pair-effect ridge selects an optimum on 15 of 22 sources and limits every effect error to 10% of score range on three. Fitting a Gaussian process (GP) to the same agent observations raises effect recovery, accuracy relative to true effect magnitude, from 0.632 to 0.698 in the original Flash cohort and from 0.621 to 0.720 in an additional cohort. On six completed beat-detection and graph submissions, the same-observation GP raises family-macro recovery from 0.303 to 0.455. On six workflows with six binary options at 20 new measurements, encoding code equivalences, configurations with identical behavior, raises GP recovery from 0.248 to 0.462. WhatWorkedBench supports research on experimental agents, adaptive experimental design, numerical inference, and use of program structure.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/6310a812a23f0327bce68778/WcBCzCBxqgjQge9mTWzjD.png"
              ],
              "organization": {
                "_id": "691d9a1012cc4d473e1c862f",
                "name": "CarnegieMellonU",
                "fullname": "Carnegie Mellon University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68e396f2b5bb631e9b2fac9a/6I146aJvxxlRCEbYFFAeQ.png"
              },
              "link": "https://huggingface.co/papers/2609.27490",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 8
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27490"
    },
    {
      "id": "8364dc3d3db28af3d9c7",
      "title": "InternW0: A Foundational Physical World Model for Efficient Real-World Interactions",
      "content_text": "Physical intelligence requires more than predicting how the world may evolve: predictions must remain actionable as the world continues to change. We introduce InternW0, the first instantiation of the InternW physical world model series from Shanghai AI Laboratory, built around omnimodal interfaces, asynchronous multi-frequency processing, and local physical modeling under partial observations and external influences. InternW0 jointly learns future visual dynamics and continuous robot control…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "8364dc3d3db28af3d9c7",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27656",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "InternW0: A Foundational Physical World Model for Efficient Real-World Interactions",
            "item_type": "record",
            "summary": "added: InternW0: A Foundational Physical World Model for Efficient Real-World Interactions",
            "after": {
              "title": "InternW0: A Foundational Physical World Model for Efficient Real-World Interactions",
              "summary": "Physical intelligence requires more than predicting how the world may evolve: predictions must remain actionable as the world continues to change. We introduce InternW0, the first instantiation of the InternW physical world model series from Shanghai AI Laboratory, built around omnimodal interfaces, asynchronous multi-frequency processing, and local physical modeling under partial observations and external influences. InternW0 jointly learns future visual dynamics and continuous robot control through an asymmetric video--action architecture with flow matching. A high-capacity video expert provides longer-horizon predictive context, while a lightweight action expert operates at a faster timescale. Instead of regenerating the future for every action update, InternW0 reuses layerwise K/V and adapts it to newly observed states through observation-conditioned context routing. Domain-specific interfaces and soft prompts support heterogeneous embodiments, while contact-aware post-training incorporates force and tactile signals for contact-rich manipulation. We train InternW0 on approximately 7,200 hours of heterogeneous robot and egocentric data, including EgoLab, a 275-hour real-laboratory egocentric dataset. Evaluation spans simulation benchmarks and real-world scientific tasks, including a 15-stage metal--organic framework synthesis workflow and 5-stage contact- and force-aware dexterous manipulation for general-purpose quantitative pipetting. These results advance scalable, asynchronous, and science-native physical world models for universal and efficient real-world interactions.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/6039478ab3ecf716b1a5fd4d/ESs9otzT66BaGbgG-VekR.png"
              ],
              "link": "https://huggingface.co/papers/2609.27656",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27656"
    },
    {
      "id": "4030974f24d9e1851617",
      "title": "The Past Frames the Future: Memory for Autoregressive Video Generation",
      "content_text": "Advances in generative models have improved video fidelity, enabling long-horizon generation, interactive world modeling, and evolving visual environments. Autoregressive (AR) video generation extends visual sequences through causal rollouts. However, a fundamental bottleneck emerges: as the generated sequence expands, practical models must operate under strictly bounded context windows, storage, and computational limits. Consequently, critical historical information, e.g., entity identities…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "4030974f24d9e1851617",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.28466",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "The Past Frames the Future: Memory for Autoregressive Video Generation",
            "item_type": "record",
            "summary": "added: The Past Frames the Future: Memory for Autoregressive Video Generation",
            "after": {
              "title": "The Past Frames the Future: Memory for Autoregressive Video Generation",
              "summary": "Advances in generative models have improved video fidelity, enabling long-horizon generation, interactive world modeling, and evolving visual environments. Autoregressive (AR) video generation extends visual sequences through causal rollouts. However, a fundamental bottleneck emerges: as the generated sequence expands, practical models must operate under strictly bounded context windows, storage, and computational limits. Consequently, critical historical information, e.g., entity identities, dynamic states, and intervention-induced causal changes, often leaves the active context long before its relevance diminishes. Overcoming this limitation and maintaining temporal persistence constitutes a fundamental memory problem. We present a systematic and comprehensive review of memory mechanisms in AR video generation. We formulate memory operationally as persistent historical information maintained across outer AR steps, capable of influencing future generation even after the originating evidence is no longer locally accessible. Building upon this unified framework, we organize the literature through five complementary perspectives: (I) Forms, the representational carriers of history; (II) Functions, the specific semantic and physical information requiring preservation; (III) Operations, the lifecycle of writing, reading, updating, managing, and integrating memory; (IV) Learning, the optimization of memory behaviors under closed-loop rollouts; and (V) Evaluation, the paradigms for diagnosing genuine memory capabilities. We conclude by synthesizing open challenges, including composable and resource-aware memory architectures, trustworthy state updating, self-rollout learning, and standardized evaluation. By bridging representations, mechanisms, and learning paradigms, this paper establishes a structured foundation for developing reliable, memory-conditioned video generation systems.",
              "link": "https://huggingface.co/papers/2609.28466",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 35
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.28466"
    },
    {
      "id": "6ef3f04bcf369794ed1a",
      "title": "Schrödinger's Code Repository: Have LLMs Learned SWE-bench or Memorized It?",
      "content_text": "Repository-level coding benchmarks have become the standard for evaluating coding agents, yet they inherently suffer from data leakage because they are built upon popular open-source repositories repeatedly used for training. Consequently, strong performance may reflect memorization of canonical repository cues rather than robust repository reasoning. We propose SchrodingerRepo (Schrödinger's Repository), an evaluation framework for testing coding agents under dynamically instantiated…",
      "date_published": "2026-09-24T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "6ef3f04bcf369794ed1a",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.27891",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Schrödinger's Code Repository: Have LLMs Learned SWE-bench or Memorized It?",
            "item_type": "record",
            "summary": "added: Schrödinger's Code Repository: Have LLMs Learned SWE-bench or Memorized It?",
            "after": {
              "title": "Schrödinger's Code Repository: Have LLMs Learned SWE-bench or Memorized It?",
              "summary": "Repository-level coding benchmarks have become the standard for evaluating coding agents, yet they inherently suffer from data leakage because they are built upon popular open-source repositories repeatedly used for training. Consequently, strong performance may reflect memorization of canonical repository cues rather than robust repository reasoning. We propose SchrodingerRepo (Schrödinger's Repository), an evaluation framework for testing coding agents under dynamically instantiated repository representations. Instead of repeatedly using a static representation of the test repository, SchrodingerRepo treats the test repository as an evaluation-time latent variable that is dynamically instantiated only when the agent enters the evaluation environment. The instantiated repository preserves the original executable behavior while eroding familiar cues such as naming conventions, file layouts, and implementation patterns through four transformation levels: problem statement reconstruction, namespace remapping, intra-file layout reordering, and functionality-preserving code rewriting. We evaluate popular LLMs on SWE-bench Verified and SWE-QA. Results show that removing familiar repository cues consistently degrades agent performance and substantially increases interaction costs across models. Further analysis reveals that the additional cost is primarily caused by increased difficulty in repository exploration and localization. These findings suggest that current coding agents may partially rely on memorized repository-side cues, highlighting the need for evaluation under dynamically instantiated repository representations.",
              "organization": {
                "_id": "63e5ef7bf2e9a8f22c515654",
                "name": "SJTU",
                "fullname": "Shanghai Jiao Tong University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/1676013394657-63e5ee22b6a40bf941da0928.png"
              },
              "link": "https://huggingface.co/papers/2609.27891",
              "published_at": "2026-09-24T00:00:00.000Z",
              "upvotes": 14
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.27891"
    },
    {
      "id": "c2dfdd2ca458cf124b32",
      "title": "HARMONY: Hierarchical Agentic Reasoning for MONocular Image-to-Scene Synthesis",
      "content_text": "Compositional 3D scene reconstruction has recently been explored from two directions: agentic reasoning that provides semantic understanding of spatial relationships but lacks precise alignment with input images; and visual geometry foundation models that predict dense point maps from input images but the reconstruction quality is limited. Therefore, recovering a complete 3D scene from a single monocular image with accurate inter-object relationships and high-fidelity reconstruction quality…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "c2dfdd2ca458cf124b32",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26793",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "HARMONY: Hierarchical Agentic Reasoning for MONocular Image-to-Scene Synthesis",
            "item_type": "record",
            "summary": "added: HARMONY: Hierarchical Agentic Reasoning for MONocular Image-to-Scene Synthesis",
            "after": {
              "title": "HARMONY: Hierarchical Agentic Reasoning for MONocular Image-to-Scene Synthesis",
              "summary": "Compositional 3D scene reconstruction has recently been explored from two directions: agentic reasoning that provides semantic understanding of spatial relationships but lacks precise alignment with input images; and visual geometry foundation models that predict dense point maps from input images but the reconstruction quality is limited. Therefore, recovering a complete 3D scene from a single monocular image with accurate inter-object relationships and high-fidelity reconstruction quality remains challenging. In this paper, we present HARMONY, a hierarchical chain-of-thought framework that leverages both agentic reasoning and visual geometry foundation. Given an image of an indoor scene, starting from an empty 3D floorplan, HARMONY first calibrates the camera against the reference image to establish a semantically-grounded spatial frame, then uses agentic VLM reasoning to recover the 3D room layout and an initial placement order. It then places the objects in a hierarchical order, from wall-mounted elements, free-standing furniture, to dependent decorations on top of furniture. We also use depth-first traversal for furniture so each placement conditions on previously resolved structure and a reflective feedback loop to avoid error accumulation. After each object placement by VLM, we use the point cloud estimations to perform geometry-based refinement so that the rendered image aligns better with the input. HARMONY can produce 3D scenes that are semantically consistent and perceptually aligned with the reference image, extending single-image compositional reconstruction to complex indoor scene images. Experiments on synthetic and real-world images demonstrate that HARMONY outperforms the evaluated reconstruction baselines, while qualitative comparisons with GPT-6 Astra suggest more faithful object arrangements and better preservation of scene details.",
              "organization": {
                "_id": "633cfd005d7b7741bef6aa99",
                "name": "upenn",
                "fullname": "University of Pennsylvania",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68e396f2b5bb631e9b2fac9a/FFSbROS6R8vPMIyXb7iIU.png"
              },
              "link": "https://huggingface.co/papers/2609.26793",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 0
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26793"
    },
    {
      "id": "58e36c201a27e9588b1f",
      "title": "Tri-PvP: Exposing Modality Bias in Omni-Modal Large Language Models through Perceptual-Propositional Evidence Conflicts",
      "content_text": "Omni-modal large language models (OLLMs) jointly process vision, audio, and text, yet their modality bias under cross-modal conflict remains underexplored. Existing benchmarks conflate two distinct forms of evidence within a single modality: perceptual signals (e.g., a photograph or recording of a dog) and propositional signals (e.g., the declarative claim \"this is a dog\"), such that any measured modality bias is inherently confounded with evidence-form bias, precluding clean attribution to…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "58e36c201a27e9588b1f",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.06011",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Tri-PvP: Exposing Modality Bias in Omni-Modal Large Language Models through Perceptual-Propositional Evidence Conflicts",
            "item_type": "record",
            "summary": "added: Tri-PvP: Exposing Modality Bias in Omni-Modal Large Language Models through Perceptual-Propositional Evidence Conflicts",
            "after": {
              "title": "Tri-PvP: Exposing Modality Bias in Omni-Modal Large Language Models through Perceptual-Propositional Evidence Conflicts",
              "summary": "Omni-modal large language models (OLLMs) jointly process vision, audio, and text, yet their modality bias under cross-modal conflict remains underexplored. Existing benchmarks conflate two distinct forms of evidence within a single modality: perceptual signals (e.g., a photograph or recording of a dog) and propositional signals (e.g., the declarative claim \"this is a dog\"), such that any measured modality bias is inherently confounded with evidence-form bias, precluding clean attribution to either source. To address this, we introduce Tri-PvP, an 8,000-sample tri-modal conflict benchmark crossing vision, audio, and text, where vision and audio each take perceptual or propositional form. Evaluating five OLLMs, we find robust visual bias across most models and evidence-type conditions. Crucially, we reveal a systematic asymmetry in evidence-form bias: models exhibit a stronger bias toward perceptual signal in vision but propositional in audio. Further analyses via layer-wise linear probing and contrastive decoding reveal that modality bias is already linearly decodable from early representation layers and can only be partially mitigated, calling for mitigation strategies beyond surface-level interventions.",
              "organization": {
                "_id": "69c519398b490144e0d7c658",
                "name": "Tri-PvP",
                "fullname": "Tri-PvP",
                "avatar": "https://www.gravatar.com/avatar/6e9fe42b0fb8c12bb68cfceb3b028c10?d=retro&size=100"
              },
              "link": "https://huggingface.co/papers/2609.06011",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 5
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.06011"
    },
    {
      "id": "89377b60d41595a5435c",
      "title": "Embedding Physics Priors in Robot Learning: A Survey",
      "content_text": "The rapid progress of artificial intelligence is reshaping robotics and accelerating the adoption of learning-based approaches. While purely data-driven methods have achieved remarkable success in computer vision and natural language processing, robotics remains constrained by limited data, complex real-world interactions, and the need for reliable operation. These challenges have motivated the exploration of physics-embedded robot learning, which embeds physics priors into learning algorithms…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "89377b60d41595a5435c",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.22319",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Embedding Physics Priors in Robot Learning: A Survey",
            "item_type": "record",
            "summary": "added: Embedding Physics Priors in Robot Learning: A Survey",
            "after": {
              "title": "Embedding Physics Priors in Robot Learning: A Survey",
              "summary": "The rapid progress of artificial intelligence is reshaping robotics and accelerating the adoption of learning-based approaches. While purely data-driven methods have achieved remarkable success in computer vision and natural language processing, robotics remains constrained by limited data, complex real-world interactions, and the need for reliable operation. These challenges have motivated the exploration of physics-embedded robot learning, which embeds physics priors into learning algorithms. By encoding the underlying physical laws and constraints, physics priors can complement limited data with robotics-specific inductive biases, potentially improving generalization, interpretability, and sample efficiency. However, the literature on physics-embedded robot learning remains fragmented across terminology, methodologies, and application domains, making it difficult to assess this growing body of work. This survey reviews physics-embedded robot learning across a broad range of physics priors, robotics applications, and machine learning models, from single-layer perceptrons to generative foundation models. We adopt a unified taxonomy that classifies existing approaches according to their physics embedding: physics-guided inputs, data, and representations; physics-encoded model architectures; and physics-informed training loss functions. Building on this taxonomy, we review methods for robot dynamics learning, trajectory planning, prediction, control, and estimation, together with the corresponding open-source software ecosystem. We identify key open challenges, and outline promising future research directions. Overall, we argue that physics priors provide a particularly relevant robotics-specific inductive bias, complementing rather than replacing data-driven learning, and paving the way toward more generalizable, data-efficient, and trustworthy robotic systems.",
              "organization": {
                "_id": "6a4f83663e43ae629a202b05",
                "name": "TUM-AVS",
                "fullname": "TUM - Professorship of Autonomous Vehicle Systems",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6723459e7ddd97700c1c1c6a/AJSXS1Qq8ywx4puQ91Fhm.png"
              },
              "link": "https://huggingface.co/papers/2609.22319",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 2
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.22319"
    },
    {
      "id": "7dca61c56ea4c432b7fc",
      "title": "Agensh: Scaling Organizational Intelligence to 1,024 Agents",
      "content_text": "A multi-agent system can reduce latency on complex tasks by executing work concurrently. Several pioneering harness frameworks support multi-agent systems. However, the scalability of current multi-agent harnesses is often constrained by a central orchestrator's capacity to allocate tasks and coordinate workers. To address this limitation, we introduce Agensh, a scalable self-organized multi-agent harness without a central orchestrator: concurrent workers execute a multi-agent cooperation loop…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "7dca61c56ea4c432b7fc",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26781",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Agensh: Scaling Organizational Intelligence to 1,024 Agents",
            "item_type": "record",
            "summary": "added: Agensh: Scaling Organizational Intelligence to 1,024 Agents",
            "after": {
              "title": "Agensh: Scaling Organizational Intelligence to 1,024 Agents",
              "summary": "A multi-agent system can reduce latency on complex tasks by executing work concurrently. Several pioneering harness frameworks support multi-agent systems. However, the scalability of current multi-agent harnesses is often constrained by a central orchestrator's capacity to allocate tasks and coordinate workers. To address this limitation, we introduce Agensh, a scalable self-organized multi-agent harness without a central orchestrator: concurrent workers execute a multi-agent cooperation loop, continuously gathering context, claiming and self-assigning sub-tasks, taking action and sharing findings, verifying results, and merging progress in an asynchronous manner. The loop is supported by the agentic organization infrastructure comprising three components: a shared workspace holds proposed, ongoing, and completed work; a message interface lets workers communicate; and shared context retains reusable findings and work intentions. To test the scalability of Agensh, we evaluate it on the five hardest ProgramBench tasks with GPT-5.6-sol (high). Scaling from 1 to 128 agents raises the mean final test-pass rate from 19.31% to 28.78%, an approximately 49% relative improvement. Larger organizations reach comparable test-pass rates earlier. On pandoc, scaling from 1 to 1,024 agents raises the final test-pass rate from 33.89% to 55.06%. Worker trajectories further show that different forms of self-organized cooperation gradually emerges and standardizes as the organization grows. These results reveal the number of agents as a new scaling dimension for multi-agent organizations to expand the frontier of general intelligence, offering a practical solution for complex tasks under hard latency constraints or time budgets.",
              "organization": {
                "_id": "68151d0f51add3813f3f7d1b",
                "name": "MicrosoftResearch",
                "fullname": "Microsoft Research",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6529a4f2f1205983224fa513/PeuVr7jSuJflmDBBGxoDX.png"
              },
              "link": "https://huggingface.co/papers/2609.26781",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 14
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26781"
    },
    {
      "id": "2c1d2e932ecac4dc7a44",
      "title": "JEV-as-a-Judge: Accept When Confident, Escalate When Unsure",
      "content_text": "LLM-as-a-judge enables evaluation across diverse tasks, but inference cost and confidence reliability become critical at scale. We study whether a decision-only judge can provide an economical first pass and identify when stronger evaluation is needed. Comparing jev-as-a-judge with sixteen generative and reward-model judges, with blinded human adjudication, we find it within three percentage points of a state-of-the-art LLM judge, our strongest comparator, on ordinary preference and…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "2c1d2e932ecac4dc7a44",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26550",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "JEV-as-a-Judge: Accept When Confident, Escalate When Unsure",
            "item_type": "record",
            "summary": "added: JEV-as-a-Judge: Accept When Confident, Escalate When Unsure",
            "after": {
              "title": "JEV-as-a-Judge: Accept When Confident, Escalate When Unsure",
              "summary": "LLM-as-a-judge enables evaluation across diverse tasks, but inference cost and confidence reliability become critical at scale. We study whether a decision-only judge can provide an economical first pass and identify when stronger evaluation is needed. Comparing jev-as-a-judge with sixteen generative and reward-model judges, with blinded human adjudication, we find it within three percentage points of a state-of-the-art LLM judge, our strongest comparator, on ordinary preference and evidence-grounded factuality at 0.36% of the comparator's fee. Larger gaps arise when judgments require checking a derivation or resisting an elaborately written wrong answer. On several benchmarks, JEV's gap to this comparator is concentrated in low-confidence decisions. A frozen cascade that accepts confident verdicts and escalates uncertain ones retains 99% of the comparator's accuracy at lower cost.",
              "organization": {
                "_id": "691d9a1012cc4d473e1c862f",
                "name": "CarnegieMellonU",
                "fullname": "Carnegie Mellon University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68e396f2b5bb631e9b2fac9a/6I146aJvxxlRCEbYFFAeQ.png"
              },
              "link": "https://huggingface.co/papers/2609.26550",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 25
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26550"
    },
    {
      "id": "840825d18a4406976093",
      "title": "LatentPort: Beyond KV Cache - Cross-Model Transfer of Recurrent Memory in Hybrid Language Models: A 4B-to-9B Hybrid-State Handoff Without Target Prefix Replay",
      "content_text": "Can one language model hand its live memory to another without the receiver rereading the context? We demonstrate useful persistent hybrid-state transfer across one architecture-matched Qwen3.5 4B-to-9B sibling pair. To our knowledge, this is the first demonstrated cross-model handoff of persistent recurrent inference state between differently sized hybrid language models without target prefix replay. Translated attention KV alone leaves a large gap; adding the Gated DeltaNet (GDN)…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "840825d18a4406976093",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25053",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "LatentPort: Beyond KV Cache - Cross-Model Transfer of Recurrent Memory in Hybrid Language Models: A 4B-to-9B Hybrid-State Handoff Without Target Prefix Replay",
            "item_type": "record",
            "summary": "added: LatentPort: Beyond KV Cache - Cross-Model Transfer of Recurrent Memory in Hybrid Language Models: A 4B-to-9B Hybrid-State Handoff Without Target Prefix Replay",
            "after": {
              "title": "LatentPort: Beyond KV Cache - Cross-Model Transfer of Recurrent Memory in Hybrid Language Models: A 4B-to-9B Hybrid-State Handoff Without Target Prefix Replay",
              "summary": "Can one language model hand its live memory to another without the receiver rereading the context? We demonstrate useful persistent hybrid-state transfer across one architecture-matched Qwen3.5 4B-to-9B sibling pair. To our knowledge, this is the first demonstrated cross-model handoff of persistent recurrent inference state between differently sized hybrid language models without target prefix replay. Translated attention KV alone leaves a large gap; adding the Gated DeltaNet (GDN) persistent-state package lowers teacher-forced negative log-likelihood (NLL), the average next-token log-loss, by 0.747 nats/token (95% paired document bootstrap CI [0.6921, 0.8047]), improving all 64 PG19 documents. Direct recurrent and convolution reuse outperforms the tested learned GDN maps, consistent with partial functional compatibility of persistent-state coordinates. A fresh component factorial selects translated KV with direct recurrent and convolution state. An additional 434,176-parameter correction improves that base on 64 fresh web documents: continuation loss is 0.076 nats/token above native 9B (excess NLL), Jensen-Shannon (JS) divergence is 0.022, and native context recovery (NCR) is 0.918. Corrected 9B significantly beats continued 4B inference while processing zero historical prefix tokens. Evidence covers one direction, one geometry-matched Base-model pair, and 4K teacher-forced continuation; the near-native gate failed, the 16K branch was not run, and free-generation equivalence and a general state interface remain unproven.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/63e3175e88181ba2ab0a0268/xdVkqSIAwKbt6nrvD5goH.png"
              ],
              "link": "https://huggingface.co/papers/2609.25053",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25053"
    },
    {
      "id": "90ba15dd8eb5594942c0",
      "title": "RoboFollow: Unveiling the Instruction Following Mirage in Embodied Agents",
      "content_text": "Modern embodied agents achieve impressive success rates, yet their actual instruction-following ability is far weaker than these numbers suggest. We trace this illusion to a structural property we term low scene entropy: when a visual scene admits only one valid task, language becomes redundant and a policy can score highly while barely using it. We introduce RoboFollow, a diagnostic benchmark with three principles: (1) High Scene Entropy: each training scene supports multiple kinematically…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "90ba15dd8eb5594942c0",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25636",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "RoboFollow: Unveiling the Instruction Following Mirage in Embodied Agents",
            "item_type": "record",
            "summary": "added: RoboFollow: Unveiling the Instruction Following Mirage in Embodied Agents",
            "after": {
              "title": "RoboFollow: Unveiling the Instruction Following Mirage in Embodied Agents",
              "summary": "Modern embodied agents achieve impressive success rates, yet their actual instruction-following ability is far weaker than these numbers suggest. We trace this illusion to a structural property we term low scene entropy: when a visual scene admits only one valid task, language becomes redundant and a policy can score highly while barely using it. We introduce RoboFollow, a diagnostic benchmark with three principles: (1) High Scene Entropy: each training scene supports multiple kinematically distinct task branches, making vision alone insufficient and forcing reliance on language. (2) Hierarchical Diagnostic Protocol: a four-level protocol (L0--L3) progressively perturbs visual layout and semantics, probing whether equivalent instructions yield consistent behavior and distinct ones yield discriminable behavior across spatial relations, attributes, trajectory constraints, and logic. (3) Confound-Controlled Diagnosis: we simplify interaction objects, restrict actions to the trained repertoire and report stage-wise Intent and Execution scores, isolating comprehension from motor execution. Evaluation of nine VLA and WAM policies shows that strong L0 performance, where attained, does not reliably transfer to L1--L3 under our fine-tuning setup. Representative mitigations, including stronger VLM backbones, QA co-training, LangForce, and Classifier-Free Guidance, all fail to close this gap. RoboFollow exposes genuine instruction following as a critical, overlooked bottleneck. Code and dataset are available at https://github.com/AutoLab-SAI-SJTU/RoboFollow and https://huggingface.co/datasets/AutoLab-SJTU/robofollow-data.",
              "organization": {
                "_id": "68ee0edd23dc954f7744ac27",
                "name": "AutoLab-SJTU",
                "fullname": "AutoLab",
                "avatar": "https://www.gravatar.com/avatar/d35be2364b0e0b9b57d2487a06bfe26a?d=retro&size=100"
              },
              "link": "https://huggingface.co/papers/2609.25636",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25636"
    },
    {
      "id": "1d7f17b1181f9250fc0e",
      "title": "Blaming Across the Aisle: Political Contrasting and Blame Attribution in the Danish Parliament",
      "content_text": "Political discourse is widely perceived to be growing more hostile, yet robust evidence remains scarce. This study examines blame attribution in the Danish Parliament from 1997 to 2026, combining a purpose-built classifier, BlameBERT (F1: 0.80), with multilevel statistical modeling. The classifier is constructed using an annotation-efficient pipeline for blame attribution in low-to-mid resource languages. The results reveal a banana-shaped trajectory, with blame declining until around 2016…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "1d7f17b1181f9250fc0e",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26346",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Blaming Across the Aisle: Political Contrasting and Blame Attribution in the Danish Parliament",
            "item_type": "record",
            "summary": "added: Blaming Across the Aisle: Political Contrasting and Blame Attribution in the Danish Parliament",
            "after": {
              "title": "Blaming Across the Aisle: Political Contrasting and Blame Attribution in the Danish Parliament",
              "summary": "Political discourse is widely perceived to be growing more hostile, yet robust evidence remains scarce. This study examines blame attribution in the Danish Parliament from 1997 to 2026, combining a purpose-built classifier, BlameBERT (F1: 0.80), with multilevel statistical modeling. The classifier is constructed using an annotation-efficient pipeline for blame attribution in low-to-mid resource languages. The results reveal a banana-shaped trajectory, with blame declining until around 2016 before entering a significant and sustained increase in recent years (2019-2026). Government status consistently influenced blame attribution - an effect we term political contrasting - with opposition parties blaming substantially more than governing parties. This effect was moderated by ideology: The blame-dampening effect of governing was less pronounced among right-wing parties, and ideological extremity amplified blame more strongly on the right. In recent years, the interaction between political wing and ideological extremity intensified, suggesting an ideological hardening of the blame rhetoric concentrated on the right of the political spectrum. Taken together, these patterns suggest that the perceived rise in harsh political language reflects not merely a general rhetorical drift, but an ideologically asymmetric hardening of political discourse. A sensitivity analysis showed that the conclusions were robust to varying classification thresholds.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/5ff5943752c26e9bc240bada/tbVLxlIwTO8z2ln8nZFwV.png"
              ],
              "organization": {
                "_id": "60f54b8c433e05f1f88f52d5",
                "name": "chcaa",
                "fullname": "Center for Humanities Computing Aarhus",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/5ff5943752c26e9bc240bada/qxjBLoiMIg4TX4AA4JvUH.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.26346",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26346"
    },
    {
      "id": "70cac9fb1618d7a5f84f",
      "title": "ImIR: Image-Instruction Tuning for All-in-One Image Restoration",
      "content_text": "Degradations vary widely across images, so a practical restoration system has to handle many degradation types with one model. A recent and effective recipe adapts a large pretrained image-editing model to restoration using a small low-rank adapter with a text prompt. We replace that prompt with an instruction derived from the degraded image itself. The image reaches the editor through two paths: its structure comes from the model's VAE, and its semantic instruction comes from a lightweight…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "70cac9fb1618d7a5f84f",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25267",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "ImIR: Image-Instruction Tuning for All-in-One Image Restoration",
            "item_type": "record",
            "summary": "added: ImIR: Image-Instruction Tuning for All-in-One Image Restoration",
            "after": {
              "title": "ImIR: Image-Instruction Tuning for All-in-One Image Restoration",
              "summary": "Degradations vary widely across images, so a practical restoration system has to handle many degradation types with one model. A recent and effective recipe adapts a large pretrained image-editing model to restoration using a small low-rank adapter with a text prompt. We replace that prompt with an instruction derived from the degraded image itself. The image reaches the editor through two paths: its structure comes from the model's VAE, and its semantic instruction comes from a lightweight token mapper that shifts the degraded image's vision-language embedding toward the embedding a clean image would produce. Because the instruction is a continuous vector, scaling it yields a family of valid restorations for tasks whose target is not unique, such as low-light enhancement. We adapt one Qwen-Image-Edit model to six tasks with a single adapter trained in about three hours on one GPU. The image instruction outperforms text conditioning under a matched comparison, and it supports task agnostic restoration without a degradation label, which the text variant does not.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/633e0e76fb435ad2334429c8/CBE5OYWofqpoFF7MXgZXT.png"
              ],
              "link": "https://huggingface.co/papers/2609.25267",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25267"
    },
    {
      "id": "89071ed2fa4a06c7a4e2",
      "title": "Emergent Collusion in Long-Horizon LLM Agent Interaction",
      "content_text": "LLM agents are increasingly deployed in collaborative settings, yet long-term interaction may give rise to undesirable coordination. We study the emergence of collusion in a long-horizon multi-agent environment: two agents repeatedly complete individual tasks, share task logs, verify each other's work, and receive rewards. We introduce realistic constraints that make compliance with the verification protocol incompatible with reward maximization, and find that agents increasingly deviate from…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "89071ed2fa4a06c7a4e2",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24967",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Emergent Collusion in Long-Horizon LLM Agent Interaction",
            "item_type": "record",
            "summary": "added: Emergent Collusion in Long-Horizon LLM Agent Interaction",
            "after": {
              "title": "Emergent Collusion in Long-Horizon LLM Agent Interaction",
              "summary": "LLM agents are increasingly deployed in collaborative settings, yet long-term interaction may give rise to undesirable coordination. We study the emergence of collusion in a long-horizon multi-agent environment: two agents repeatedly complete individual tasks, share task logs, verify each other's work, and receive rewards. We introduce realistic constraints that make compliance with the verification protocol incompatible with reward maximization, and find that agents increasingly deviate from the protocol over repeated interactions. Collusion emerges in 94% of trajectories across 10 models, and more capable models within the same family reach it earlier. Controlled peer interventions show that collusion is shaped by peer behavior, while ablations reveal additional effects of reward structure, the verification feedback agents receive, and their interaction history. In particular, restricting the amount and scope of interaction history available to agents reduces collusion. Overall, our findings show that long-horizon interaction can reshape how agents coordinate in ways that create safety risks.",
              "organization": {
                "_id": "63213141145cfa4c04ce8f5f",
                "name": "SALT-NLP",
                "fullname": "Social And Language Technology Lab",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/632116accafe12f481a473cb/6s98xP2tbaxg5uHpV66m7.png"
              },
              "link": "https://huggingface.co/papers/2609.24967",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 6
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24967"
    },
    {
      "id": "229cfba08eba62c358cd",
      "title": "Lean Pool: An AI-Maintained Archive of Formalized Mathematics",
      "content_text": "Lean Pool is a repository of formalized mathematics. It is grown, maintained and optimized by AI agents.",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "229cfba08eba62c358cd",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25199",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Lean Pool: An AI-Maintained Archive of Formalized Mathematics",
            "item_type": "record",
            "summary": "added: Lean Pool: An AI-Maintained Archive of Formalized Mathematics",
            "after": {
              "title": "Lean Pool: An AI-Maintained Archive of Formalized Mathematics",
              "summary": "Lean Pool is a repository of formalized mathematics. It is grown, maintained and optimized by AI agents.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/66bff1188c3816c563f42d20/rj1ouOigPMQL6zjmtXNrF.png"
              ],
              "link": "https://huggingface.co/papers/2609.25199",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 13
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25199"
    },
    {
      "id": "87c2bec57b3ebae2ee4d",
      "title": "The Tasteful Agent: Measuring and Improving Taste in Long-Horizon Tasks",
      "content_text": "LLM agents increasingly work on long-horizon tasks, and the decisions they make along the way, such as which hypothesis to test or which implementation to build on, determine the outcome of the whole run. Making these decisions well is becoming a key capability for both engineering and research agents. We refer to the ability to make good long-horizon decisions as the taste of an agent. While existing benchmarks measure the end-to-end success of agents on long-horizon tasks, none of them…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "87c2bec57b3ebae2ee4d",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25804",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "The Tasteful Agent: Measuring and Improving Taste in Long-Horizon Tasks",
            "item_type": "record",
            "summary": "added: The Tasteful Agent: Measuring and Improving Taste in Long-Horizon Tasks",
            "after": {
              "title": "The Tasteful Agent: Measuring and Improving Taste in Long-Horizon Tasks",
              "summary": "LLM agents increasingly work on long-horizon tasks, and the decisions they make along the way, such as which hypothesis to test or which implementation to build on, determine the outcome of the whole run. Making these decisions well is becoming a key capability for both engineering and research agents. We refer to the ability to make good long-horizon decisions as the taste of an agent. While existing benchmarks measure the end-to-end success of agents on long-horizon tasks, none of them measures the taste of an agent. To address this problem, we build Taste-Bench, a benchmark of taste questions constructed automatically from trajectories that agents produced in engineering and research tasks. Each question presents a decision fork, a point in a trajectory where multiple directions are available and one of them leads to a better outcome, and the evaluated model chooses among these directions without seeing what happens after the fork. We mine these forks automatically from parallel attempts at the same task and from detours inside a single trajectory, without needing human annotation. We evaluate frontier models on Taste-Bench and find that the best model answers only 59.7% of the questions correctly. We further find that forks whose deciding evidence appears later in the trajectory are much harder for every model, and that a larger reasoning budget does not improve the accuracy. Finally, we show that taste can be trained. We distill the judgment of a teacher that has seen the outcome into a student model, and the student makes better decisions on unseen tasks and improves end-to-end success on held-out SWE-bench Pro tasks.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/62cd3a3691d27e60db0698b0/rpQVNjJOt5YO0UODQflht.png"
              ],
              "organization": {
                "_id": "5e6485f787403103f9f1055e",
                "name": "microsoft",
                "fullname": "Microsoft",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/1583646260758-5e64858c87403103f9f1055d.png"
              },
              "link": "https://huggingface.co/papers/2609.25804",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 118
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25804"
    },
    {
      "id": "d3cf889b5fb4505cf933",
      "title": "StableVQ: Practical Guidelines for Stable Vector-Quantized Tokenizer Training",
      "content_text": "Vector Quantization (VQ) is fundamental to discrete visual tokenizers that power modern autoregressive and masked image generation models. While recent shared-projection codebook methods have substantially advanced codebook utilization, training stability remains a critical and underexplored challenge. We argue that the root cause lies in the entanglement of the Encoder--Decoder and Codebook training: because neither module can reliably fulfill its own responsibility in isolation, the system…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "d3cf889b5fb4505cf933",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.26774",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "StableVQ: Practical Guidelines for Stable Vector-Quantized Tokenizer Training",
            "item_type": "record",
            "summary": "added: StableVQ: Practical Guidelines for Stable Vector-Quantized Tokenizer Training",
            "after": {
              "title": "StableVQ: Practical Guidelines for Stable Vector-Quantized Tokenizer Training",
              "summary": "Vector Quantization (VQ) is fundamental to discrete visual tokenizers that power modern autoregressive and masked image generation models. While recent shared-projection codebook methods have substantially advanced codebook utilization, training stability remains a critical and underexplored challenge. We argue that the root cause lies in the entanglement of the Encoder--Decoder and Codebook training: because neither module can reliably fulfill its own responsibility in isolation, the system can only function when the two subsystems happen to cooperate---a fragile condition that breaks down precisely when training is most stressed. We propose StableVQ, which revisits the proper learning objective of each module and resolves the problems that arise when each is trained to fulfill its own role independently. Concretely, (1) Dynamic STE corrects the instability in the Encoder's learning objective, enabling it to robustly optimize the reconstruction space under discrete regularization even when codebook utilization is low. (2) Region VQ Loss reconceives the Codebook's learning objective so that it can independently guarantee full tracking of the encoder output distribution, without relying on encoder oscillations to drive activation. (3) Decoupled Schedule recognizes that the distinct responsibilities of the Encoder--Decoder and the Codebook demand distinct optimization dynamics, and assigns each an independent learning rate schedule to ensure robust system-level behavior. Built on top of shared-projection codebooks, StableVQ is lightweight and introduces no learnable parameters. Experiments on ImageNet demonstrate consistent improvements in training stability, codebook utilization, and reconstruction quality across diverse codebook sizes and initialization settings.",
              "organization": {
                "_id": "665f02ce9f9e5b38d0a256a8",
                "name": "Kwai-Kolors",
                "fullname": "Kolors Team, Kuaishou Technology",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/62f0babaef9cc6810cec02ff/sVnELkcfVo5kxg5308rkr.png"
              },
              "link": "https://huggingface.co/papers/2609.26774",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 25
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.26774"
    },
    {
      "id": "04a58b37c58dc6bd266e",
      "title": "ALPINE: Adaptive Localization for Parameter- and Sample-Efficient Few-Shot Learning",
      "content_text": "Few-shot learning research is predominantly evaluated on accuracy alone, with limited attention to the parameter and training-sample budgets required to reach that accuracy - a real constraint for practitioners without large-scale compute. We present an ultra-lightweight (22,249-34,917 parameter) spatial-relational architecture for few-shot image classification that combines fixed Gabor edge-energy guidance with a windowed, content-adaptive patch locator. Under a strictly matched…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "04a58b37c58dc6bd266e",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.22323",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "ALPINE: Adaptive Localization for Parameter- and Sample-Efficient Few-Shot Learning",
            "item_type": "record",
            "summary": "added: ALPINE: Adaptive Localization for Parameter- and Sample-Efficient Few-Shot Learning",
            "after": {
              "title": "ALPINE: Adaptive Localization for Parameter- and Sample-Efficient Few-Shot Learning",
              "summary": "Few-shot learning research is predominantly evaluated on accuracy alone, with limited attention to the parameter and training-sample budgets required to reach that accuracy - a real constraint for practitioners without large-scale compute. We present an ultra-lightweight (22,249-34,917 parameter) spatial-relational architecture for few-shot image classification that combines fixed Gabor edge-energy guidance with a windowed, content-adaptive patch locator. Under a strictly matched, iso-episode-budget protocol (250 meta-training episodes, 5 canonical seeds, 600 evaluation episodes per seed), our architecture achieves 5-shot accuracy gains, consistent across all five seeds, over Prototypical Networks, Relation Networks, and MAML on both CIFAR-FS and MiniImageNet, while using 27-53% fewer parameters than any baseline. It also converges in fewer training episodes, generalizes better to an unseen fine-grained domain (CUB-200-2011 birds, zero retraining), and is more robust to 50% occlusion and 25% spatial translation than all three baselines. A series of falsification ablations - zeroing relational tokens at inference and retraining without them entirely - shows that the architecture's pairwise relational computation, while present, is not the primary driver of its performance; the content-adaptive patch locator is. We report this honestly, together with a capacity sweep showing a genuine accuracy plateau near 22-35k parameters, and release full seed-level results and checkpoint hashes for reproducibility.",
              "link": "https://huggingface.co/papers/2609.22323",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 3
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.22323"
    },
    {
      "id": "48d98d490368ed3bd951",
      "title": "All-in-One Multilingual Scene Text Recognition with Script-aware Mixture-of-Experts",
      "content_text": "Multilingual scene text recognition (STR) remains challenging due to the scarcity of training data for most languages and the difficulty of serving diverse scripts within a single model. Existing solutions either deploy one recognizer per language, inflating cost and introducing error accumulation, or rely on massive vision-language models (VLMs) that are expensive and still inaccurate on many scripts. In this work, we pursue an all-in-one multilingual recognizer that is simpler than…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "48d98d490368ed3bd951",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24058",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "All-in-One Multilingual Scene Text Recognition with Script-aware Mixture-of-Experts",
            "item_type": "record",
            "summary": "added: All-in-One Multilingual Scene Text Recognition with Script-aware Mixture-of-Experts",
            "after": {
              "title": "All-in-One Multilingual Scene Text Recognition with Script-aware Mixture-of-Experts",
              "summary": "Multilingual scene text recognition (STR) remains challenging due to the scarcity of training data for most languages and the difficulty of serving diverse scripts within a single model. Existing solutions either deploy one recognizer per language, inflating cost and introducing error accumulation, or rely on massive vision-language models (VLMs) that are expensive and still inaccurate on many scripts. In this work, we pursue an all-in-one multilingual recognizer that is simpler than per-language experts, lighter than VLMs, and more accurate than both. First, we construct TextMuSS-10M, a large-scale synthetic scene text dataset spanning 10 scripts and 229 languages. It provides balanced and sufficient supervision where real data is unavailable. Second, we propose ScriptMoE, a script-aware Mixture-of-Experts (MoE) architecture. It shares a single visual encoder and replaces the dense decoder with a sparse MoE block, which consists of an image-level router dispatches each image to the top-2 script-aligned experts and a shared expert absorbs cross-script knowledge. Extensive experiments on our assembled TextMuSS-Bench (10 scripts, 10,899 images) show that ScriptMoE achieves the highest accuracy of 82.06%, outperforming the strongest STR baseline by 1.31%. On the CC-OCR end-to-end multilingual task, replacing only the recognizer in PP-OCRv5 with ScriptMoE lifts F1 score from 65.71% to 80.89%, slightly surpassing the best VLM (80.73%) at a fraction of the parameter count.",
              "organization": {
                "_id": "643cb0625fcffe09fb6ca688",
                "name": "Fudan-University",
                "fullname": "Fudan University",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/6437eca0819f3ab20d162e14/kWv0cGlAhAG3iNWVxowkJ.png"
              },
              "link": "https://huggingface.co/papers/2609.24058",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 39
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24058"
    },
    {
      "id": "51497f13e1d317f8072e",
      "title": "Geometric and Semantic Coupling for Interaction Understanding in 3D Scenes",
      "content_text": "Interaction understanding in 3D scenes requires a joint description of movable parts, their motion, and the regions through which they can be operated. We present Segment-Snap, which connects these outputs through the physical relationship between parts and handles. Learned predictors identify broad part surfaces and small handles. A geometric decoder uses planar and upright priors to constrain motion, then selects hinge lines using predicted handle locations, without training a motion…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "51497f13e1d317f8072e",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25247",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Geometric and Semantic Coupling for Interaction Understanding in 3D Scenes",
            "item_type": "record",
            "summary": "added: Geometric and Semantic Coupling for Interaction Understanding in 3D Scenes",
            "after": {
              "title": "Geometric and Semantic Coupling for Interaction Understanding in 3D Scenes",
              "summary": "Interaction understanding in 3D scenes requires a joint description of movable parts, their motion, and the regions through which they can be operated. We present Segment-Snap, which connects these outputs through the physical relationship between parts and handles. Learned predictors identify broad part surfaces and small handles. A geometric decoder uses planar and upright priors to constrain motion, then selects hinge lines using predicted handle locations, without training a motion regressor. Conversely, a joint part-and-handle predictor supplies additional handle candidates, whose motion classes are refined using containing parts. Each information transfer is applied once, without iterative feedback. On Articulate3D validation, handle guidance raises motion-gated AP from 13.74% to 40.98% at fixed masks and axes. Additional handle candidates raise handle AP from 24.63% to 29.65%; part-based class correction adds 0.98 points, and full context reaches 30.99%. Repeated training, learned-decoder controls and paired visualizations establish the benefits and limitations of combining geometric and semantic evidence for interaction understanding.",
              "link": "https://huggingface.co/papers/2609.25247",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 4
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25247"
    },
    {
      "id": "4fa7c182ec611f3a3a94",
      "title": "Circuit Hypernetworks for Quantum-Augmented Diffusion Language Models",
      "content_text": "Language models can be adapted by changing the computations applied to individual tokens. Quantum circuits offer one such approach, but evaluating wider circuits inside a large model can be computationally demanding. Here we introduce HyperQ, which adds token-conditioned quantum residual branches to a frozen masked-diffusion language model. A quantum residual branch is a module in each transformer block that reads a token's hidden state, emits the coordinates of that token's circuit, executes…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "4fa7c182ec611f3a3a94",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24657",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Circuit Hypernetworks for Quantum-Augmented Diffusion Language Models",
            "item_type": "record",
            "summary": "added: Circuit Hypernetworks for Quantum-Augmented Diffusion Language Models",
            "after": {
              "title": "Circuit Hypernetworks for Quantum-Augmented Diffusion Language Models",
              "summary": "Language models can be adapted by changing the computations applied to individual tokens. Quantum circuits offer one such approach, but evaluating wider circuits inside a large model can be computationally demanding. Here we introduce HyperQ, which adds token-conditioned quantum residual branches to a frozen masked-diffusion language model. A quantum residual branch is a module in each transformer block that reads a token's hidden state, emits the coordinates of that token's circuit, executes it, and adds the measured values back through a residual connection. The backbone remains frozen, and only the added branches are trained. Within each branch, a lightweight circuit hypernetwork emits token-specific rotation angles, coupling strengths, and measurement axes in a shared sparse circuit structure. The required expectation values have an exact classical expression whose evaluation cost grows linearly with the qubit count, enabling circuits from 16 to 64 qubits to be trained within a 1.1-billion-parameter backbone. Across downstream benchmarks, increasing circuit width raises the average score from 47.65 to 54.30. At 64 qubits, HyperQ exceeds the backbone and its low-rank-adapted counterpart by 4.71 and 3.67 points, respectively. HyperQ is fine-tuned on 20,000 prompt-response pairs, compared with 200,000 for the classical baselines. These findings support token-conditioned circuit emission as a tractable architectural approach to quantum-augmented language modelling.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/654a97282d2fcd6bf2851173/eWMzNg7KtFjVZ7Fw9TRUk.png"
              ],
              "organization": {
                "_id": "636e93488ba65db4a0987ab4",
                "name": "Universite-de-Montreal",
                "fullname": "Université de Montréal",
                "avatar": "https://www.gravatar.com/avatar/989f06cbcbc4e7af75108e247e7a9abf?d=retro&size=100"
              },
              "link": "https://huggingface.co/papers/2609.24657",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 26
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24657"
    },
    {
      "id": "8c419c552dc7900e3de1",
      "title": "GAE: Learning a Geometry-Native Latent Space for 3D-Consistent World Generation",
      "content_text": "We present a compact geometry-native latent space as a shared foundation for perception and generation. Visual generators can produce photorealistic frames without preserving a consistent 3D scene. We argue that this is not only a modeling problem but also a representation problem: generators typically evolve appearance-centric latents, while perception models recover geometry in a semantically rich space that encodes cross-view structure. Rather than adding geometry as another output, we…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "8c419c552dc7900e3de1",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.24981",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "GAE: Learning a Geometry-Native Latent Space for 3D-Consistent World Generation",
            "item_type": "record",
            "summary": "added: GAE: Learning a Geometry-Native Latent Space for 3D-Consistent World Generation",
            "after": {
              "title": "GAE: Learning a Geometry-Native Latent Space for 3D-Consistent World Generation",
              "summary": "We present a compact geometry-native latent space as a shared foundation for perception and generation. Visual generators can produce photorealistic frames without preserving a consistent 3D scene. We argue that this is not only a modeling problem but also a representation problem: generators typically evolve appearance-centric latents, while perception models recover geometry in a semantically rich space that encodes cross-view structure. Rather than adding geometry as another output, we reparameterize a geometry foundation model's features into a compact latent space for generation. We realize this shift with the geometry-native autoencoder (GAE), whose latent is jointly decodable to appearance, depth, cameras, and point maps. With this state, a standard conditional flow supports diverse generation tasks. In controlled comparisons that hold the generator and training protocol fixed, replacing the latent with GAE improves both visual quality and independently measured 3D coherence: FVD falls by 12.7% and 23.1% on RealEstate10K and DL3DV, and camera-trajectory error is halved on RealEstate10K. Together, these results show that the latent space is central to geometry-consistent generation and can serve as a shared interface between perception and generation.",
              "mediaUrls": [
                "https://cdn-uploads.huggingface.co/production/uploads/64912d6b2549fd68a774ad3d/FQ6oupZh1V4N8tJZnxOOd.mp4"
              ],
              "organization": {
                "_id": "60e3f7f641ca131919975fe5",
                "name": "TencentARC",
                "fullname": "ARC Lab, Tencent",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/1625552871844-60e272ca6c78a8c122b12127.png"
              },
              "link": "https://huggingface.co/papers/2609.24981",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 43
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.24981"
    },
    {
      "id": "b17b17349e1e03f63c69",
      "title": "Bellman Policy Optimization",
      "content_text": "Reinforcement learning with verifiable rewards (RLVR) improves the reasoning capabilities of large language models (LLMs). We introduce Bellman Policy Optimization (BPO), a critic-free method derived from Policy Mirror Descent (PMD). For autoregressive generation with terminal rewards, BPO uses the Bellman equations to reformulate PMD as a trajectory-level objective. The reformulation avoids estimating state values at intermediate states. We prove that it has the same unique optimal solution as…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "b17b17349e1e03f63c69",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.15987",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Bellman Policy Optimization",
            "item_type": "record",
            "summary": "added: Bellman Policy Optimization",
            "after": {
              "title": "Bellman Policy Optimization",
              "summary": "Reinforcement learning with verifiable rewards (RLVR) improves the reasoning capabilities of large language models (LLMs). We introduce Bellman Policy Optimization (BPO), a critic-free method derived from Policy Mirror Descent (PMD). For autoregressive generation with terminal rewards, BPO uses the Bellman equations to reformulate PMD as a trajectory-level objective. The reformulation avoids estimating state values at intermediate states. We prove that it has the same unique optimal solution as the original PMD objective. We derive the practical BPO loss by approximating this objective. Its mismatch-correction weight is a smoothed ratio of complementary token probabilities. Experiments on mathematical reasoning benchmarks demonstrate the effectiveness of BPO.",
              "organization": {
                "_id": "69e06d3f92a1019939f7e7a0",
                "name": "apodex",
                "fullname": "Apodex",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/68bf96f0ca298a370d66adbb/T4Bt5ELjozQxgkKsHM_gg.jpeg"
              },
              "link": "https://huggingface.co/papers/2609.15987",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 27
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.15987"
    },
    {
      "id": "17cf334726251f8d67b1",
      "title": "From Pattern Recognizers to Personalized Companions: A Survey of Large Language Models in Mental Health",
      "content_text": "The rising global prevalence of mental health conditions, together with longstanding barriers in traditional healthcare, such as limited resources, high cost, stigma, and privacy concerns, has created an urgent need for accessible and scalable support. Large Language Models (LLMs) have emerged as a transformative technology with strong potential to democratize mental health support through advanced natural language understanding and generation. However, the rapidly expanding, fragmented body of…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "17cf334726251f8d67b1",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25186",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "From Pattern Recognizers to Personalized Companions: A Survey of Large Language Models in Mental Health",
            "item_type": "record",
            "summary": "added: From Pattern Recognizers to Personalized Companions: A Survey of Large Language Models in Mental Health",
            "after": {
              "title": "From Pattern Recognizers to Personalized Companions: A Survey of Large Language Models in Mental Health",
              "summary": "The rising global prevalence of mental health conditions, together with longstanding barriers in traditional healthcare, such as limited resources, high cost, stigma, and privacy concerns, has created an urgent need for accessible and scalable support. Large Language Models (LLMs) have emerged as a transformative technology with strong potential to democratize mental health support through advanced natural language understanding and generation. However, the rapidly expanding, fragmented body of work in this area lacks a coherent evolutionary narrative, making it difficult to contextualize current progress and identify future directions. This survey addresses this gap by organizing and analyzing the literature around a central thesis: the role of LLMs in mental health is evolving through three distinct, increasingly sophisticated phases. We trace this trajectory from Phase I, in which LLMs act primarily as passive Information Tools and Pattern Recognizers for assessment; through Phase II, where they function as Empathetic Conversationalists for in-the-moment, stateless interactions; to the current frontier, Phase III, which seeks Longitudinal, Personalized Companions implemented as stateful cognitive agents. To support this framework, we systematically review core technologies, agent architectures (Profile, Memory, Reasoning, and Planning), and the critical infrastructure of datasets and benchmarks, highlighting how their evolution underpins this developmental path. Viewing the field through this developmental lens, we provide a comprehensive synthesis of existing work, an insightful narrative of its trajectory, and a clear roadmap for future innovation in responsible, effective, and human-centered AI for mental healthcare. A curated collection of the resources reviewed in this survey is available at our project repository: https://github.com/Emo-gml/Awesome-Mental-Health-LLMs.",
              "link": "https://huggingface.co/papers/2609.25186",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 16
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25186"
    },
    {
      "id": "8e3c8d37066072026f9f",
      "title": "Ovis-Embedding: Pushing the Frontiers of Universal Omni-Modal Embeddings",
      "content_text": "In this report, we introduce Ovis-Embedding, a state-of-the-art omni-modal embedding family built on native integration of text, image, video, and audio. Instead of assembling separate modality towers, Ovis-Embedding uses a shared multimodal backbone to encode different modalities in a common representation space. Specifically, we make three key advances: (1) native omni-modal initialization: we adopt a pretrained Qwen-omni model as the embedding backbone and adapt it through contrastive…",
      "date_published": "2026-09-23T00:00:00Z",
      "_unlimitedpipe": {
        "event": {
          "schema": "unlimitedpipe.event/1",
          "id": "8e3c8d37066072026f9f",
          "source": "web",
          "type": "change",
          "key": "https://huggingface.co/api/daily_papers#https://huggingface.co/papers/2609.25165",
          "source_url": "https://huggingface.co/api/daily_papers",
          "timestamp": null,
          "observed_at": "2026-09-24T17:26:00Z",
          "data": {
            "change": "added",
            "label": "Ovis-Embedding: Pushing the Frontiers of Universal Omni-Modal Embeddings",
            "item_type": "record",
            "summary": "added: Ovis-Embedding: Pushing the Frontiers of Universal Omni-Modal Embeddings",
            "after": {
              "title": "Ovis-Embedding: Pushing the Frontiers of Universal Omni-Modal Embeddings",
              "summary": "In this report, we introduce Ovis-Embedding, a state-of-the-art omni-modal embedding family built on native integration of text, image, video, and audio. Instead of assembling separate modality towers, Ovis-Embedding uses a shared multimodal backbone to encode different modalities in a common representation space. Specifically, we make three key advances: (1) native omni-modal initialization: we adopt a pretrained Qwen-omni model as the embedding backbone and adapt it through contrastive training with low-rank initialization; (2) data-centric omni-modal training: we construct a broad, high-quality corpus spanning text, images, video, audio, and interleaved multimodal data. To improve data efficiency, we introduce homogeneous-source sampling to form task-consistent batches with informative in-batch negatives; and (3) embedding-specific training and inference optimization: we use focal loss to emphasize hard examples and similarity-based Embedding Distillation to transfer fine-grained similarity structure from complementary experts. At inference time, low-rank feature decomposition enables compact embeddings with flexible dimensionality and minimal performance loss. Empirical evaluations show that the Ovis-Embedding family achieves state-of-the-art performance on MMEB-v3, MMEB-v2, MVEB, MAEB, and RTEB, demonstrating its effectiveness across text, image, video, and audio modalities. These results highlight the potential of unified omni-modal training to overcome modality fragmentation and advance universal embedding models for any-to-any retrieval.",
              "organization": {
                "_id": "6662a91edd706a226d18cc5a",
                "name": "ATH-MaaS",
                "fullname": "ATH-MaaS",
                "avatar": "https://cdn-avatars.huggingface.co/v1/production/uploads/666a9d46a638e57bb7907929/CRc-9MCuH2q9hjTScyTPE.png"
              },
              "link": "https://huggingface.co/papers/2609.25165",
              "published_at": "2026-09-23T00:00:00.000Z",
              "upvotes": 36
            }
          },
          "metadata": {
            "status": 200,
            "final_url": "https://huggingface.co/api/daily_papers",
            "content_type": "application/json",
            "elapsed_ms": 95,
            "not_modified": false,
            "method": "json"
          },
          "provenance": [
            {
              "step": "web",
              "version": "0.3.1"
            },
            {
              "step": "map",
              "version": "0.3.1",
              "args": {
                "assign": [
                  "link=\"https://huggingface.co/papers/\" + paper.id",
                  "published_at=paper.submittedOnDailyAt",
                  "upvotes=paper.upvotes"
                ],
                "drop": [
                  "paper",
                  "publishedAt",
                  "thumbnail",
                  "submittedBy",
                  "isAuthorParticipating",
                  "numComments"
                ]
              }
            },
            {
              "step": "diff",
              "version": "0.3.1",
              "args": {
                "key": "link",
                "namespace": "ai-papers",
                "only": [
                  "added"
                ],
                "emit_initial": true
              }
            }
          ]
        }
      },
      "url": "https://huggingface.co/papers/2609.25165"
    }
  ]
}
