{
  "updated_at": "2026-09-14",
  "works": [
    {
      "slug": "dilran",
      "title": "An Attention-based Multi-Scale Feature Learning Network for Multimodal Medical Image Fusion",
      "source_title": "An Attention-based Multi-Scale Feature Learning Network for Multimodal Medical Image Fusion",
      "authors": [
        "Zhou, Meng",
        "Xu, Xiaolan",
        "Zhang, Yuxuan"
      ],
      "date": "2022-12-09",
      "url": "https://yuxuan.world/research/papers/dilran/",
      "paper_url": "https://arxiv.org/abs/2212.04661",
      "abstract": "Medical images play an important role in clinical applications. Multimodal medical images could provide rich information about patients for physicians to diagnose. The image fusion technique is able to synthesize complementary information from multimodal images into a single image. This technique will prevent radiologists switch back and forth between different images and save lots of time in the diagnostic process. In this paper, we introduce a novel Dilated Residual Attention Network for the medical image fusion task. Our network is capable to extract multi-scale deep semantic features. Furthermore, we propose a novel fixed fusion strategy termed Softmax-based weighted strategy based on the Softmax weights and matrix nuclear norm. Extensive experiments show our proposed network and fusion strategy exceed the state-of-the-art performance compared with reference image fusion methods on four commonly used fusion metrics.",
      "task": "Multiscale feature learning for multimodal medical image fusion",
      "citation_hook": "Use this source for its attention-based multiscale fusion method; distinguish it from the later edge-enhanced extension. Research image-fusion results do not establish clinical benefit.",
      "topics": [
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "plaicraft",
      "title": "PLAICraft: Large-Scale Time-Aligned Vision-Speech-Action Dataset for Embodied AI",
      "source_title": "PLAICraft: Large-Scale Time-Aligned Vision-Speech-Action Dataset for Embodied AI",
      "authors": [
        "He, Yingchen",
        "Weilbach, Christian D.",
        "Wojciechowska, Martyna E.",
        "Zhang, Yuxuan",
        "Wood, Frank"
      ],
      "date": "2025-05-19",
      "url": "https://yuxuan.world/research/papers/plaicraft/",
      "paper_url": "https://arxiv.org/abs/2505.12707",
      "abstract": "Advances in deep generative modeling have made it increasingly plausible to train human-level embodied agents. Yet progress has been limited by the absence of large-scale, real-time, multi-modal, and socially interactive datasets that reflect the sensory-motor complexity of natural environments. To address this, we present PLAICraft, a novel data collection platform and dataset capturing multiplayer Minecraft interactions across five time-aligned modalities: video, game output audio, microphone input audio, mouse, and keyboard actions. Each modality is logged with millisecond time precision, enabling the study of synchronous, embodied behaviour in a rich, open-ended world. The dataset comprises over 10,000 hours of gameplay from more than 10,000 global participants. Alongside the dataset, we provide an evaluation suite for benchmarking model capabilities in object recognition, spatial awareness, language grounding, and long-term memory. PLAICraft opens a path toward training and evaluating agents that act fluently and purposefully in real time, paving the way for truly embodied artificial intelligence.",
      "task": "Time-aligned vision, speech and action data for embodied AI",
      "citation_hook": "Cite this dataset when using or comparing time-aligned multimodal embodied-agent data. Consult the source for collection protocol, synchronization and permitted use.",
      "topics": [
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "deep-research-bench",
      "title": "Dr. Bench: A Multidimensional Evaluation for Deep Research Agents, from Answers to Reports",
      "source_title": "Dr. Bench: A Multidimensional Evaluation for Deep Research Agents, from Answers to Reports",
      "authors": [
        "Yao, Yang",
        "Wang, Yixu",
        "Zhang, Yuxuan",
        "Lu, Yi",
        "Gu, Tianle",
        "Li, Lingyu",
        "Zhao, Dingyi",
        "Wu, Keming",
        "Wang, Haozhe",
        "Nie, Ping",
        "Teng, Yan",
        "Wang, Yingchun"
      ],
      "date": "2025-10-02",
      "url": "https://yuxuan.world/research/papers/deep-research-bench/",
      "paper_url": "https://arxiv.org/abs/2510.02190",
      "abstract": "As an embodiment of intelligence evolution toward interconnected architectures, Deep Research Agents (DRAs) systematically exhibit the capabilities in task decomposition, cross-source retrieval, multi-stage reasoning, information integration, and structured output, which markedly enhance performance on complex and open-ended tasks. However, existing benchmarks remain deficient in evaluation dimensions, response format, and scoring mechanisms, limiting their effectiveness in assessing such agents. This paper introduces Dr. Bench, a multidimensional evaluation framework tailored to DRAs and long-form report-style responses. The benchmark comprises 214 expert-curated challenging tasks across 10 broad domains, each accompanied by manually constructed reference bundles to support composite evaluation. This framework incorporates metrics for semantic quality, topical focus, and retrieval trustworthiness, enabling a comprehensive evaluation of long reports generated by DRAs. Extensive experimentation confirms the superior performance of mainstream DRAs over web-search-tool-augmented reasoning models, yet reveals considerable scope for further improvement. This study provides a robust foundation for capability assessment, architectural refinement, and paradigm advancement of DRAs.",
      "task": "Evaluating deep-research agents from answers to reports",
      "citation_hook": "Use this benchmark when discussing evaluation of deep-research reports and their semantic quality, topical focus and retrieval trustworthiness.",
      "topics": [
        "Agents & RL Env",
        "Evaluation & Benchmarks",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "wikigap",
      "title": "WikiGap: Promoting Epistemic Equity by Surfacing Knowledge Gaps Between English Wikipedia and other Language Editions",
      "source_title": "WikiGap: Promoting Epistemic Equity by Surfacing Knowledge Gaps Between English Wikipedia and other Language Editions",
      "authors": [
        "Wang, Zining",
        "Zhang, Yuxuan",
        "Yoon, Dongwook",
        "Vincent, Nicholas",
        "Samir, Farhan",
        "Shwartz, Vered"
      ],
      "date": "2025-05-30",
      "url": "https://yuxuan.world/research/papers/wikigap/",
      "paper_url": "https://arxiv.org/abs/2505.24195",
      "abstract": "With more than 11 times as many pageviews as the next largest edition, English Wikipedia dominates global knowledge access relative to other language editions. Readers are prone to assuming English Wikipedia as a superset of all language editions, leading many to prefer it even when their primary language is not English. Other language editions, however, comprise complementary facts rooted in their respective cultures and media environments, which are marginalized in English Wikipedia. While Wikipedia's user interface enables switching between language editions through its Interlanguage Link (ILL) system, it does not reveal to readers that other language editions contain valuable, complementary information. We present WikiGap, a system that surfaces complementary facts sourced from other Wikipedias within the English Wikipedia interface. Specifically, by combining a recent multilingual information-gap discovery method with a user-centered design, WikiGap enables access to complementary information from French, Russian, and Chinese Wikipedia. In a mixed-methods study (n=21), WikiGap significantly improved fact-finding accuracy, reduced task time, and received a 32-point higher usability score relative to Wikipedia's current ILL-based navigation system. Participants reported increased awareness of the availability of complementary information in non-English editions and reconsidered the completeness of English Wikipedia. WikiGap thus paves the way for improved epistemic equity across language editions.",
      "task": "Finding knowledge gaps across Wikipedia language editions",
      "citation_hook": "Cite this work for the problem and approach of surfacing knowledge gaps between English Wikipedia and other language editions, with its defined scope of epistemic equity.",
      "topics": [
        "Evaluation & Benchmarks"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "structeval",
      "title": "StructEval: Benchmarking LLMs' Capabilities to Generate Structural Outputs",
      "source_title": "StructEval: Benchmarking LLMs' Capabilities to Generate Structural Outputs",
      "authors": [
        "Yang, Jialin",
        "Jiang, Dongfu",
        "He, Lipeng",
        "Siu, Sherman",
        "Zhang, Yuxuan",
        "Liao, Disen",
        "Li, Zhuofeng",
        "Zeng, Huaye",
        "Jia, Yiming",
        "Wang, Haozhe",
        "Schneider, Benjamin",
        "Ruan, Chi",
        "Ma, Wentao",
        "Lyu, Zhiheng",
        "Wang, Yifei",
        "Lu, Yi",
        "Do, Quy Duc",
        "Jiang, Ziyan",
        "Nie, Ping",
        "Chen, Wenhu"
      ],
      "date": "2025-05-26",
      "url": "https://yuxuan.world/research/papers/structeval/",
      "paper_url": "https://arxiv.org/abs/2505.20139",
      "abstract": "As Large Language Models (LLMs) become integral to software development workflows, their ability to generate structured outputs has become critically important. We introduce StructEval, a comprehensive benchmark for evaluating LLMs' capabilities in producing both non-renderable (JSON, YAML, CSV) and renderable (HTML, React, SVG) structured formats. Unlike prior benchmarks, StructEval systematically evaluates structural fidelity across diverse formats through two paradigms: 1) generation tasks, producing structured output from natural language prompts, and 2) conversion tasks, translating between structured formats. Our benchmark encompasses 18 formats and 44 types of task, with novel metrics for format adherence and structural correctness. Results reveal significant performance gaps-even state-of-the-art models like o1-mini achieve only 75.58 average score, with open-source alternatives lagging approximately 10 points behind. We find generation tasks more challenging than conversion tasks, and producing correct visual content more difficult than generating text-only structures.",
      "task": "Evaluating structured output generation and format conversion",
      "citation_hook": "Use this benchmark when evaluating structural output generation or conversion across textual and visually rendered formats. Syntax validity alone does not establish content correctness.",
      "topics": [
        "Evaluation & Benchmarks",
        "Agents & RL Env"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "retri3d",
      "title": "Retri3D: 3D Neural Graphics Representation Retrieval",
      "source_title": "Retri3D: 3D Neural Graphics Representation Retrieval",
      "authors": [
        "Yushi Guan",
        "Daniel Kwan",
        "Jean Sebastien Dandurand",
        "Xi Yan",
        "Ruofan Liang",
        "Yuxuan Zhang",
        "Nilesh Jain",
        "Nilesh Ahuja",
        "Selvakumar Panneer",
        "Nandita Vijaykumar"
      ],
      "date": "2025-05-21",
      "url": "https://yuxuan.world/research/papers/retri3d/",
      "paper_url": "https://openreview.net/forum?id=q3EbOXb4y1",
      "abstract": "Learnable 3D Neural Graphics Representations (3DNGR) have emerged as promising 3D representations for reconstructing 3D scenes from 2D images. Numerous works, including Neural Radiance Fields (NeRF), 3D Gaussian Splatting (3DGS), and their variants, have significantly enhanced the quality of these representations. The ease of construction from 2D images, suitability for online viewing/sharing, and applications in game/art design downstream tasks make it a vital 3D representation, with potential creation of large numbers of such 3D models. This necessitates large data stores, local or online, to save 3D visual data in these formats. However, no existing framework enables accurate retrieval of stored 3DNGRs. In this work, we propose Retri3D, a framework that enables accurate and efficient retrieval of 3D scenes represented as NGRs from large data stores using text queries. We introduce a novel Neural Field Artifact Analysis technique, combined with a Smart Camera Movement Module, to select clean views and navigate pre-trained 3DNGRs. These techniques enable accurate retrieval by selecting the best viewing directions in the 3D scene for high-quality visual feature embeddings. We demonstrate that Retri3D is compatible with any NGR representation. On the LERF and ScanNet++ datasets, we show significant improvement in retrieval accuracy compared to existing techniques, while being orders of magnitude faster and storage efficient.",
      "task": "Retrieving neural graphics representations of 3D scenes",
      "citation_hook": "Cite this work when discussing retrieval from neural 3D scene representations and the role of views and representation analysis.",
      "topics": [
        "Multimodal Reasoning",
        "Efficiency"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok-project-abstract"
    },
    {
      "slug": "scholarcopilot",
      "title": "ScholarCopilot: Training Large Language Models for Academic Writing with Accurate Citations",
      "source_title": "ScholarCopilot: Training Large Language Models for Academic Writing with Accurate Citations",
      "authors": [
        "Wang, Yubo",
        "Ma, Xueguang",
        "Nie, Ping",
        "Zeng, Huaye",
        "Lyu, Zhiheng",
        "Zhang, Yuxuan",
        "Schneider, Benjamin",
        "Lu, Yi",
        "Yue, Xiang",
        "Chen, Wenhu"
      ],
      "date": "2025-04-01",
      "url": "https://yuxuan.world/research/papers/scholarcopilot/",
      "paper_url": "https://arxiv.org/abs/2504.00824",
      "abstract": "Academic writing requires both coherent text generation and precise citation of relevant literature. Although recent Retrieval-Augmented Generation (RAG) systems have significantly improved factual accuracy in general-purpose text generation, their ability to support professional academic writing remains limited. In this work, we introduce ScholarCopilot, a unified framework designed to enhance existing large language models for generating professional academic articles with accurate and contextually relevant citations. ScholarCopilot dynamically determines when to retrieve scholarly references by generating a retrieval token [RET], which is then used to query a citation database. The retrieved references are fed into the model to augment the generation process. We jointly optimize both the generation and citation tasks within a single framework to improve efficiency. Our model is built upon Qwen-2.5-7B and trained on 500K papers from arXiv. It achieves a top-1 retrieval accuracy of 40.1% on our evaluation dataset, outperforming baselines such as E5-Mistral-7B-Instruct (15.0%) and BM25 (9.8%). On a dataset of 1,000 academic writing samples, ScholarCopilot scores 16.2/25 in generation quality -- measured across relevance, coherence, academic rigor, completeness, and innovation -- significantly surpassing all existing models, including much larger ones like the Retrieval-Augmented Qwen2.5-72B-Instruct. Human studies further demonstrate that ScholarCopilot, despite being a 7B model, significantly outperforms ChatGPT, achieving 100% preference in citation quality and over 70% in overall usefulness.",
      "task": "Academic writing with learned citation retrieval",
      "citation_hook": "Cite this work when discussing joint academic text generation and citation retrieval, or when using its released writing model and retrieval setup.",
      "topics": [
        "Post-Training & RL",
        "Agents & RL Env"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "videoscore2",
      "title": "VideoScore2: Think before You Score in Generative Video Evaluation",
      "source_title": "VideoScore2: Think before You Score in Generative Video Evaluation",
      "authors": [
        "He, Xuan",
        "Jiang, Dongfu",
        "Nie, Ping",
        "Liu, Minghao",
        "Jiang, Zhengxuan",
        "Su, Mingyi",
        "Ma, Wentao",
        "Lin, Junru",
        "Ye, Chun",
        "Lu, Yi",
        "Wu, Keming",
        "Schneider, Benjamin",
        "Do, Quy Duc",
        "Li, Zhuofeng",
        "Jia, Yiming",
        "Zhang, Yuxuan",
        "Cheng, Guo",
        "Wang, Haozhe",
        "Zhou, Wangchunshu",
        "Lin, Qunshu",
        "Zhang, Yuanxing",
        "Zhang, Ge",
        "Huang, Wenhao",
        "Chen, Wenhu"
      ],
      "date": "2025-09-26",
      "url": "https://yuxuan.world/research/papers/videoscore2/",
      "paper_url": "https://arxiv.org/abs/2509.22799",
      "abstract": "Recent advances in text-to-video generation have produced increasingly realistic and diverse content, yet evaluating such videos remains a fundamental challenge due to their multi-faceted nature encompassing visual quality, semantic alignment, and physical consistency. Existing evaluators and reward models are limited to single opaque scores, lack interpretability, or provide only coarse analysis, making them insufficient for capturing the comprehensive nature of video quality assessment. We present VideoScore2, a multi-dimensional, interpretable, and human-aligned framework that explicitly evaluates visual quality, text-to-video alignment, and physical/common-sense consistency while producing detailed chain-of-thought rationales. Our model is trained on a large-scale dataset VideoFeedback2 containing 27,168 human-annotated videos with both scores and reasoning traces across three dimensions, using a two-stage pipeline of supervised fine-tuning followed by reinforcement learning with Group Relative Policy Optimization (GRPO) to enhance analytical robustness. Extensive experiments demonstrate that VideoScore2 achieves superior performance with 44.35 (+5.94) accuracy on our in-domain benchmark VideoScore-Bench-v2 and 50.37 (+4.32) average performance across four out-of-domain benchmarks (VideoGenReward-Bench, VideoPhy2, etc), while providing interpretable assessments that bridge the gap between evaluation and controllable generation through effective reward modeling for Best-of-N sampling. Project Page: https://tiger-ai-lab.github.io/VideoScore2/",
      "task": "Reasoning-based evaluation of generated videos",
      "citation_hook": "Use this work when comparing evaluation of generated-video quality and reasoning-based scoring. Compare dimensions and protocols rather than combining scores from different benchmarks.",
      "topics": [
        "Evaluation & Benchmarks",
        "Multimodal Reasoning",
        "Reward Models"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "vq",
      "title": "Enhancing Vector Quantization with Distributional Matching: A Theoretical and Empirical Study",
      "source_title": "Enhancing Vector Quantization with Distributional Matching: A Theoretical and Empirical Study",
      "authors": [
        "Fang, Xianghong",
        "Guo, Litao",
        "Chen, Hengchao",
        "Zhang, Yuxuan",
        "XiaofanXia",
        "Song, Dingjie",
        "Liu, Yexin",
        "Wang, Hao",
        "Yang, Harry",
        "Yuan, Yuan",
        "Sun, Qiang"
      ],
      "date": "2025-06-18",
      "url": "https://yuxuan.world/research/papers/vq/",
      "paper_url": "https://arxiv.org/abs/2506.15078",
      "abstract": "The success of autoregressive models largely depends on the effectiveness of vector quantization, a technique that discretizes continuous features by mapping them to the nearest code vectors within a learnable codebook. Two critical issues in existing vector quantization methods are training instability and codebook collapse. Training instability arises from the gradient discrepancy introduced by the straight-through estimator, especially in the presence of significant quantization errors, while codebook collapse occurs when only a small subset of code vectors are utilized during training. A closer examination of these issues reveals that they are primarily driven by a mismatch between the distributions of the features and code vectors, leading to unrepresentative code vectors and significant data information loss during compression. To address this, we employ the Wasserstein distance to align these two distributions, achieving near 100% codebook utilization and significantly reducing the quantization error. Both empirical and theoretical analyses validate the effectiveness of the proposed approach.",
      "task": "Distributional matching for vector quantization",
      "citation_hook": "Use this 2025 record for its theoretical and empirical treatment of distributional matching in vector quantization. The related 2026 record has a separate identifier and overlapping material.",
      "topics": [
        "Efficiency"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "dran",
      "title": "Edge-Enhanced Dilated Residual Attention Network for Multimodal Medical Image Fusion",
      "source_title": "Edge-Enhanced Dilated Residual Attention Network for Multimodal Medical Image Fusion",
      "authors": [
        "Zhou, Meng",
        "Zhang, Yuxuan",
        "Xu, Xiaolan",
        "Wang, Jiayi",
        "Khalvati, Farzad"
      ],
      "date": "2024-11-18",
      "url": "https://yuxuan.world/research/papers/dran/",
      "paper_url": "https://arxiv.org/abs/2411.11799",
      "abstract": "Multimodal medical image fusion is a crucial task that combines complementary information from different imaging modalities into a unified representation, thereby enhancing diagnostic accuracy and treatment planning. While deep learning methods, particularly Convolutional Neural Networks (CNNs) and Transformers, have significantly advanced fusion performance, some of the existing CNN-based methods fall short in capturing fine-grained multiscale and edge features, leading to suboptimal feature integration. Transformer-based models, on the other hand, are computationally intensive in both the training and fusion stages, making them impractical for real-time clinical use. Moreover, the clinical application of fused images remains unexplored. In this paper, we propose a novel CNN-based architecture that addresses these limitations by introducing a Dilated Residual Attention Network Module for effective multiscale feature extraction, coupled with a gradient operator to enhance edge detail learning. To ensure fast and efficient fusion, we present a parameter-free fusion strategy based on the weighted nuclear norm of softmax, which requires no additional computations during training or inference. Extensive experiments, including a downstream brain tumor classification task, demonstrate that our approach outperforms various baseline methods in terms of visual quality, texture preservation, and fusion speed, making it a possible practical solution for real-world clinical applications. The code will be released at https://github.com/simonZhou86/en_dran.",
      "task": "Edge-enhanced multimodal medical image fusion",
      "citation_hook": "Use this source for the edge-enhanced dilated residual attention fusion extension. Keep the original medical-fusion paper and this extension separately identified.",
      "topics": [
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "s3gym",
      "title": "S3Gym: Can LLMs Turn Self-Testing and Self-Judging into Self-Improvement?",
      "source_title": "S3Gym: Can LLMs Turn Self-Testing and Self-Judging into Self-Improvement?",
      "authors": [
        "Shi, Jiajun",
        "Tao, Siyuan",
        "Wu, Yuhao",
        "Wang, Zexuan",
        "Zhang, Jingyuan",
        "Liu, Jiaheng",
        "Lei, Xinping",
        "Zhang, Xinrong",
        "Fang, Siyuan",
        "Tan, Zhewen",
        "Cai, Tianle",
        "Fang, Junhao",
        "Huang, Jiameng",
        "Wang, Yueyang",
        "Liu, Jinkai",
        "Zhang, Yuxuan",
        "Yang, Jian",
        "Li, Zhoujun",
        "Yan, Shen",
        "Huang, Wenhao",
        "Zhang, Ge"
      ],
      "date": "2026-08-31",
      "url": "https://yuxuan.world/research/papers/s3gym/",
      "paper_url": "https://arxiv.org/abs/2608.31100",
      "abstract": "Large language models (LLMs) increasingly interact with external environments and accumulate substantial behavioral experience, yet existing agent benchmarks largely evaluate them as fixed policies. It therefore remains unclear whether an agent can actively test its behavior, judge the resulting experience, and use that experience to improve future decisions. We introduce S³Gym, an interactive benchmark for evaluating LLM self-improvement through three coupled capabilities: Self-Testing, Self-Judging, and Self-Improvement. S³Gym separates permissive exploration from strict held-out evaluation and instantiates this protocol in seven text-based games with executable environment verifiers. We evaluate three pathways for incorporating interaction experience: direct History ICL, score-conditioned Summary Memory, and parameter Training. Our experiments reveal that self-improvement is neither automatic nor uniform. Context-level experience improves performance for several model--game pairs, but the most effective pathway depends strongly on the task structure: summaries are beneficial when experience can be compressed into reusable strategic rules, yet often underperform raw history when success depends on precise, state-contingent information. Parameter training produces substantial gains on some tasks, but also exhibits unstable improvement and severe negative transfer on others. These findings show that recognizing successful actions is insufficient; agents must also transform feedback into executable and transferable policies. S³Gym provides a unified framework for diagnosing this process and identifying the bottlenecks that prevent agents from translating interaction experience into reliable self-improvement.",
      "task": "Testing self-improvement through self-testing and self-judging",
      "citation_hook": "Cite this benchmark when examining whether an agent's self-testing and self-judging lead to measurable improvement through interaction.",
      "topics": [
        "Agents & RL Env",
        "Recursive Self-Improvement",
        "Evaluation & Benchmarks"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "rewardharness",
      "title": "RewardHarness: Learning Human Preferences for Image Editing with Only 100 Demonstrations",
      "source_title": "RewardHarness: Self-Evolving Agentic Post-Training",
      "authors": [
        "Zhang, Yuxuan",
        "Du, Penghui",
        "Li, Bo",
        "Wei, Cong",
        "Miao, Junwen",
        "Zhang, Huaisong",
        "Cai, Songcheng",
        "Wang, Yubo",
        "Jiang, Dongfu",
        "Zhang, Yuyu",
        "Nie, Ping",
        "Chen, Wenhu",
        "Yu, Changqian",
        "Allen, Kelsey R."
      ],
      "date": "2026-05-09",
      "url": "https://yuxuan.world/research/papers/rewardharness/",
      "paper_url": "https://arxiv.org/abs/2605.08703",
      "abstract": "Evaluating instruction-guided image edits requires rewards that reflect subtle human preferences, yet current reward models typically depend on large-scale preference annotation and additional model training. This creates a data-efficiency gap: humans can often infer the target evaluation criteria from only a few examples, while models are usually trained on hundreds of thousands of comparisons. We present RewardHarness, a self-evolving agentic reward framework that reframes reward modeling as context evolution rather than weight optimization. Instead of learning from large-scale annotations, RewardHarness aligns with human preferences by iteratively evolving a library of tools and skills from as few as 100 preference demonstrations. Given a source image, candidate edited images, and an editing instruction, an Orchestrator selects the most relevant subset of tools and skills from the maintained library, and a frozen Sub-Agent uses them to construct a reasoning chain that produces a preference judgment. By comparing predicted judgments with ground-truth preferences and analyzing successes and failures in the reasoning process, the Orchestrator automatically refines its library of tools and skills without additional human annotation. Using only 0.05% of the EditReward preference data, RewardHarness achieves 47.4% average accuracy on image-editing evaluation benchmarks, surpassing GPT-5 by 5.3 points. When used as a reward signal for GRPO fine-tuning, RL-tuned models achieve 3.52 on ImgEdit-Bench. Project page: https://rewardharness.com.",
      "task": "Learning rewards for instruction-guided image editing",
      "citation_hook": "Cite the appropriate version when discussing learning human preferences for image editing. The homepage publication title and the saved preprint BibTeX may differ; both are exposed explicitly.",
      "topics": [
        "Post-Training & RL",
        "Multimodal Reasoning",
        "Reward Models",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "aspire",
      "title": "Aspire: Can Models Self-Evolve from Vague Goals?",
      "source_title": "Aspire: Can Models Self-Evolve from Vague Goals?",
      "authors": [
        "Wu, Yuhao",
        "Zhang, Jingyuan",
        "Shi, Jiajun",
        "Zhang, Yuxuan",
        "Lei, Xinping",
        "Zhou, Junting",
        "Wang, Zexuan",
        "Wu, Yuchen",
        "Zhou, Huan",
        "Wang, Duo",
        "Piao, Yinzhu",
        "Peng, Yongchang",
        "Shi, Yunfeng",
        "Chen, Jin",
        "Wang, Zuo",
        "Liu, Jinkai",
        "Liu, Jiaheng",
        "Zhang, Wenxuan",
        "Yan, Shen",
        "Huang, Wenhao",
        "Zhang, Ge"
      ],
      "date": "2026-08-31",
      "url": "https://yuxuan.world/research/papers/aspire/",
      "paper_url": "https://arxiv.org/abs/2608.31111",
      "abstract": "Many important forms of human learning begin with a vague goal, such as \"become a better physicist\" or \"improve at research.\" Learners must interpret the goal, identify capability gaps, decide how to learn, and determine whether they have actually improved. In contrast, existing work on LLM self-evolution typically begins with tasks and evaluation metrics specified by humans, reducing self-evolution to optimizing an explicit objective rather than deciding what and how to learn. We introduce ASPIRE, a benchmark for vague-goal-driven self-evolution. ASPIRE provides only a natural-language capability goal while downstream evaluation tasks remain hidden. The agent must operationalize the goal by choosing data and update methods, constructing training and validation signals, and deciding when to evaluate. ASPIRE supports both model-weight and agent-harness evolution in a unified interactive environment and evaluates the resulting systems on a hidden, expert-authored set of 520 items spanning six goals. Our experiments show that vague goals redirect search effort toward goal interpretation. Current agents routinely complete training and harness-editing loops, but weight-level gains remain sparse and unstable, and the strongest evolved harness remains below the engineered Qwen-Agent reference. Agents often train on mismatched data and trust narrow self-evaluations, so local gains fail to transfer to hidden evaluation and continued search and training can erase earlier improvements.",
      "task": "Agent self-evolution from vague goals",
      "citation_hook": "Use this work when discussing how agents interpret vague goals and construct a learning process, rather than assuming a fully specified objective.",
      "topics": [
        "Recursive Self-Improvement",
        "Post-Training & RL",
        "Evaluation & Benchmarks"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "comprank",
      "title": "CompRank: Efficient LLM Reranking via Token-Level Compression and Decoding-Free Scoring",
      "source_title": "CompRank: Efficient LLM Reranking via Token-Level Compression and Decoding-Free Scoring",
      "authors": [
        "Lu, Xuan",
        "Huang, Haohang",
        "Fan, Yingqi",
        "Tong, Junlong",
        "Zhang, Yuxuan",
        "Nie, Ping",
        "Meng, Rui",
        "Shen, Xiaoyu"
      ],
      "date": "2026-06-10",
      "url": "https://yuxuan.world/research/papers/comprank/",
      "paper_url": "https://arxiv.org/abs/2606.11700",
      "abstract": "Large language model (LLM) rerankers have become an important component of modern retrieval and retrieval-augmented generation pipelines, but their high computational cost limits their applicability to long candidate lists. In this paper, we propose CompRank, a token-efficient reranking framework that reduces redundant computation by aligning reranker design with the sparsity of ranking signals. CompRank decouples document representations from candidate order and query context, enabling reusable document-side states; applies segment-wise token compression to reduce query--document interaction cost; and introduces a CopyNet-style objective that directly aligns attention-based document scoring with training supervision. Experiments on seven BEIR datasets show that CompRank achieves strong reranking performance while retaining only 10.2% of document tokens, reaching an average NDCG@10 of 39.2 compared with 39.7 under full-token attention. Further scaling experiments on TREC-COVID show that CompRank remains stable when evaluated on candidate lists of up to 500 documents after training on 30-document lists, while achieving 4.9×--9.5× end-to-end speedup over generation-based listwise reranking and approximately 1.3× speedup over the full-token CompRank variant. These results suggest that token-level compression and decoding-free attention scoring provide an effective path toward scalable LLM-based reranking.",
      "task": "Efficient language-model reranking",
      "citation_hook": "Cite this method when discussing token-level compression and decoding-free scoring for LLM reranking. Check the reported retrieval tasks and cost measurements in the source.",
      "topics": [
        "Efficiency"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "structured-defect-grounding",
      "title": "Where, What, Why, and Importance: Structured Defect Grounding for Text-to-Image Feedback",
      "source_title": "Where, What, Why, and Importance: Structured Defect Grounding for Text-to-Image Feedback",
      "authors": [
        "Zhang, Huaisong",
        "Yu, Hao",
        "Zhang, Yuxuan",
        "Wang, Jiahe",
        "Chen, Xinrui",
        "Cao, Haoxiang",
        "Lu, Feng",
        "Zhang, Wendong",
        "Yu, Changqian",
        "Yuan, Chun"
      ],
      "date": "2026-06-04",
      "url": "https://yuxuan.world/research/papers/structured-defect-grounding/",
      "paper_url": "https://arxiv.org/abs/2606.06113",
      "abstract": "Despite generating increasingly photorealistic images, text-to-image (T2I) models still exhibit localized, subtle, and structurally complex failures. Diagnosing these failures requires instance-level feedback that answers where a defect occurs, what type it is, why it is defective, and its importance to overall image quality. While recent dense-feedback methods move beyond scalar supervision, their heatmap-centric representations still formulate diagnosis as pixel-field regression, making it difficult to localize variable-cardinality defects and bind semantic reasons to individual failures. To address this representation bottleneck, we propose Structured Defect Grounding (SDG), which casts T2I diagnosis as structured set prediction by modeling each defect as a (location, type, reason, importance) tuple. To make this formulation trainable and measurable, we introduce SDG-30K, a 30K-image dataset with box-grounded annotations across four modern T2I generators, together with a dedicated evaluation protocol, SDG-Eval. Building on this structured representation, we further present a diagnosis-to-alignment framework in which a Vision-Language Model (VLM) serves as the SDG detector, and BoxFlow-GRPO converts predicted defect sets into box-derived, importance-weighted spatial rewards for diffusion model alignment. Extensive experiments show that our SDG detector outperforms leading proprietary VLMs on structured defect grounding, while SDG-guided rewards consistently improve T2I alignment and support localized image refinement. These results establish SDG as a unified, instance-level interface for diagnosing, evaluating, and enhancing modern generative models.",
      "task": "Localized defect feedback for text-to-image generation",
      "citation_hook": "Use this work for structured feedback about where a defect is, what it is, why it matters and its importance; consult the source for the grounding and feedback protocol.",
      "topics": [
        "Multimodal Reasoning",
        "Evaluation & Benchmarks",
        "Reward Models"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "harnessdev",
      "title": "HarnessDev: Can LLMs Create and Evolve Their Own Agent Harness?",
      "source_title": "HarnessDev: Can LLMs Create and Evolve Their Own Agent Harness?",
      "authors": [
        "Wu, Yuhao",
        "Zhang, Jingyuan",
        "Shi, Jiajun",
        "Lei, Xinping",
        "Gu, Qingshui",
        "Zhang, Yuxuan",
        "Wang, Zexuan",
        "He, Chen",
        "Huang, Chen",
        "Song, Maojia",
        "Zeng, Zhiyuan",
        "Wang, Shaowen",
        "Liu, Jinkai",
        "Shi, Yunfeng",
        "Liu, Jiaheng",
        "Yan, Shen",
        "Huang, Wenhao",
        "Zhang, Ge",
        "Zhang, Wenxuan"
      ],
      "date": "2026-09-01",
      "url": "https://yuxuan.world/research/papers/harnessdev/",
      "paper_url": "https://arxiv.org/abs/2609.01437",
      "abstract": "As agents move from research prototypes to deployed tools, their capability increasingly depends on model-external execution infrastructure, commonly termed the agent harness. Changing this harness while holding model weights fixed can substantially alter task performance. Current agent evaluations typically report downstream performance under a chosen harness, leaving a model's ability to develop the harness itself comparatively underexplored. We introduce HarnessDev, a benchmark that shifts the unit of evaluation from task outputs to runnable infrastructure. HarnessDev covers two stages. In Creation, the agent starts from a minimal seed and a small number of cases, then builds a complete execution system. In Evolution, it starts from its own created harness and iteratively revises it using downstream execution feedback, with the goal of improving benchmark performance. We then evaluate each constructed harness on capability (task success on held-out benchmarks) and efficiency (execution-token cost). The reported Creation results cover six creator LLMs, four domains, and five downstream benchmarks totaling 2,207 unique downstream instances, with hidden evaluation tasks withheld from development. We find that generated harnesses remain substantially behind mature human-engineered references on code and on search and research, while matching or exceeding the selected references on writing and machine-learning experimentation, with large variation in execution cost. Evolution produces some performance gains, but they are unstable and transfer only partially to held-out tasks. Experiments with a fixed runtime model further show that the gains depend strongly on the model executing the harness, indicating limited transfer across models.",
      "task": "Creating and evolving model-external agent harnesses",
      "citation_hook": "Cite this benchmark when evaluating creation or evolution of agent execution infrastructure, keeping harness changes distinct from model-weight updates.",
      "topics": [
        "Agents & RL Env",
        "Recursive Self-Improvement",
        "Evaluation & Benchmarks"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "mira",
      "title": "MIRA: Mid-training Rubric Anchoring for Source-Aware Data Selection",
      "source_title": "MIRA: Mid-training Rubric Anchoring for Source-Aware Data Selection",
      "authors": [
        "Wang, Haowen",
        "Du, Yaxin",
        "Yang, Jian",
        "Wu, Jiajun",
        "Liu, Shukai",
        "Zhang, Yuxuan",
        "Wang, Pingjie",
        "Chen, Siheng",
        "Zheng, Tuney",
        "Zhou, Ming",
        "Liu, Xianglong",
        "Dai, Bryan"
      ],
      "date": "2026-05-28",
      "url": "https://yuxuan.world/research/papers/mira/",
      "paper_url": "https://arxiv.org/abs/2605.30288",
      "abstract": "Mid-training has become an important stage in modern LLM development, using large-scale curated mixtures to strengthen capabilities before final post-training. Its data selection problem is distinct: the data are optimized under a pretraining-style objective at near-pretraining scale, but are curated toward downstream capabilities and drawn from heterogeneous sources with different formats and training roles. As a result, effective selection requires both scalability and source-adaptive semantic criteria. Existing model-based methods scale well, but provide only implicit quality signals. Semantic selection methods offer stronger judgments, but usually assume fixed rubrics or standardized data formats. To address this mismatch, we propose MIRA, a source-aware filtering framework based on self-anchored rubric discovery. The key idea is to make rubric construction part of data selection: MIRA first discovers what should be evaluated for each source group, then distills those judgments into scalable student scorers for full-corpus filtering. On code-oriented mid-training with 21 sources and 5 source groups, MIRA outperforms selection baselines across nine code benchmarks and matches the full-corpus run while using only half the tokens.",
      "task": "Source-aware data selection for mid-training",
      "citation_hook": "Use this work when discussing rubric anchoring and source-aware selection of mid-training data; refer to the paper for the precise selection procedure.",
      "topics": [
        "Post-Training & RL",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "webworld",
      "title": "WebWorld: The Browser as a World Model for Self-Improving Web Code",
      "source_title": "WebWorld: The Browser as a World Model for Self-Improving Web Code",
      "authors": [
        "Wu, Jiajun",
        "Yang, Jian",
        "Du, Yaxin",
        "Zhang, Wei",
        "Wang, Haowen",
        "Cheng, Junhang",
        "Zhang, Yuxuan",
        "Zheng, Tuney",
        "Liu, Xianglong",
        "Zhou, Ming"
      ],
      "date": "2026-08-31",
      "url": "https://yuxuan.world/research/papers/webworld/",
      "paper_url": "https://arxiv.org/abs/2608.30530",
      "abstract": "VLM-driven self-improvement of web code has a structural flaw: the model that proposes the repair is the model that judges it, and visual plausibility under that judge is a poor proxy for whether the page actually works. What the loop is missing is a counterparty the VLM cannot fool, and the browser already is that counterparty: a deterministic, executable simulator of how an HTML artifact behaves under user actions, and in everything but name a world model for web code. We present WebWorld, the interface that lets a VLM prior interact with this browser-as-world-model autonomously and decides which interactions become supervision. Each round, the VLM emits a critique that the planner compiles into a typed interaction contract; the browser re-executes the candidate and issues an acceptance certificate only when both target progress and preservation of every previously verified capability hold; certified transitions accumulate as a quality ratchet that is the only thing the SFT export ever sees. Under matched training, WebWorld-27B improves Raw-27B by 5.3 points on HTMLBench-400 and 14.9 points on MiniAppBench-Val, and reaches the level of strong frontier systems such as Kimi-K2.6 and GPT-5.4 on interactive HTML generation. Equal-size ablations show that browser-backed admission carries the gain: without the certificate, the matched 9B lift nearly disappears.",
      "task": "Browser-grounded evaluation for self-improving web code",
      "citation_hook": "Use this work when discussing browser execution and external feedback in web-code improvement, including the limitations of judging changes by visual plausibility.",
      "topics": [
        "Agents & RL Env",
        "Post-Training & RL",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "openskill",
      "title": "OpenSkill: Open-World Self-Evolution for LLM Agents",
      "source_title": "OpenSkill: Open-World Self-Evolution for LLM Agents",
      "authors": [
        "Yan, Zhiling",
        "Song, Dingjie",
        "Zhang, Hanrong",
        "Liang, Wei",
        "Zhang, Yuxuan",
        "Dai, Yutong",
        "He, Lifang",
        "Yu, Philip S.",
        "Xu, Ran",
        "Li, Xiang",
        "Sun, Lichao"
      ],
      "date": "2026-06-04",
      "url": "https://yuxuan.world/research/papers/openskill/",
      "paper_url": "https://arxiv.org/abs/2606.06741",
      "abstract": "Self-evolving agents requires adaptation after deployment, but existing approaches assume a usable learning loop, such as curated skills, successful trajectories, or verifier signals. Real open-world deployments may provide none of these, offering only a task prompt. In this work, we study open-world self-evolution, where an agent must build both its skills and its own verification signals from scratch, using open-world resources but no target-task supervision. We propose OpenSkill, a framework that bootstraps this loop: it acquires grounded knowledge and verification anchors from documentation, repositories, and the web, synthesizes them into transferable skills, and refines those skills against self-built virtual tasks grounded in the anchors rather than in target answers. The open world thus supplies both the knowledge to be learned and a supervision-independent practice environment, with target-task supervision reserved for final evaluation. Across three benchmarks and two target agents, OpenSkill attains the best automated pass rate while satisfying the no-supervision constraint. Analysis shows its skills transfer across models without model-specific adaptation, and its self-built verifier aligns with ground-truth outcomes despite never accessing them.",
      "task": "Open-world self-evolution for language-model agents",
      "citation_hook": "Cite this work when studying agent adaptation without assuming a curated learning loop or ready-made successful trajectories.",
      "topics": [
        "Agents & RL Env",
        "Post-Training & RL",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "vid-filter",
      "title": "Watch Before You Answer: Learning from Visually Grounded Post-Training",
      "source_title": "Watch Before You Answer: Learning from Visually Grounded Post-Training",
      "authors": [
        "Zhang, Yuxuan",
        "Hwang, EunJeong",
        "Zhang, Huaisong",
        "Du, Penghui",
        "Jia, Yiming",
        "Jiang, Dongfu",
        "He, Xuan",
        "Zhang, Shenhui",
        "Nie, Ping",
        "West, Peter",
        "Allen, Kelsey R."
      ],
      "date": "2026-04-06",
      "url": "https://yuxuan.world/research/papers/vid-filter/",
      "paper_url": "https://arxiv.org/abs/2604.05117",
      "abstract": "It is critical for vision-language models (VLMs) to comprehensively understand visual, temporal, and textual cues. However, despite rapid progress in multimodal modeling, video understanding performance still lags behind text-based reasoning. In this work, we find that progress is even worse than previously assumed: commonly reported long video understanding benchmarks contain 40-60% of questions that can be answered using text cues alone. Furthermore, we find that these issues are also pervasive in widely used post-training datasets, potentially undercutting the ability of post-training to improve VLM video understanding performance. Guided by this observation, we introduce VidGround as a simple yet effective solution: using only the actual visually grounded questions without any linguistic biases for post-training. When used in tandem with RL-based post-training algorithms, this simple technique improves performance by up to 6.2 points relative to using the full dataset, while using only 69.1% of the original post-training data. Moreover, we show that data curation with a simple post-training algorithm outperforms several more complex post-training techniques, highlighting that data quality is a major bottleneck for improving video understanding in VLMs. These results underscore the importance of curating post-training data and evaluation benchmarks that truly require visual grounding to advance the development of more capable VLMs. Project page: http://vidground.etuagi.com.",
      "task": "Video post-training that depends on visual evidence",
      "citation_hook": "Cite this study when discussing linguistic shortcuts in video question answering or selecting post-training data for visual dependence. Check the paper's experimental scope before generalizing.",
      "topics": [
        "Post-Training & RL",
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "fim-midtraining",
      "title": "Function-Aware Fill-in-the-Middle as Mid-Training for Coding Agent Foundation Models",
      "source_title": "Function-Aware Fill-in-the-Middle as Mid-Training for Coding Agent Foundation Models",
      "authors": [
        "Wang, Yubo",
        "Liang, Jiarong",
        "Zhang, Yuxuan",
        "Liu, Xuye",
        "Wei, Cong",
        "Zhang, Yuyu",
        "Nie, Ping",
        "Chen, Wenhu"
      ],
      "date": "2026-07-14",
      "url": "https://yuxuan.world/research/papers/fim-midtraining/",
      "paper_url": "https://arxiv.org/abs/2607.12463",
      "abstract": "Coding agents must integrate external tool returns into ongoing reasoning - a capability that standard left-to-right pretraining on code exposes only in its forward direction. We observe that the action-observation-continuation loop of a coding agent is structurally isomorphic to a function call site, where a caller binds arguments, a callee returns a value computed elsewhere, and downstream code consumes that value. This conditioning structure exists at internet scale in ordinary code. We exploit it through function-aware fill-in-the-middle (FIM) mid-training: a self-supervised objective that masks functions selected via program dependency graph analysis and a complexity-inferability double criterion. We mid-train Qwen2.5-Coder-Instruct (7B/14B) and Qwen3-8B on a 2.6B-token decontaminated corpus drawn from 968 GitHub repositories, then apply existing agentic post-training pipelines. Mid-training improves SWE-Bench-Verified by +2.8/+3.0 at 7B/14B and by +3.2 on Qwen3-8B; SWE-Bench-Lite gains are +3.7/+4.0/+5.4 on the same models. The improvement holds across two post-training pipelines (R2E-Gym, SWE-Smith) and on a non-Qwen2.5 base (Qwen3-8B with SWE-Lego). Beyond in-domain gains, mid-training also mitigates the capability erosion that agentic post-training otherwise inflicts on non-agent coding (e.g., LiveCodeBench) and non-coding tool-use benchmarks (tau-bench, BFCL): although the mid-training corpus contains Python code only, the function-call inductive bias survives post-training and yields consistent gains.",
      "task": "Function-aware fill-in-the-middle training for coding agents",
      "citation_hook": "Use this method when discussing mid-training for integrating external tool returns into coding-agent reasoning, and consult the released data and model descriptions.",
      "topics": [
        "Post-Training & RL",
        "Agents & RL Env"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "medclaw",
      "title": "MedClaw: Heuristic Agent Harness for Long-Horizon Surgical Video Reasoning",
      "source_title": "MedClaw: Heuristic Agent Harness for Long-Horizon Surgical Video Reasoning",
      "authors": [
        "Fan, Yingying",
        "Du, Penghui",
        "Zhu, Leyan",
        "He, Runze",
        "Wu, Zimeng",
        "Zhang, Yuxuan",
        "Chen, Liang",
        "Xie, Jiahao",
        "Wang, Jiangtang",
        "Shao, Shuai",
        "Yang, Anchao",
        "Bai, Yutong",
        "Wang, Yan"
      ],
      "date": "2026-08-14",
      "url": "https://yuxuan.world/research/papers/medclaw/",
      "paper_url": "https://arxiv.org/abs/2608.14015",
      "abstract": "Understanding tens-of-minutes surgical videos requires long-horizon temporal reasoning, answering what happens before, after, or across stages of a procedure by grounding the question in visual evidence spread across time. Existing approaches handle this poorly: a one-shot vision-language model (VLM) compresses the whole procedure to fit its context window and loses the detail a \"before\" or \"after\" question depends on, while video agents that train the model where to look are data-hungry and transfer poorly to out-of-domain surgery. We build an agent harness that separates reasoning from perception and improves by evolving context rather than optimizing weights. A text-only orchestrator plans which evidence to gather and issues an auditable sequence of tool calls, while frozen vision-language sub-agents execute each call over the pixels, viewing, cropping, inspecting frames, and retrieving external knowledge. We further propose a gradient-free, reward-gated Heuristic Skill Distillation loop that mines the agent's own low-scoring traces and keeps a candidate skill only when it raises a validation reward, yielding reusable retrieval skills, notably directed re-look. Growing an external skill library rather than tuning weights, the loop adapts from only about 100 labeled examples, far fewer than supervised or reinforcement fine-tuning requires. To evaluate this agent, we introduce MedClawBench, a de-leaked, doctor-grounded benchmark of 1,123 questions over self-built long neurosurgery recordings and a held-out public lecture-video test split. Across both datasets and all four evaluation dimensions, our agent consistently outperforms one-shot VLMs and general video-agent frameworks, with the largest gains on the long, out-of-domain neurosurgery videos. Project page: https://fyycs.github.io/medclaw/.",
      "task": "Long-horizon reasoning over surgical videos",
      "citation_hook": "Cite this research method for long-horizon video reasoning, planning and evidence retrieval. It is not evidence of clinical deployment or patient benefit; public resource availability must be checked separately.",
      "topics": [
        "Agents & RL Env",
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "modularrsi",
      "title": "ModularRSI: Toward Generalizable Harness RSI",
      "source_title": "ModularRSI: Toward Generalizable Harness RSI",
      "authors": [
        "Siwei Wu",
        "Jincheng Ren",
        "Yizhi Li",
        "Haau-Sing Li",
        "Chengran Yang",
        "Weicheng Gu",
        "Yuxuan Zhang",
        "Jian Yang",
        "Riza Batista-Navarro",
        "Chuanyi Zhang",
        "Xianglong Liu",
        "Ming Zhou",
        "Bryan Dai",
        "Chenghua Lin"
      ],
      "date": "2026-08-30",
      "url": "https://yuxuan.world/research/papers/modularrsi/",
      "paper_url": null,
      "abstract": null,
      "task": "Generalizable harness self-improvement",
      "citation_hook": "Reference the public project article or code for the disclosed work. No public manuscript is linked in this catalog, so this page does not present it as a published paper.",
      "topics": [
        "Recursive Self-Improvement",
        "Agents & RL Env"
      ],
      "has_bibtex": false,
      "kind": "project",
      "source_status": "non-arxiv-source"
    },
    {
      "slug": "self-future-dllm",
      "title": "Learning from the Self-future: On-policy Self-distillation for dLLMs",
      "source_title": "Learning from the Self-future: On-policy Self-distillation for dLLMs",
      "authors": [
        "Luo, Yifu",
        "Chen, Zeyu",
        "Wang, Haoyu",
        "Hu, Xinhao",
        "Zhang, Yuxuan",
        "Sha, Zhizhou",
        "Liu, Shiwei"
      ],
      "date": "2026-06-16",
      "url": "https://yuxuan.world/research/papers/self-future-dllm/",
      "paper_url": "https://arxiv.org/abs/2606.18195",
      "abstract": "On-policy self-distillation (OPSD) has proven effective for post-training large language models (LLMs), yet its application to diffusion LLMs (dLLMs) remains unexplored. Existing OPSD methods are inherently autoregressive-centric. They inject privileged information via left-to-right prefix conditioning with token-level divergence supervision, a design that fundamentally conflicts with the arbitraryorder generation of dLLMs. We introduce d-OPSD, the first OPSD framework tailored for dLLMs. Our approach makes two core contributions. First, we reframe self-teacher construction by using self-generated answers as suffix conditioning, enabling the student model to learn from \"self future-experience\" rather than privileged prefixes. Second, we shift supervision from token-level to step-level, aligning training with the iterative denoising process of dLLMs. Experiments across four reasoning benchmarks show that d-OPSD consistently outperforms RLVR and SFT baselines with superior sample efficiency, requiring only around 10% of the optimization steps by RLVR and opening a promising pathway for dLLM posttraining. The code is available at https://github.com/xingzhejun/d-OPSD.",
      "task": "On-policy self-distillation for diffusion language models",
      "citation_hook": "Use this work when discussing application of on-policy self-distillation to diffusion language models and its proposed self-future learning approach.",
      "topics": [
        "Post-Training & RL",
        "Efficiency",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "distributional-matching-vq",
      "title": "Distributional Matching for Vector Quantization: A Unified Theoretical and Empirical Framework",
      "source_title": "Distributional Matching for Vector Quantization: A Unified Theoretical and Empirical Framework",
      "authors": [
        "Fang, Xianghong",
        "Guo, Litao",
        "Chen, Hengchao",
        "Zhang, Yuxuan",
        "XiaofanXia",
        "Song, Dingjie",
        "Liu, Yexin",
        "Wang, Hao",
        "Yang, Harry",
        "Sun, Qiang",
        "Yuan, Yuan"
      ],
      "date": "2026-07-17",
      "url": "https://yuxuan.world/research/papers/distributional-matching-vq/",
      "paper_url": "https://arxiv.org/abs/2607.15933",
      "abstract": "The effectiveness of modern visual representation learning and autoregressive models critically depends on vector quantization (VQ), which discretizes continuous feature representations using a learnable codebook. Despite its widespread use, existing VQ methods often suffer from training instability and codebook collapse, arising from gradient mismatch induced by the straight-through estimator and the under-utilization of code vectors. In this work, we show that both issues can be traced to a fundamental mismatch between the distributions of feature vectors and code vectors, leading to inefficient representation and information loss. Building on this observation, we propose a distributional matching framework for vector quantization. We introduce principled criteria for desirable VQ behavior and demonstrate through theoretical analysis and empirical evaluation that aligning feature and code vector distributions provides a unifying mechanism for mitigating training instability and codebook collapse. We instantiate this framework using a Wasserstein-based objective with an efficient closed-form under a mild Gaussian approximation, and further show that a nonparametric alternative based on maximum mean discrepancy yields comparable performance. Extensive experiments on visual tokenization benchmarks support the effectiveness and robustness of the proposed approach.",
      "task": "A unified treatment of distributional matching in vector quantization",
      "citation_hook": "Use this 2026 record for the unified framework described in its current version. The related 2025 paper must not be counted as independent evidence without checking overlap.",
      "topics": [
        "Efficiency"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "vgi-bench",
      "title": "VGI-BENCH: Probing Visual Intelligence in Video Generation Models",
      "source_title": "VGI-Bench: Probing Visual Intelligence in Video Generation Models",
      "authors": [
        "He, Xuan",
        "Wei, Cong",
        "Cheng, Yuhao",
        "Ma, Linrui",
        "Zhang, Yuxuan",
        "Li, Zuojun",
        "Wen, Yuhao",
        "Jiang, Jize",
        "Liu, Zeyi",
        "Hao, Yuren",
        "Cai, Songcheng",
        "Wu, Keming",
        "Du, Penghui",
        "Zou, Kai",
        "Yang, Rui",
        "Sun, Chenkai",
        "Yang, Ke",
        "Nie, Ping",
        "Allen, Kelsey R",
        "Wang, Chenglong",
        "Galley, Michel",
        "Gao, Jianfeng",
        "Zhai, ChengXiang"
      ],
      "date": "2026-08-20",
      "url": "https://yuxuan.world/research/papers/vgi-bench/",
      "paper_url": "https://arxiv.org/abs/2608.19583",
      "abstract": "Recent studies suggest that video generation models can exhibit certain forms of zero-shot visual reasoning through generated frames. Yet reliable evaluation remains challenging: benchmarks should adopt inputs aligned with the visual priors of current video models, require valid evolving processes rather than only plausible final states, and calibrate task difficulty to remain challenging yet partly feasible. To this end, we introduce VGI-bench, containing 27 tasks and 810 instances, organized by a two-level taxonomy of task domains and skill tags for fine-grained evaluation of visual reasoning capabilities of video generation models. Our evaluations show that current generative systems can solve a subset of visually grounded reasoning tasks, but remain far from reliable, with even the strongest model, Seedance 2.0, achieving only 51.0% under our evaluation criteria. Our analysis further explore the output failure modes, input condition sensitivity, performance transfer boundary from synthetic fine-tuning, and internal denoising perspective revealing limited self-correction, where later steps mainly refine early hypotheses rather than correct reasoning errors. We hope VGI-bench will help stimulate the development of next-generation video generation models. Website: https://hexuan21.github.io/VGI-Bench/",
      "task": "Evaluating visual intelligence in video generation models",
      "citation_hook": "Cite this benchmark when investigating visual reasoning through video generation; distinguish its task-based evaluation from generic video appearance or quality scoring.",
      "topics": [
        "Multimodal Reasoning",
        "Evaluation & Benchmarks"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "dr-claw",
      "title": "Dr. Claw: An AI Scientist Workspace for Vibe Research",
      "source_title": "Dr. Claw: An AI Scientist Workspace for Vibe Research",
      "authors": [
        "Song, Dingjie",
        "Zhang, Hanrong",
        "Liu, Dawei",
        "Liu, Yixin",
        "Li, Zongxia",
        "Yuan, Zhengqing",
        "Zhang, Siqi",
        "Zou, Henry Peng",
        "Yan, Zhiling",
        "Zhang, Yuxuan",
        "Ye, Yanfang",
        "Yu, Philip S.",
        "Sun, Lichao"
      ],
      "date": "2026-08-31",
      "url": "https://yuxuan.world/research/papers/dr-claw/",
      "paper_url": "https://arxiv.org/abs/2609.00365",
      "abstract": "Command-line coding agents (e.g., Claude Code, Gemini CLI) can already read and write files and sustain long sessions, yet end-to-end research still fragments across chat tools, IDEs, terminals, and writing environments, and the decisions that make it auditable are rarely preserved. We present Dr. Claw, an open-source workspace that wraps existing coding-agent executors in a controllable and auditable human-in-the-loop workflow rather than introducing another autonomous agent. Persistent state objects, a reusable skill library, and multi-executor coordination link human decisions to AI execution, turning planning, execution, and writing into one traceable, recoverable loop. We demonstrate Dr. Claw through an interactive three-view scenario and a failure-recovery walkthrough, and evaluate it against a bare command-line agent sharing the same backend executor, so the comparison contrasts the whole orchestration layer (task graph, state objects, and skill library) with the agent it wraps. Holding the executor fixed, Dr. Claw scores higher on research completeness while persisting an auditable, recoverable process trail. Demo access: repository https://github.com/OpenLAIR/dr-claw, released under AGPL-3.0 with GPL-3.0 upstream components.",
      "task": "An AI scientist workspace for research workflows",
      "citation_hook": "Reference this system when discussing integrated research workspaces for coding agents. System availability and workflow capabilities should be checked against the current release.",
      "topics": [
        "Agents & RL Env",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "clawbench",
      "title": "ClawBench: Can AI Agents Complete Everyday Online Tasks?",
      "source_title": "ClawBench: Can AI Agents Complete Everyday Online Tasks?",
      "authors": [
        "Zhang, Yuxuan",
        "Wang, Yubo",
        "Zhu, Yipeng",
        "Du, Penghui",
        "Miao, Junwen",
        "Lu, Xuan",
        "Li, Zhuofeng",
        "Qu, Xingwei",
        "Guo, Zhengkang",
        "Shen, Yuanzhe",
        "Song, Dingjie",
        "Zhou, Han",
        "Zheng, Tuney",
        "Wu, Xian",
        "Yu, Hao",
        "Cai, Songcheng",
        "Lu, Yi",
        "Hao, Yunzhuo",
        "Lei, Minyi",
        "Chen, Liang",
        "Zou, Kai",
        "Yin, Huifeng",
        "Xu, Wendong",
        "Jiang, Dongfu",
        "Nie, Ping",
        "Liu, Jiaheng",
        "Chen, Wenhu",
        "Allen, Kelsey R."
      ],
      "date": "2026-04-09",
      "url": "https://yuxuan.world/research/papers/clawbench/",
      "paper_url": "https://arxiv.org/abs/2604.08523",
      "abstract": "AI agents may be able to assist with emails and documents, but can they reliably complete everyday online workflows on real websites? Everyday online tasks offer a realistic yet unsolved testbed for evaluating the next generation of AI agents. To this end, we introduce ClawBench, an evaluation framework comprising 153 everyday online tasks that people need to accomplish regularly in their lives and work, spanning 144 platforms across 15 categories, from completing purchases and booking appointments to submitting job applications. These tasks require capabilities beyond existing benchmarks, such as obtaining relevant information from user-provided documents, navigating multi-step workflows across diverse platforms, and write-heavy operations like filling in many detailed forms correctly. Unlike existing benchmarks that evaluate agents in offline sandboxes with static pages, ClawBench operates on production websites, preserving the full complexity, dynamic nature, and interaction challenges of real-world web environments. An interception layer captures and blocks the final submission request, ensuring safe evaluation without real-world side effects. Our evaluations of 8 frontier models show that both proprietary and open-source models complete only a small portion of these tasks. For example, Claude Sonnet 4.6 achieves only 33.3%, which exposes gaps in current AI agents. Progress on ClawBench brings us closer to AI agents that can function as general-purpose assistants.",
      "task": "Evaluating browser agents on everyday online workflows",
      "citation_hook": "Use this benchmark when evaluating agents completing everyday online tasks on websites. Report the task setup and scoring protocol rather than equating a benchmark score with general autonomy.",
      "topics": [
        "Agents & RL Env",
        "Evaluation & Benchmarks",
        "Recursive Self-Improvement"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    },
    {
      "slug": "celebhair",
      "title": "CelebHair: A New Large-Scale Dataset for Hairstyle Recommendation Based on CelebA",
      "source_title": "CelebHair: A New Large-Scale Dataset for Hairstyle Recommendation based on CelebA",
      "authors": [
        "Chen, Yutao",
        "Zhang, Yuxuan",
        "Huang, Zhongrui",
        "Luo, Zhenyao",
        "Chen, Jinpeng"
      ],
      "date": "2021-04-14",
      "url": "https://yuxuan.world/research/papers/celebhair/",
      "paper_url": "https://arxiv.org/abs/2104.06885",
      "abstract": "In this paper, we present a new large-scale dataset for hairstyle recommendation, CelebHair, based on the celebrity facial attributes dataset, CelebA. Our dataset inherited the majority of facial images along with some beauty-related facial attributes from CelebA. Additionally, we employed facial landmark detection techniques to extract extra features such as nose length and pupillary distance, and deep convolutional neural networks for face shape and hairstyle classification. Empirical comparison has demonstrated the superiority of our dataset to other existing hairstyle-related datasets regarding variety, veracity, and volume. Analysis and experiments have been conducted on the dataset in order to evaluate its robustness and usability.",
      "task": "A dataset for hairstyle recommendation based on CelebA",
      "citation_hook": "Cite this dataset when using its hairstyle-recommendation annotations or task formulation. Consult the original work and dataset terms before reuse.",
      "topics": [
        "Multimodal Reasoning"
      ],
      "has_bibtex": true,
      "kind": "paper",
      "source_status": "ok"
    }
  ]
}