{
  "schemaVersion": 1,
  "provenance": {
    "sourceCommit": "4dec0787977dcca6fb4ab8235f498d7c72be1837",
    "sourceFiles": [
      "research.html",
      "index.html"
    ],
    "verificationStatus": "Paper content and citation exports checked against the primary sources recorded per paper; conference display labels remain imported unless independently verified.",
    "missingValuesPolicy": "Omit unavailable fields; do not infer abstracts, DOI, publication dates, results or citation statistics.",
    "primaryImportSource": "research.html visible paper cards",
    "reconciliation": "Homepage links and legacy author metadata aligned to the Publications body with user approval.",
    "contentVerifiedOn": "2026-09-12",
    "citationPolicy": "Prefer verified proceedings or journal records. Citation titles and author lists follow the published version when they differ from the expanded preprint. Conference display years identify the event; BibTeX years follow the publisher. Retain preprint citations when no formal record was verified.",
    "evidencePolicy": "Selected results are checked against the linked full-text revision, with explicit experimental conditions and table, section or PDF-page locators."
  },
  "author": {
    "name": "Yuhang Zang",
    "givenName": "Yuhang",
    "familyName": "Zang",
    "alternateName": [
      "臧宇航",
      "Zang Yuhang"
    ],
    "jobTitle": "Researcher",
    "description": "Young researcher at Shanghai AI Laboratory focusing on post-training for multimodal LLMs and vision-language pre-training. Area Chair for NeurIPS, ICLR, CVPR, AAAI, and COLM.",
    "image": "https://yuhangzang.github.io/imgs/yuhang_zang.jpg",
    "affiliation": {
      "name": "Shanghai AI Laboratory",
      "url": "https://www.shlab.org.cn/"
    },
    "alumniOf": [
      {
        "name": "Nanyang Technological University",
        "url": "https://www.ntu.edu.sg/"
      },
      {
        "name": "University of Electronic Science and Technology of China",
        "alternateName": "UESTC",
        "url": "https://en.uestc.edu.cn/"
      }
    ],
    "identifiers": {
      "orcid": "0000-0003-1110-5062",
      "googleScholar": "hW23VKIAAAAJ",
      "dblp": "230/4433",
      "openalex": "A5005200501",
      "semanticScholar": "12862495",
      "github": "yuhangzang",
      "huggingface": "yuhangzang",
      "twitter": "yuhangzang",
      "linkedin": "yuhang-zang"
    },
    "knowsAbout": [
      "Multimodal Large Language Models",
      "Vision-Language Models",
      "Reinforcement Learning from Human Feedback",
      "Machine Learning",
      "Computer Vision",
      "Artificial Intelligence"
    ]
  },
  "homepage": {
    "defaultVisibleCount": 5,
    "selectedPaperIds": [
      "arxiv:2603.12252",
      "arxiv:2505.03318",
      "arxiv:2503.01785",
      "arxiv:2501.12368",
      "arxiv:2502.05173",
      "arxiv:2404.06512",
      "arxiv:2407.01523",
      "arxiv:2401.15914",
      "arxiv:2305.18279",
      "arxiv:2210.07225",
      "arxiv:2305.14813",
      "arxiv:2203.11876",
      "arxiv:2102.12867"
    ]
  },
  "papers": [
    {
      "id": "arxiv:2606.19338",
      "title": "Beyond the Current Observation: Evaluating Multimodal Large Language Models in Controllable Non-Markov Games",
      "authors": [
        {
          "name": "Shengyuan Ding"
        },
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Xinyu Fang"
        },
        {
          "name": "Haodong Duan",
          "corresponding": true
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "EMNLP",
        "citationText": "Empirical Methods in Natural Language Processing (EMNLP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-06-17",
        "arxivLastUpdated": "2026-06-17"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2606.19338",
        "googleScholar": "hW23VKIAAAAJ:yB1At4FlUx8C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2606.19338"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:yB1At4FlUx8C"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/RNGBench",
          "repository": "InternLM/RNGBench"
        }
      ],
      "display": {
        "new": true,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 244,
        "order": 1
      },
      "shortName": "RNG-Bench",
      "keywords": [
        "RNG-Bench",
        "Non-Markov games",
        "Multimodal agents",
        "Visual memory",
        "Hidden-state reconstruction",
        "Memory Gap",
        "Closed-loop interaction",
        "Spatial mapping",
        "Long-context evaluation"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260619338",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2606.19338",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2606.19338"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2606.19338v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2606.19338v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2606.19338v1"
          }
        },
        "abstract": {
          "text": "Deploying multimodal foundation models as closed-loop policies increasingly requires conditioning actions on observations that are no longer visible. However, existing benchmarks either expose the full state, conflate hidden-state reconstruction with other agent skills, or test recall only after an episode has ended. We introduce RNG-Bench (Reconstructive Non-Markov Games), a benchmark suite designed to isolate a base model's ability to reconstruct past observations and act on them during multi-step interaction. RNG-Bench includes two complementary games: Matching Pairs, where card identities briefly revealed at specific locations must later be recalled, and 3D Maze, where egocentric views must be integrated into a spatial map. Both games are evaluated under a unified harness with three controlled difficulty axes: grid size, visual pattern, and observation modality. The benchmark further introduces a head-to-head duel protocol to control for instance-level variance and a Memory Gap metric that disentangles forgetting from poor action selection. The hardest configurations require contexts of roughly 128K tokens and 350 image inputs per episode, and remain far from saturated by frontier MLLMs. Memory Gap analysis shows that most residual errors stem from forgetting earlier observations rather than from suboptimal decision making. Finally, fine-tuning Qwen3.5-9B on optimal-policy rollouts and filtered model demonstrations improves performance on RNG-Bench and transfers to existing benchmarks without degrading general multimodal capability.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "RNG-Bench separates memory failures from action-selection errors in interactive visual games, finding that forgotten observations account for most residual errors in the evaluated models.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Closed-loop agents must act on observations that are no longer visible. RNG-Bench isolates this requirement using Matching Pairs and egocentric 3D Maze games with controlled difficulty and access to past observations.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Controls grid size, visual pattern and observation modality within one evaluation harness.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Introduces head-to-head duels and the Memory Gap metric to distinguish forgetting from poor decisions.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On 10×10 Matching Pairs, GPT-5.4 scores 62.3% and Gemini-3.1-Pro 50.0%; on the 13×13 3D Maze, their aggregate game scores are 30.5 and 49.7, respectively. The ranking changes with the game and metric, so the benchmark does not support a single model winning every hidden-state task.",
            "fragment": "S4.T2",
            "locator": "Table 2 · single-player evaluation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "For Qwen3.5-397B on Matching Pairs, the score falls from 100% with text input to 75.8% with ASCII images and 38.3% with noisy images. This controlled comparison isolates a substantial visual-input bottleneck within the same game.",
            "fragment": "S4.T4",
            "locator": "Table 4 · input-modality ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Full-state or post-episode evaluation",
              "design": "Exposes the current state or tests recall after interaction, rather than isolating the use of hidden past observations."
            },
            {
              "method": "RNG-Bench",
              "design": "Requires memory during closed-loop games and separates forgetting from action selection with Memory Gap."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2606.03890",
      "title": "OVO-S-Bench: A Hierarchical Benchmark for Streaming Spatial Intelligence in Multimodal LLMs",
      "authors": [
        {
          "name": "Yifei Li"
        },
        {
          "name": "Pengyiang Liu"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zhongyue Shi"
        },
        {
          "name": "Qi Fu"
        },
        {
          "name": "Hongye Hao"
        },
        {
          "name": "Jiwen Lu"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "EMNLP",
        "citationText": "Empirical Methods in Natural Language Processing (EMNLP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-06-02",
        "arxivLastUpdated": "2026-08-26"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2606.03890",
        "googleScholar": "hW23VKIAAAAJ:XD-gHx7UXLsC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2606.03890"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:XD-gHx7UXLsC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/OVO-S-Bench",
          "repository": "InternLM/OVO-S-Bench"
        }
      ],
      "display": {
        "new": true,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 266,
        "order": 2
      },
      "shortName": "OVO-S-Bench",
      "keywords": [
        "OVO-S-Bench",
        "Streaming video understanding",
        "Spatial intelligence",
        "Egocentric video",
        "Allocentric mapping",
        "Prefix-only evaluation",
        "Spatiotemporal reasoning",
        "Multimodal benchmarks"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260603890",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2606.03890",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2606.03890"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2606.03890v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2606.03890v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2606.03890v2"
          }
        },
        "abstract": {
          "text": "Multimodal agents in robotics, AR, and autonomous driving must reason about places and layouts from continuous egocentric streams, often using evidence outside the current view. Existing benchmarks either evaluate offline over full videos or target events rather than spatial structure. We introduce OVO-S-Bench, a fully human-annotated benchmark for streaming spatial intelligence, comprising 1,680 questions over 348 source videos. Annotation involves 12 trained annotators (each also serving as a blind cross-reviewer) across roughly 804 person-hours of multi-round quality assurance. Each question carries a query timestamp and an evidence interval, and at evaluation, the model sees only the prefix preceding the query. Questions span four levels of increasing abstraction: instantaneous egocentric perception, spatiotemporal context tracking, generative spatial reasoning, and allocentric spatial mapping. Across 38 proprietary and open-source MLLMs, Gemini-3.1-Pro trails human experts evaluated under the same prefix-access protocol by 33 points (59.2 vs. 92.2), with allocentric spatial mapping as the dominant bottleneck. Notably, streaming and spatially fine-tuned MLLMs underperform their own backbones. We further find that chain-of-thought reasoning amplifies spatial errors when ungrounded in the stream. By exposing these limitations, OVO-S-Bench establishes a demanding testbed for next-generation streaming spatial MLLMs.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "OVO-S-Bench shows that streaming spatial reasoning remains difficult even for strong multimodal models, with allocentric mapping a major bottleneck under prefix-only video access.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Offline video evaluation can expose evidence unavailable when a real-time query arrives. OVO-S-Bench evaluates spatial questions using only the video prefix preceding each query timestamp.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Provides 1,680 human-annotated questions over 348 videos, with query timestamps and evidence intervals.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Organizes questions into perception, context tracking, generative spatial reasoning and allocentric mapping.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Gemini-3.1-Pro reaches 59.19 overall, compared with 92.20 for humans with prefix access and 86.61 for humans under streaming observation. These human protocols provide distinct reference points and should not be conflated.",
            "fragment": "S4.T2",
            "locator": "Table 2 · spatial understanding benchmark · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "On the L4 allocentric category, Gemini-3.1-Pro scores 54.90 and GPT-5.4 40.53; their overall scores are 59.19 and 50.89. The category-level results expose a world-centered reasoning gap beyond aggregate performance.",
            "fragment": "S4.T2",
            "locator": "Table 2 · allocentric spatial reasoning · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Offline spatial video evaluation",
              "design": "Can provide video evidence that would not yet exist at the time of a real-time query."
            },
            {
              "method": "OVO-S-Bench",
              "design": "Restricts every query to its preceding video prefix and tests four spatial capability levels."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2602.08439",
      "title": "Demo-ICL: In-Context Learning for Procedural Video Knowledge Acquisition",
      "authors": [
        {
          "name": "Yuhao Dong"
        },
        {
          "name": "Shulin Tian"
        },
        {
          "name": "Shuai Liu"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Ziwei Liu",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "EMNLP",
        "citationText": "Findings of Empirical Methods in Natural Language Processing (Findings of EMNLP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-02-09",
        "arxivLastUpdated": "2026-02-09"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2602.08439",
        "googleScholar": "hW23VKIAAAAJ:tKAzc9rXhukC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2602.08439"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:tKAzc9rXhukC"
        },
        {
          "type": "code",
          "url": "https://github.com/dongyh20/Demo-ICL",
          "repository": "dongyh20/Demo-ICL"
        }
      ],
      "display": {
        "new": true,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 288,
        "order": 3
      },
      "shortName": "Demo-ICL",
      "keywords": [
        "Demo-ICL",
        "Video in-context learning",
        "Procedural knowledge",
        "Instructional videos",
        "Demonstration learning",
        "Video question answering",
        "Direct preference optimization",
        "Multimodal adaptation"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260208439",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2602.08439",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2602.08439"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2602.08439v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2602.08439v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2602.08439v1"
          }
        },
        "abstract": {
          "text": "Despite the growing video understanding capabilities of recent Multimodal Large Language Models (MLLMs), existing video benchmarks primarily assess understanding based on models' static, internal knowledge, rather than their ability to learn and adapt from dynamic, novel contexts from few examples. To bridge this gap, we present Demo-driven Video In-Context Learning, a novel task focused on learning from in-context demonstrations to answer questions about the target videos. Alongside this, we propose Demo-ICL-Bench, a challenging benchmark designed to evaluate demo-driven video in-context learning capabilities. Demo-ICL-Bench is constructed from 1200 instructional YouTube videos with associated questions, from which two types of demonstrations are derived: (i) summarizing video subtitles for text demonstration; and (ii) corresponding instructional videos as video demonstrations. To effectively tackle this new challenge, we develop Demo-ICL, an MLLM with a two-stage training strategy: video-supervised fine-tuning and information-assisted direct preference optimization, jointly enhancing the model's ability to learn from in-context examples. Extensive experiments with state-of-the-art MLLMs confirm the difficulty of Demo-ICL-Bench, demonstrate the effectiveness of Demo-ICL, and thereby unveil future research directions.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "Demo-ICL trains video-language models to acquire procedural knowledge from in-context demonstrations, rather than relying only on knowledge already stored in their parameters.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Existing video benchmarks mainly test static knowledge. Demo-ICL-Bench asks whether a model can learn from text or video demonstrations and apply that knowledge to questions about a target video.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Builds a demonstration-driven benchmark from 1,200 instructional YouTube videos.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Combines video-supervised fine-tuning with information-assisted direct preference optimization.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With the 7B Ola-VideoBase backbone and 32-frame input, Demo-ICL raises the reported task average from 24.8 to 33.1. Text-demonstration performance rises from 31.4 to 43.4, and video-demonstration performance from 25.0 to 32.0.",
            "fragment": "S3.T1",
            "locator": "Table 1 · demonstration-conditioned evaluation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "The task average is 26.4 without instructional-video training, 29.8 after Demo-ICL SFT, 30.7 with vanilla DPO, and 33.1 with the final method. The ablation separates the contributions of instructional data and preference optimization.",
            "fragment": "S4.T4",
            "locator": "Table 4 · training ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Static-knowledge video question answering",
              "design": "Primarily tests information already available in model parameters and the target video."
            },
            {
              "method": "Demo-ICL",
              "design": "Supplies demonstrations as context and trains the model to acquire and apply procedural information from them."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2605.27955",
      "title": "Skill-as-Pseudocode: Refactoring Skill Libraries to Pseudocode for LLM Agents",
      "authors": [
        {
          "name": "Xinze Li"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yixin Cao"
        },
        {
          "name": "Aixin Sun"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "EMNLP",
        "citationText": "Findings of Empirical Methods in Natural Language Processing (Findings of EMNLP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-05-27",
        "arxivLastUpdated": "2026-08-31"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2605.27955",
        "googleScholar": "hW23VKIAAAAJ:uJ-U7cs_P_0C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2605.27955"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:uJ-U7cs_P_0C"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/Skill-as-Pseudocode",
          "repository": "InternLM/Skill-as-Pseudocode"
        }
      ],
      "display": {
        "new": true,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 310,
        "order": 4
      },
      "shortName": "Skill-as-Pseudocode",
      "keywords": [
        "Skill-as-Pseudocode",
        "LLM agents",
        "Agent skill libraries",
        "Typed contracts",
        "Pseudocode",
        "Action templates",
        "ALFWorld",
        "Token efficiency",
        "Procedural knowledge"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260527955",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2605.27955",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2605.27955"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2605.27955v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2605.27955v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2605.27955v2"
          }
        },
        "abstract": {
          "text": "Markdown skill libraries for LLM agents ship as free-form prose, forcing the agent to re-derive both the input schema and the concrete invocation syntax on every retrieval. This produces a \"confused → re-retrieve → still confused\" loop: the agent issues a partially-correct action, receives uninformative feedback, and re-retrieves the same prose. We propose Skill-as-Pseudocode (SaP), an automatic conversion of markdown skill libraries into typed pseudocode with deterministic quality control. From each cluster of similar procedural passages, SaP extracts a typed contract and filters it through a four-check deterministic verifier (coverage, binding, replacement, risk). Promoted contracts are inlined into a rewritten skill skeleton alongside restored action templates, giving the agent two complementary signals: a typed signature for what a skill does and a concrete template for how to invoke it. On the ALFWorld unseen split (134 games, gpt-4o-mini, three seeds), SaP wins 82/402 paired games versus 47/402 for the Graph-of-Skills (GoS) baseline (pooled McNemar p = 8.2 × 10⁻⁵), at -22.8 ± 6.4% input tokens and -14.5 ± 4.1% LLM calls per game. A bundle-component ablation attributes the gain to the pairing of typed contracts with concrete action templates: the contract alone falls below the prose baseline.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Skill-as-Pseudocode improves agent use of procedural libraries by pairing typed contracts with concrete action templates; contracts alone do not deliver the same benefit.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Free-form skill descriptions force agents to repeatedly infer input schemas and invocation syntax. SaP converts procedural passages into typed pseudocode and restores executable action templates in the rewritten library.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Extracts typed contracts from clusters of similar procedures and verifies coverage, binding, replacement and risk.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses component ablations to separate the effects of contracts and action templates.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On 134 unseen ALFWorld games with three seeds, GPT-4o-mini and a 30-step limit, SaP succeeds in 82/402 runs (20.4%) versus 47/402 (11.7%) for GoS. Mean input tokens per game decrease from 247.8k to 191.4k and calls from 39.7 to 33.9.",
            "fragment": "S5.T1",
            "locator": "Table 1 · ALFWorld, 402 paired runs · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "In the separate single-seed ablation on 134 games, the full SaP bundle yields 30 successes, compared with 25 for templates alone, 18 for length-matched prose, 16 for raw GoS, and 13 for contracts alone. This supports the combined representation rather than attributing all gains to shorter text.",
            "fragment": "S5.T4",
            "locator": "Table 4 · skill-bundle ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Free-form prose skills",
              "design": "Leaves agents to infer input contracts and concrete invocation syntax from procedural descriptions."
            },
            {
              "method": "Skill-as-Pseudocode",
              "design": "Pairs verified typed contracts with restored action templates; the study finds contracts alone insufficient."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2601.16690",
      "title": "EMemBench: Interactive Benchmarking of Episodic Memory for VLM Agents",
      "authors": [
        {
          "name": "Xinze Li"
        },
        {
          "name": "Ziyue Zhu"
        },
        {
          "name": "Siyuan Liu"
        },
        {
          "name": "Yubo Ma"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yixin Cao"
        },
        {
          "name": "Aixin Sun"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "EMNLP",
        "citationText": "Findings of Empirical Methods in Natural Language Processing (Findings of EMNLP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-01-23",
        "arxivLastUpdated": "2026-08-31"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2601.16690",
        "googleScholar": "hW23VKIAAAAJ:PR6Y55bgFSsC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2601.16690"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:PR6Y55bgFSsC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/EMemBench",
          "repository": "InternLM/EMemBench"
        }
      ],
      "display": {
        "new": true,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 332,
        "order": 5
      },
      "shortName": "EMemBench",
      "keywords": [
        "EMemBench",
        "Episodic memory",
        "Interactive evaluation",
        "VLM agents",
        "Long-term memory",
        "Memory induction",
        "Trajectory-grounded questions",
        "Spatial memory"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260116690",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2601.16690",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2601.16690"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2601.16690v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2601.16690v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2601.16690v2"
          }
        },
        "abstract": {
          "text": "We introduce EMemBench, a programmatic benchmark generator for evaluating long-term episodic memory of agents through interactive games. Rather than using a fixed set of questions, EMemBench generates questions from environment-grounded trajectories, covering both text-only and visual game environments. Each template computes verifiable ground truth from underlying game signals, with controlled answerability and balanced coverage over memory skills: single/multi-hop recall, induction, temporal, spatial, logical, and adversarial. We evaluate memory agents with strong LMs/VLMs as backbones, using in-context prompting as baselines. Across 15 text games and multiple visual seeds, results are far from saturated: induction and spatial reasoning are persistent bottlenecks, especially in visual settings. Persistent memory yields clear gains for open backbones on text games, but improvements are less consistent for VLM agents, suggesting that visually grounded episodic memory remains an open challenge. A human study further contextualizes the difficulty and interpretability of EMemBench.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "EMemBench finds that induction and spatial reasoning remain persistent weaknesses of episodic-memory agents, especially when memories must be grounded in visual environments.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "A fixed question set cannot fully probe interactive episodic memory. EMemBench generates verifiable questions from game trajectories across text-only and visual environments.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Controls answerability and coverage across recall, induction, temporal, spatial, logical and adversarial memory skills.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Compares persistent-memory agents with in-context baselines and includes a human study.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "For text-based Qwen2.5-32B, A-MEM raises overall accuracy from the in-context baseline of 40.8 to 51.4 and F1 from 39.9 to 48.1. For visual GPT-5.1, accuracy instead changes from 43.8 to 42.1 and F1 from 39.0 to 35.5, showing that external memory is not uniformly beneficial.",
            "fragment": "S3.T3",
            "locator": "Table 3 · episodic-memory evaluation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Human open-book accuracy is 65.6, versus 23.2 in the closed-book condition; corresponding F1 scores are 63.8 and 19.8. These results quantify the effect of access to past experience under the benchmark protocol.",
            "fragment": "S4.T4",
            "locator": "Table 4 · human reference conditions · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Fixed memory questions",
              "design": "Uses a predetermined question set with limited coverage of interactive experience."
            },
            {
              "method": "EMemBench",
              "design": "Generates verifiable questions grounded in game trajectories across text and visual environments."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2505.14677",
      "title": "Visionary-R1: Mitigating Shortcuts in Visual Reasoning with Reinforcement Learning",
      "authors": [
        {
          "name": "Jiaer Xia"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Peng Gao"
        },
        {
          "name": "Sharon Li"
        },
        {
          "name": "Kaiyang Zhou"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "TMLR",
        "citationText": "Transactions on Machine Learning Research (TMLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-05-20",
        "arxivLastUpdated": "2025-10-26"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2505.14677",
        "googleScholar": "hW23VKIAAAAJ:70eg2SAEIzsC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2505.14677"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:70eg2SAEIzsC"
        },
        {
          "type": "code",
          "url": "https://github.com/maifoundations/Visionary-R1",
          "repository": "maifoundations/Visionary-R1"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 354,
        "order": 6
      },
      "shortName": "Visionary-R1",
      "keywords": [
        "Visionary-R1",
        "Visual reasoning",
        "Shortcut learning",
        "Reinforcement learning",
        "GRPO",
        "Caption–reason–answer",
        "CoT-free supervision",
        "Out-of-distribution generalization"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv250514677",
        "year": 2026,
        "journal": "Transactions on Machine Learning Research",
        "source": {
          "label": "TMLR published citation record",
          "url": "https://jmlr.org/tmlr/papers/bib/JWkZXBgh5a.bib"
        },
        "status": "published",
        "pdfURL": "https://openreview.net/pdf?id=JWkZXBgh5a",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2505.14677v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2505.14677v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2505.14677v3"
          }
        },
        "abstract": {
          "text": "Learning general-purpose reasoning capabilities has long been a challenging problem in AI. Recent research in large language models (LLMs), such as DeepSeek-R1, has shown that reinforcement learning techniques like GRPO can enable pre-trained LLMs to develop reasoning capabilities using simple question-answer pairs. In this paper, we aim to train visual language models (VLMs) to perform reasoning on image data through reinforcement learning and visual question-answer pairs, without any explicit chain-of-thought (CoT) supervision. Our findings indicate that simply applying reinforcement learning to a VLM -- by prompting the model to produce a reasoning chain before providing an answer -- can lead the model to develop shortcuts from easy questions, thereby reducing its ability to generalize across unseen data distributions. We argue that the key to mitigating shortcut learning is to encourage the model to interpret images prior to reasoning. Therefore, we train the model to adhere to a caption-reason-answer output format: initially generating a detailed caption for an image, followed by constructing an extensive reasoning chain. When trained on 273K CoT-free visual question-answer pairs and using only reinforcement learning, our model, named Visionary-R1, outperforms strong multimodal models, such as GPT-4o, Claude3.5-Sonnet, and Gemini-1.5-Pro, on multiple visual reasoning benchmarks.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "Visionary-R1 reduces shortcut learning in visual reinforcement learning by requiring image interpretation before reasoning through a caption–reason–answer output format.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Applying reinforcement learning directly to visual question answering can encourage shortcuts learned from easy questions. Visionary-R1 explicitly encourages image interpretation before constructing a reasoning chain.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Analyzes shortcut learning and failures to generalize to unseen data distributions.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Trains with reinforcement learning on question-answer pairs without explicit chain-of-thought supervision.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On the 3B backbone, Visionary-R1 scores 69.4 on MathVista, 24.7 on MathVision and 84.1 on MMBench, compared with 61.8, 20.3 and 78.6 for the GRPO baseline in the same table.",
            "fragment": "S4.T1",
            "locator": "Table 1 · 3B backbone comparison · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "In the ChartQA-training ablation, adding captions to GRPO changes MathVista/MathVision from 59.0/18.2 to 62.6/20.9; the caption-reward configuration reaches 64.6/22.7. Adding only a caption-length reward gives 62.0/20.3, so longer captions alone do not explain the improvement.",
            "fragment": "S4.T2",
            "locator": "Table 2 · perception-reward ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Direct visual-question RL",
              "design": "Can exploit shortcuts from easier questions instead of learning robust visual reasoning."
            },
            {
              "method": "Visionary-R1",
              "design": "Uses a caption–reason–answer format to encourage image interpretation before reasoning, without CoT labels."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2603.12252",
      "title": "EndoCoT: Scaling Endogenous Chain-of-Thought Reasoning in Diffusion Models",
      "authors": [
        {
          "name": "Xuanlang Dai"
        },
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Long Xing"
        },
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Yuhong Liu"
        },
        {
          "name": "Beichen Zhang"
        },
        {
          "name": "Kai Chen",
          "corresponding": true
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ECCV",
        "citationText": "European Conference on Computer Vision (ECCV), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-03-12",
        "arxivLastUpdated": "2026-06-18"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2603.12252",
        "googleScholar": "hW23VKIAAAAJ:uLbwQdceFCQC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2603.12252"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:uLbwQdceFCQC"
        },
        {
          "type": "project",
          "url": "https://internlm.github.io/EndoCoT/"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/EndoCoT",
          "repository": "InternLM/EndoCoT"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": true,
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 375,
        "order": 7
      },
      "shortName": "EndoCoT",
      "keywords": [
        "Endogenous chain-of-thought",
        "Latent reasoning",
        "Diffusion models",
        "Diffusion Transformers (DiT)",
        "Multimodal large language models",
        "Visual reasoning",
        "Spatial planning",
        "Iterative thought guidance",
        "Terminal thought grounding",
        "Progressive training",
        "Test-time scaling",
        "Image-to-image generation"
      ],
      "citation": {
        "key": "dai2026endocot",
        "type": "article",
        "journal": "arXiv preprint arXiv:2603.12252",
        "source": {
          "label": "arXiv record",
          "url": "https://arxiv.org/abs/2603.12252"
        },
        "status": "preprint",
        "year": 2026,
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "html": {
            "label": "EndoCoT paper · arXiv v4",
            "url": "https://arxiv.org/html/2603.12252v4"
          },
          "paper": {
            "label": "EndoCoT paper · arXiv v4",
            "url": "https://arxiv.org/pdf/2603.12252v4",
            "encodingFormat": "application/pdf"
          },
          "citation": {
            "label": "Author-provided citation",
            "url": "https://github.com/InternLM/EndoCoT#-citation"
          }
        },
        "abstract": {
          "text": "Recently, Multimodal Large Language Models (MLLMs) have been widely integrated into diffusion frameworks primarily as text encoders to tackle complex tasks such as spatial reasoning. However, this paradigm suffers from two critical limitations: (i) MLLMs text encoder exhibit insufficient reasoning depth. Single-step encoding fails to activate the Chain-of-Thought process, which is essential for MLLMs to provide accurate guidance for complex tasks. (ii) The guidance remains invariant during the decoding process. Invariant guidance during decoding prevents DiT from progressively decomposing complex instructions into actionable denoising steps, even with correct MLLM encodings. To this end, we propose Endogenous Chain-of-Thought (EndoCoT), a novel framework that first activates MLLMs’ reasoning potential by iteratively refining latent thought states through an iterative thought guidance module, and then bridges these states to the DiT’s denoising process. Second, a terminal thought grounding module is applied to ensure the reasoning trajectory remains grounded in textual supervision by aligning the final state with ground-truth answers. With these two components, the MLLM text encoder delivers meticulously reasoned guidance, enabling the DiT to execute it progressively and ultimately solve complex tasks in a step-by-step manner. Extensive evaluations across diverse benchmarks (e.g., Maze, TSP, VSP, and Sudoku) achieve an average accuracy of 92.1%, outperforming the strongest baseline by 8.3 percentage points.",
          "source": "html",
          "fragment": "abstract1",
          "locator": "Author abstract"
        },
        "takeaway": {
          "text": "EndoCoT couples iterative latent reasoning with diffusion generation, achieving 92.1% average accuracy under task-specific training across the visual reasoning settings in Table 1, compared with 83.8% for DiffThinker.",
          "source": "html",
          "fragment": "S5.T1",
          "locator": "Table 1"
        },
        "summary": {
          "text": "Using an MLLM as a single-pass text encoder leaves diffusion generation with shallow reasoning and fixed guidance. EndoCoT repeatedly updates latent thought states, uses them to condition the diffusion model, and grounds the final state in textual supervision so that generation can follow a multi-step reasoning process.",
          "source": "html",
          "fragment": "S4.SS2",
          "locator": "Section 4.2"
        },
        "contributions": [
          {
            "text": "Iterative thought guidance updates latent reasoning states and conditions visual generation at each reasoning step; LoRA adapts both the MLLM and the diffusion transformer.",
            "source": "html",
            "fragment": "S4.SS2.SSS1",
            "locator": "Section 4.2.1"
          },
          {
            "text": "Terminal thought grounding aligns the final latent state with a reference representation of the ground-truth reasoning, complementing visual supervision.",
            "source": "html",
            "fragment": "S4.SS2.SSS2",
            "locator": "Section 4.2.2"
          },
          {
            "text": "Progressive training first supervises intermediate and final reasoning steps, then optimizes the terminal output. At inference, latent thought states can be updated without decoding every intermediate image.",
            "source": "html",
            "fragment": "S4.SS2.SSS3",
            "locator": "Sections 4.2.3–4.2.4"
          }
        ],
        "methodComparison": {
          "source": "html",
          "fragment": "S4.SS2",
          "locator": "Section 4.2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "signal",
              "label": "Conditioning"
            },
            {
              "key": "optimization",
              "label": "Reasoning process"
            },
            {
              "key": "data",
              "label": "Supervision"
            }
          ],
          "rows": [
            {
              "method": "Standard diffusion conditioning",
              "signal": "Static text embeddings",
              "optimization": "Denoising under fixed guidance",
              "data": "Final visual target"
            },
            {
              "method": "EndoCoT",
              "signal": "Iteratively refined latent thought states",
              "optimization": "Multiple latent reasoning steps guide generation",
              "data": "Intermediate visual targets and textual grounding of the final state"
            }
          ]
        },
        "resultsCaption": "Qwen-Image-Edit-2511 · arXiv v4, Table 1. Comparisons use the same training setting in each row.",
        "resultsNote": "Average accuracy follows the 18 settings reported in Table 1, covering Maze, TSP, Sudoku and VSP (including VSP-Super). Gains are percentage points over DiffThinker.",
        "resultColumns": [
          {
            "key": "baseline",
            "label": "DiffThinker"
          },
          {
            "key": "result",
            "label": "EndoCoT"
          }
        ],
        "results": [
          {
            "setting": "Task-specific training · all evaluated settings",
            "metric": "Average accuracy (%)",
            "baseline": 83.8,
            "result": 92.1,
            "source": "html",
            "fragment": "S5.T1",
            "locator": "Table 1"
          },
          {
            "setting": "Unified training · one model for all tasks",
            "metric": "Average accuracy (%)",
            "baseline": 77.1,
            "result": 84.2,
            "source": "html",
            "fragment": "S5.T1",
            "locator": "Table 1"
          },
          {
            "setting": "Task-specific training · Maze, scale 32",
            "metric": "Accuracy (%)",
            "baseline": 65,
            "result": 90,
            "source": "html",
            "fragment": "S5.T1",
            "locator": "Table 1"
          },
          {
            "setting": "Task-specific training · Sudoku, scale 35",
            "metric": "Accuracy (%)",
            "baseline": 55,
            "result": 95,
            "source": "html",
            "fragment": "S5.T1",
            "locator": "Table 1"
          }
        ]
      }
    },
    {
      "id": "arxiv:2603.12648",
      "title": "From Sparse to Dense: Multi-View GRPO for Flow Models via Augmented Condition Space",
      "authors": [
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Yibin Wang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Tianyi Wei"
        },
        {
          "name": "Xiaohang Zhan"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Xingang Pan"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ECCV",
        "citationText": "European Conference on Computer Vision (ECCV), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-03-13",
        "arxivLastUpdated": "2026-03-13"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2603.12648",
        "googleScholar": "hW23VKIAAAAJ:Fu2w8maKXqMC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2603.12648"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:Fu2w8maKXqMC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 397,
        "order": 8
      },
      "shortName": "MV-GRPO",
      "keywords": [
        "MV-GRPO",
        "Multi-view reward",
        "Flow models",
        "Preference alignment",
        "Condition augmentation",
        "Advantage estimation",
        "Text-to-image generation",
        "Group Relative Policy Optimization"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv260312648",
        "year": 2026,
        "journal": "arXiv preprint arXiv:2603.12648",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2603.12648"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2603.12648v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2603.12648v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2603.12648v1"
          }
        },
        "abstract": {
          "text": "Group Relative Policy Optimization (GRPO) has emerged as a powerful framework for preference alignment in text-to-image (T2I) flow models. However, we observe that the standard paradigm where evaluating a group of generated samples against a single condition suffers from insufficient exploration of inter-sample relationships, constraining both alignment efficacy and performance ceilings. To address this sparse single-view evaluation scheme, we propose Multi-View GRPO (MV-GRPO), a novel approach that enhances relationship exploration by augmenting the condition space to create a dense multi-view reward mapping. Specifically, for a group of samples generated from one prompt, MV-GRPO leverages a flexible Condition Enhancer to generate semantically adjacent yet diverse captions. These captions enable multi-view advantage re-estimation, capturing diverse semantic attributes and providing richer optimization signals. By deriving the probability distribution of the original samples conditioned on these new captions, we can incorporate them into the training process without costly sample regeneration. Extensive experiments demonstrate that MV-GRPO achieves superior alignment performance over state-of-the-art methods.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "MV-GRPO obtains richer preference-learning signals by evaluating the same generated samples against multiple related captions, without regenerating those samples.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Single-condition group evaluation leaves relationships between generated samples underexplored. MV-GRPO augments the condition space and re-estimates advantages from multiple semantic views.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Uses a Condition Enhancer to produce semantically adjacent but diverse captions.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Derives conditional probabilities for existing samples under new captions to avoid additional generation.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With HPSv3 reward training, MV-GRPO with VLM-generated views raises the HPSv3 score from Flow-GRPO’s 0.147 to 0.155 and HPSv2 from 0.326 to 0.340. CLIP and UnifiedReward alignment do not both improve in this comparison, so this is not a uniform gain across every evaluator.",
            "fragment": "S4.T1",
            "locator": "Table 1 · FLUX.1-dev, HPSv3 reward · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Adding MV-GRPO keeps the reported denoiser-evaluation count at 13 but raises iteration time from 156.26 to 191.95 seconds. Explicit data augmentation uses 156 evaluations and 1,931.15 seconds; the method avoids additional denoising, not all training overhead.",
            "fragment": "S4.T3",
            "locator": "Tables 2–3 · training-cost comparison · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Single-condition group evaluation",
              "design": "Estimates sample advantages under the original conditioning caption."
            },
            {
              "method": "MV-GRPO",
              "design": "Re-evaluates existing samples under multiple semantically related captions without regenerating them."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2509.22186",
      "title": "MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing",
      "authors": [
        {
          "name": "Junbo Niu"
        },
        {
          "name": "Zheng Liu"
        },
        {
          "name": "Zhuangcheng Gu"
        },
        {
          "name": "Bin Wang"
        },
        {
          "name": "Linke Ouyang"
        },
        {
          "name": "Zhiyuan Zhao"
        },
        {
          "name": "Tao Chu"
        },
        {
          "name": "Tianyao He"
        },
        {
          "name": "Fan Wu"
        },
        {
          "name": "Qintong Zhang"
        },
        {
          "name": "Zhenjiang Jin"
        },
        {
          "name": "Guang Liang"
        },
        {
          "name": "Rui Zhang"
        },
        {
          "name": "Wenzheng Zhang"
        },
        {
          "name": "Yuan Qu"
        },
        {
          "name": "Zhifei Ren"
        },
        {
          "name": "Yuefeng Sun"
        },
        {
          "name": "Yuanhong Zheng"
        },
        {
          "name": "Dongsheng Ma"
        },
        {
          "name": "Zirui Tang"
        },
        {
          "name": "Boyu Niu"
        },
        {
          "name": "Ziyang Miao"
        },
        {
          "name": "Hejun Dong"
        },
        {
          "name": "Siyi Qian"
        },
        {
          "name": "Junyuan Zhang"
        },
        {
          "name": "Jingzhou Chen"
        },
        {
          "name": "Fangdong Wang"
        },
        {
          "name": "Xiaomeng Zhao"
        },
        {
          "name": "Liqun Wei"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Shasha Wang"
        },
        {
          "name": "Ruiliang Xu"
        },
        {
          "name": "Yuanyuan Cao"
        },
        {
          "name": "Lu Chen"
        },
        {
          "name": "Qianqian Wu"
        },
        {
          "name": "Huaiyu Gu"
        },
        {
          "name": "Lindong Lu"
        },
        {
          "name": "Keming Wang"
        },
        {
          "name": "Dechen Lin"
        },
        {
          "name": "Guanlin Shen"
        },
        {
          "name": "Xuanhe Zhou"
        },
        {
          "name": "Linfeng Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Bo Zhang"
        },
        {
          "name": "Lei Bai"
        },
        {
          "name": "Pei Chu"
        },
        {
          "name": "Weijia Li"
        },
        {
          "name": "Jiang Wu"
        },
        {
          "name": "Lijun Wu"
        },
        {
          "name": "Zhenxiang Li"
        },
        {
          "name": "Guangyu Wang"
        },
        {
          "name": "Zhongying Tu"
        },
        {
          "name": "Chao Xu"
        },
        {
          "name": "Kai Chen"
        },
        {
          "name": "Yu Qiao"
        },
        {
          "name": "Bowen Zhou"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Wentao Zhang"
        },
        {
          "name": "Conghui He"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ACL",
        "citationText": "Association for Computational Linguistics (ACL), Industry Track, 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-09-26",
        "arxivLastUpdated": "2025-09-29"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2509.22186",
        "googleScholar": "hW23VKIAAAAJ:vRqMK49ujn8C",
        "doi": "10.18653/v1/2026.acl-industry.3"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2509.22186"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:vRqMK49ujn8C"
        },
        {
          "type": "code",
          "url": "https://github.com/opendatalab/MinerU",
          "repository": "opendatalab/MinerU"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 414,
        "order": 9
      },
      "shortName": "MinerU2.5",
      "keywords": [
        "MinerU2.5",
        "Document parsing",
        "Layout analysis",
        "High-resolution documents",
        "OCR",
        "Formula recognition",
        "Table recognition",
        "Coarse-to-fine inference",
        "Vision-language models"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250922186",
        "year": 2026,
        "booktitle": "Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 6: Industry Track)",
        "source": {
          "label": "ACL 2026 proceedings record",
          "url": "https://aclanthology.org/2026.acl-industry.3/"
        },
        "status": "published",
        "authors": [
          {
            "name": "Junbo Niu"
          },
          {
            "name": "Zheng Liu"
          },
          {
            "name": "Zhuangcheng Gu"
          },
          {
            "name": "Bin Wang"
          },
          {
            "name": "Linke Ouyang"
          },
          {
            "name": "Zhiyuan Zhao"
          },
          {
            "name": "Tao Chu"
          },
          {
            "name": "Tianyao He"
          },
          {
            "name": "Fan Wu"
          },
          {
            "name": "Qintong Zhang"
          },
          {
            "name": "Zhenjiang Jin"
          },
          {
            "name": "Guang Liang"
          },
          {
            "name": "Rui Zhang"
          },
          {
            "name": "Wenzheng Zhang"
          },
          {
            "name": "Yuan Qu"
          },
          {
            "name": "Zhifei Ren"
          },
          {
            "name": "Yuefeng Sun"
          },
          {
            "name": "Zirui Tang"
          },
          {
            "name": "Boyu Niu"
          },
          {
            "name": "Yuanhong Zheng"
          },
          {
            "name": "Dongsheng Ma"
          },
          {
            "name": "Ziyang Miao"
          },
          {
            "name": "Hejun Dong"
          },
          {
            "name": "Siyi Qian"
          },
          {
            "name": "Junyuan Zhang"
          },
          {
            "name": "Fangdong Wang"
          },
          {
            "name": "Jingzhou Chen"
          },
          {
            "name": "Xiaomeng Zhao"
          },
          {
            "name": "Liqun Wei"
          },
          {
            "name": "Wei Li"
          },
          {
            "name": "Shasha Wang"
          },
          {
            "name": "Ruiliang Xu"
          },
          {
            "name": "Yuanyuan Cao"
          },
          {
            "name": "Lu Chen"
          },
          {
            "name": "Qianqian Wu"
          },
          {
            "name": "Huaiyu Gu"
          },
          {
            "name": "Lindong Lu"
          },
          {
            "name": "Dechen Lin"
          },
          {
            "name": "Guanlin Shen"
          },
          {
            "name": "Xuanhe Zhou"
          },
          {
            "name": "Linfeng Zhang"
          },
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Xiaoyi Dong"
          },
          {
            "name": "Jiaqi Wang"
          },
          {
            "name": "Bo Zhang"
          },
          {
            "name": "Lei Bai"
          },
          {
            "name": "Pei Chu"
          },
          {
            "name": "Weijia Li"
          },
          {
            "name": "Jiang Wu"
          },
          {
            "name": "Lijun Wu"
          },
          {
            "name": "Zhenxiang Li"
          },
          {
            "name": "Guangyu Wang"
          },
          {
            "name": "Zhongying Tu"
          },
          {
            "name": "Chao Xu"
          },
          {
            "name": "Kai Chen"
          },
          {
            "name": "Bowen Zhou"
          },
          {
            "name": "Dahua Lin"
          },
          {
            "name": "Wentao Zhang"
          },
          {
            "name": "Conghui He"
          }
        ],
        "month": 7,
        "firstPage": 13,
        "lastPage": 42,
        "publisher": "Association for Computational Linguistics",
        "pdfURL": "https://aclanthology.org/2026.acl-industry.3.pdf",
        "verifiedOn": "2026-09-12",
        "verificationNote": "Author order and spellings follow the ACL PDF, p. 13. The Anthology metadata spells Guanlin Shen as Shenguanlin and capitalizes Ruiliang Xu inconsistently."
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2509.22186v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2509.22186v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2509.22186v2"
          }
        },
        "abstract": {
          "text": "We introduce MinerU2.5, a 1.2B-parameter document parsing vision-language model that achieves state-of-the-art recognition accuracy while maintaining exceptional computational efficiency. Our approach employs a coarse-to-fine, two-stage parsing strategy that decouples global layout analysis from local content recognition. In the first stage, the model performs efficient layout analysis on downsampled images to identify structural elements, circumventing the computational overhead of processing high-resolution inputs. In the second stage, guided by the global layout, it performs targeted content recognition on native-resolution crops extracted from the original image, preserving fine-grained details in dense text, complex formulas, and tables. To support this strategy, we developed a comprehensive data engine that generates diverse, large-scale training corpora for both pretraining and fine-tuning. Ultimately, MinerU2.5 demonstrates strong document parsing ability, achieving state-of-the-art performance on multiple benchmarks, surpassing both general-purpose and domain-specific models across various recognition tasks, while maintaining significantly lower computational overhead.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "MinerU2.5 decouples low-resolution layout analysis from native-resolution content recognition, preserving document detail while reducing high-resolution processing costs.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Dense text, formulas and tables require high-resolution inputs, but processing an entire page at native resolution is expensive. MinerU2.5 first locates page elements, then recognizes their original-resolution crops.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Introduces a 1.2B-parameter document-parsing model with a coarse-to-fine two-stage architecture.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Builds a data engine for diverse pretraining and fine-tuning corpora.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On OmniDocBench, the 1.2B-parameter MinerU2.5 reaches an overall score of 90.67, compared with 88.85 for MonkeyOCR-pro-3B and 88.41 for dots.ocr. Its formula CDM is 88.46 and table TEDS is 88.22.",
            "fragment": "S4.T5",
            "locator": "Table 5 · document parsing on OmniDocBench · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "With vLLM, MinerU2.5 processes 2.12 pages per second on an A100 80 GB GPU and 4.47 on an H200 141 GB GPU. The reported token throughputs are 2,337.25 and 4,938.31 tokens per second, respectively; speed depends on the hardware and inference setup.",
            "fragment": "S3.T3",
            "locator": "Table 3 · measured inference throughput · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Whole-page high-resolution processing",
              "design": "Processes dense page content at high resolution with substantial computational cost."
            },
            {
              "method": "MinerU2.5",
              "design": "Separates low-resolution layout localization from native-resolution recognition of selected crops."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2508.04700",
      "title": "SEAgent: Self-Evolving Computer Use Agent with Autonomous Learning from Experience",
      "authors": [
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICML",
        "citationText": "International Conference on Machine Learning (ICML), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-08-06",
        "arxivLastUpdated": "2025-08-12"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2508.04700",
        "googleScholar": "hW23VKIAAAAJ:SP6oXDckpogC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2508.04700"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:SP6oXDckpogC"
        },
        {
          "type": "code",
          "url": "https://github.com/SunzeY/SEAgent",
          "repository": "SunzeY/SEAgent"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 435,
        "order": 10
      },
      "shortName": "SEAgent",
      "keywords": [
        "SEAgent",
        "Computer-use agents",
        "Autonomous learning",
        "Experiential learning",
        "World State Model",
        "Curriculum generation",
        "GRPO",
        "Specialist-to-generalist transfer",
        "OS-World"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250804700",
        "year": 2026,
        "booktitle": "International Conference on Machine Learning (ICML)",
        "source": {
          "label": "ICML 2026 conference record",
          "url": "https://icml.cc/virtual/2026/poster/65711"
        },
        "status": "conference-record",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2508.04700v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2508.04700v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2508.04700v2"
          }
        },
        "abstract": {
          "text": "Repurposing large vision-language models (LVLMs) as computer use agents (CUAs) has led to substantial breakthroughs, primarily driven by human-labeled data. However, these models often struggle with novel and specialized software, particularly in scenarios lacking human annotations. To address this challenge, we propose SEAgent, an agentic self-evolving framework enabling CUAs to autonomously evolve through interactions with unfamiliar software. Specifically, SEAgent empowers computer-use agents to autonomously master novel software environments via experiential learning, where agents explore new software, learn through iterative trial-and-error, and progressively tackle auto-generated tasks organized from simple to complex. To achieve this goal, we design a World State Model for step-wise trajectory assessment, along with a Curriculum Generator that generates increasingly diverse and challenging tasks. The agent's policy is updated through experiential learning, comprised of adversarial imitation of failure actions and Group Relative Policy Optimization (GRPO) on successful ones. Furthermore, we introduce a specialist-to-generalist training strategy that integrates individual experiential insights from specialist agents, facilitating the development of a stronger generalist CUA capable of continuous autonomous evolution. This unified agent ultimately achieves performance surpassing ensembles of individual specialist agents on their specialized software. We validate the effectiveness of SEAgent across five novel software environments within OS-World. Our approach achieves a significant improvement of 23.2% in success rate, from 11.3% to 34.5%, over a competitive open-source CUA, i.e., UI-TARS.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "SEAgent improves computer-use agents on unfamiliar software through autonomous exploration, curriculum generation and policy updates from successful and failed experience.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Human-labeled trajectories are scarce for novel software. SEAgent generates experience through interaction, assesses trajectories with a World State Model and progressively increases task difficulty.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Combines adversarial imitation of failed actions with GRPO on successful trajectories.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Transfers experience from specialist agents into a stronger generalist computer-use agent.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Averaged over three runs on the five reported OSWorld applications, SEAgent reaches 32.2% success with specialist agents, 30.6% with general RL, and 34.5% after specialist-to-generalist training. The UI-TARS-7B-DPO baseline scores 11.3%; these figures refer to the evaluated application set.",
            "fragment": "S4.T2",
            "locator": "Table 2 · five-application OSWorld evaluation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "With the World State Model, GRPO reaches 34.8% VSCode success and adding adversarial imitation raises it to 37.7%. The corresponding SFT settings score 23.2% without and 30.4% with adversarial imitation.",
            "fragment": "S4.T3",
            "locator": "Table 3 · VSCode ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Human-labeled computer-use training",
              "design": "Relies on demonstrations that may be scarce for unfamiliar software."
            },
            {
              "method": "SEAgent",
              "design": "Builds experience autonomously, generates curricula and updates policies from successful and failed trajectories."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2512.05111",
      "title": "ARM-Thinker: Reinforcing Multimodal Generative Reward Models with Agentic Tool Use and Visual Reasoning",
      "authors": [
        {
          "name": "Shengyuan Ding"
        },
        {
          "name": "Xinyu Fang"
        },
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Xiangyu Zhao"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Jianze Liang"
        },
        {
          "name": "Bin Wang"
        },
        {
          "name": "Conghui He"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-12-04",
        "arxivLastUpdated": "2025-12-04"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2512.05111",
        "googleScholar": "hW23VKIAAAAJ:WA5NYHcadZ8C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2512.05111"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:WA5NYHcadZ8C"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/ARM-Thinker",
          "repository": "InternLM/ARM-Thinker"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 456,
        "order": 11
      },
      "shortName": "ARM-Thinker",
      "keywords": [
        "ARM-Thinker",
        "Agentic reward models",
        "Tool use",
        "Visual grounding",
        "Document retrieval",
        "Multimodal verification",
        "Reinforcement learning",
        "ARMBench-VL"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251205111",
        "year": 2026,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2026 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Ding_ARM-Thinker_Reinforcing_Multimodal_Generative_Reward_Models_with_Agentic_Tool_Use_CVPR_2026_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 22195,
        "lastPage": 22205,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2026/papers/Ding_ARM-Thinker_Reinforcing_Multimodal_Generative_Reward_Models_with_Agentic_Tool_Use_CVPR_2026_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2512.05111v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2512.05111v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2512.05111v1"
          }
        },
        "abstract": {
          "text": "Reward models are critical for aligning vision-language systems with human preferences, yet current approaches suffer from hallucination, weak visual grounding, and an inability to use tools for verification, limiting their reliability on complex multimodal reasoning tasks. We present ARM-Thinker, an Agentic multimodal Reward Model that autonomously invokes external tools (e.g., image cropping, doc page retrieval) to ground judgments in verifiable evidence, replacing static, non-interactive reward scoring. This enables the model to verify fine-grained visual details, cross-reference multi-page evidence, and validate reasoning claims, which are capabilities absent in existing reward models. We train ARM-Thinker with multi-stage reinforcement learning, jointly optimizing tool-calling decisions and judgment accuracy. To evaluate agentic reward modeling, we introduce ARMBench-VL, comprising three benchmarks that assess fine-grained visual grounding (image-level tools), multi-page document understanding (retrieval tools), and instruction following (text-level verification). ARM-Thinker achieves +16.2% average improvement on reward modeling benchmarks, +9.6% on tool-use tasks, and outperforms baselines on multimodal math and logical reasoning benchmarks. Our results demonstrate that agentic capabilities significantly enhance both accuracy and interpretability of reward models.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "ARM-Thinker turns multimodal reward scoring into an evidence-seeking process by learning when to crop images, retrieve document pages and verify reasoning claims.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Static reward models can hallucinate or miss fine visual details. ARM-Thinker uses external tools to gather verifiable evidence before judging multimodal responses.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Jointly optimizes tool-calling decisions and judgment accuracy through multi-stage reinforcement learning.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Introduces ARMBench-VL for visual grounding, multi-page evidence retrieval and instruction verification.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "ARM-Thinker-7B raises ARMBench-VL average performance from Qwen2.5-VL-7B’s 46.1 to 64.6 and VL-RewardBench overall performance from 50.1 to 67.8. The three-benchmark average rises from 47.8 to 64.0.",
            "fragment": "S5.T2",
            "locator": "Table 2 · reward-model evaluation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Enabling tools raises ARM-Thinker’s ARMBench-VL score from 59.2 to 64.6 and V* from 82.2 to 86.4. Enabling the same tool interface for the base Qwen2.5-VL-7B decreases these scores from 46.1 to 44.3 and 75.4 to 50.3, indicating that tool access alone is insufficient.",
            "fragment": "S5.T5",
            "locator": "Table 5 · tool-use ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Static reward scoring",
              "design": "Judges the supplied multimodal response without interactively gathering additional verification evidence."
            },
            {
              "method": "ARM-Thinker",
              "design": "Learns tool use and judgment jointly, gathering image crops or document pages before assigning rewards."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2511.15703",
      "title": "Think Visually, Reason Textually: Vision-Language Synergy in ARC",
      "authors": [
        {
          "name": "Beichen Zhang"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-11-19",
        "arxivLastUpdated": "2025-11-26"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2511.15703",
        "googleScholar": "hW23VKIAAAAJ:XiVPGOgt02cC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2511.15703"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:XiVPGOgt02cC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/ARC-VL",
          "repository": "InternLM/ARC-VL"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 477,
        "order": 12
      },
      "shortName": "ARC-VL",
      "keywords": [
        "ARC-VL",
        "ARC-AGI",
        "Abstract reasoning",
        "Visual abstraction",
        "Symbolic rule execution",
        "Vision-language synergy",
        "Modality-switch self-correction",
        "Few-shot reasoning"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251115703",
        "year": 2026,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2026 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Zhang_Think_Visually_Reason_Textually_Vision-Language_Synergy_in_Abstract_Reasoning_CVPR_2026_paper.html"
        },
        "status": "published",
        "title": "Think Visually, Reason Textually: Vision-Language Synergy in Abstract Reasoning",
        "month": 6,
        "firstPage": 41203,
        "lastPage": 41212,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2026/papers/Zhang_Think_Visually_Reason_Textually_Vision-Language_Synergy_in_Abstract_Reasoning_CVPR_2026_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2511.15703v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2511.15703v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2511.15703v2"
          }
        },
        "abstract": {
          "text": "Abstract reasoning from minimal examples remains a core unsolved problem for frontier foundation models such as GPT-5 and Grok 4. These models still fail to infer structured transformation rules from a handful of examples, which is a key hallmark of human intelligence. The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) provides a rigorous testbed for this capability, demanding conceptual rule induction and transfer to novel tasks. Most existing methods treat ARC-AGI as a purely textual reasoning task, overlooking the fact that humans rely heavily on visual abstraction when solving such puzzles. However, our pilot experiments reveal a paradox: naively rendering ARC-AGI grids as images degrades performance due to imprecise rule execution. This leads to our central hypothesis that vision and language possess complementary strengths across distinct reasoning stages: vision supports global pattern abstraction and verification, whereas language specializes in symbolic rule formulation and precise execution. Building on this insight, we introduce two synergistic strategies: (1) Vision-Language Synergy Reasoning (VLSR), which decomposes ARC-AGI into modality-aligned subtasks; and (2) Modality-Switch Self-Correction (MSSC), which leverages vision to verify text-based reasoning for intrinsic error correction. Extensive experiments demonstrate that our approach yields up to a 4.33% improvement over text-only baselines across diverse flagship models and multiple ARC-AGI tasks. Our findings suggest that unifying visual abstraction with linguistic reasoning is a crucial step toward achieving generalizable, human-like intelligence in future foundation models. Source code is released at https://github.com/InternLM/ARC-VL.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "ARC-VL combines visual pattern abstraction and verification with textual rule execution, addressing the complementary strengths of vision and language on ARC-AGI puzzles.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Treating ARC-AGI entirely as text ignores visual abstraction, while simply rendering grids as images can harm precise execution. The paper assigns different reasoning stages to the modality best suited to them.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Introduces Vision-Language Synergy Reasoning to decompose puzzles into modality-aligned subtasks.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses Modality-Switch Self-Correction for visual verification of text-based reasoning.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Combining vision-language synergy reasoning and modality-switch self-correction raises ARC-AGI accuracy from 8.25% to 14.5% for GPT-4o, 35.0% to 42.25% for Gemini-2.5-Pro, and 42.25% to 46.75% for o4-mini.",
            "fragment": "S4.T2",
            "locator": "Table 2 · training-free co-reasoning · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Without external feedback, o4-mini reaches 44.75% after three modality-switch correction rounds versus 43.25% after three text-only rounds, from a 42.25% baseline. GPT-4o reaches 12.0% versus 8.75%, respectively.",
            "fragment": "S4.T4",
            "locator": "Table 4 · three-round self-correction ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Text-only ARC reasoning",
              "design": "Represents puzzles symbolically but does not directly use visual pattern abstraction and verification."
            },
            {
              "method": "Naive image rendering",
              "design": "Adds visual input without separating abstraction from precise rule execution; pilot results show degraded performance."
            },
            {
              "method": "ARC-VL",
              "design": "Assigns abstraction and verification to vision, and rule formulation and execution to language."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2510.01982",
      "title": "Fine-Grained GRPO for Precise Preference Alignment in Flow Models",
      "authors": [
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Yibin Wang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Li Niu"
        },
        {
          "name": "Guangtao Zhai"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-10-02",
        "arxivLastUpdated": "2025-11-22"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2510.01982",
        "googleScholar": "hW23VKIAAAAJ:PELIpwtuRlgC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2510.01982"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:PELIpwtuRlgC"
        },
        {
          "type": "code",
          "url": "https://github.com/bcmi/Granular-GRPO",
          "repository": "bcmi/Granular-GRPO"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 498,
        "order": 13
      },
      "shortName": "G²RPO",
      "keywords": [
        "G²RPO",
        "Granular-GRPO",
        "Flow models",
        "Preference alignment",
        "Credit assignment",
        "Stochastic sampling",
        "Multi-granularity advantages",
        "Diffusion reinforcement learning"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251001982",
        "year": 2026,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2026 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Zhou_Fine-Grained_GRPO_for_Precise_Preference_Alignment_in_Flow_Models_CVPR_2026_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 20045,
        "lastPage": 20054,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2026/papers/Zhou_Fine-Grained_GRPO_for_Precise_Preference_Alignment_in_Flow_Models_CVPR_2026_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2510.01982v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2510.01982v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2510.01982v3"
          }
        },
        "abstract": {
          "text": "The incorporation of online reinforcement learning (RL) into diffusion and flow-based generative models has recently gained attention as a powerful paradigm for aligning model behavior with human preferences. By leveraging stochastic sampling via Stochastic Differential Equations (SDEs) during the denoising phase, these models can explore a variety of denoising trajectories, enhancing the exploratory capacity of RL. However, despite their ability to discover potentially high-reward samples, current approaches often struggle to effectively align with preferences due to the sparsity and narrowness of reward feedback. To overcome this limitation, we introduce a novel framework called Granular-GRPO (G²RPO), which enables fine-grained and comprehensive evaluation of sampling directions in the RL training of flow models. Specifically, we propose a Singular Stochastic Sampling mechanism that supports step-wise stochastic exploration while ensuring strong correlation between injected noise and reward signals, enabling more accurate credit assignment to each SDE perturbation. Additionally, to mitigate the bias introduced by fixed-granularity denoising, we design a Multi-Granularity Advantage Integration module that aggregates advantages computed across multiple diffusion scales, resulting in a more robust and holistic assessment of sampling trajectories. Extensive experiments on various reward models, including both in-domain and out-of-domain settings, demonstrate that our G²RPO outperforms existing flow-based GRPO baselines, highlighting its effectiveness and generalization capability.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "G²RPO improves credit assignment in flow-model reinforcement learning by isolating stochastic perturbations and integrating advantages across multiple denoising granularities.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Sparse and narrow reward feedback can obscure which denoising decisions produced a good sample. Granular-GRPO evaluates sampling directions more precisely and combines evidence from multiple diffusion scales.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Introduces Singular Stochastic Sampling to strengthen the connection between injected noise and rewards.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Aggregates advantages using Multi-Granularity Advantage Integration.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With joint HPS and CLIP rewards, G²RPO scores 0.376 HPS, 0.406 CLIP and 3.783 UnifiedReward, versus 0.363, 0.399 and 3.661 for MixGRPO. These are evaluator scores rather than task-accuracy percentages.",
            "fragment": "S4.T1",
            "locator": "Table 1 · joint HPS and CLIP reward training · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "Using granularities {1,2,3} instead of {1} changes CLIP from 0.395 to 0.406 and UnifiedReward from 3.688 to 3.783. HPS is 0.376, slightly below the {1,3} setting’s 0.378, illustrating the multi-metric trade-off.",
            "fragment": "S4.T2",
            "locator": "Table 2 · denoising-granularity ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Fixed-granularity flow-model GRPO",
              "design": "Can leave reward feedback sparse and obscure the contribution of individual stochastic perturbations."
            },
            {
              "method": "G²RPO",
              "design": "Isolates stochastic exploration and integrates advantages across denoising granularities for finer credit assignment."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2510.27606",
      "title": "Spatial-SSRL: Enhancing Spatial Understanding via Self-Supervised Reinforcement Learning",
      "authors": [
        {
          "name": "Yuhong Liu"
        },
        {
          "name": "Beichen Zhang"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Long Xing"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-10-31",
        "arxivLastUpdated": "2025-11-25"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2510.27606",
        "googleScholar": "hW23VKIAAAAJ:mvPsJ3kp5DgC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2510.27606"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:mvPsJ3kp5DgC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/Spatial-SSRL",
          "repository": "InternLM/Spatial-SSRL"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 519,
        "order": 14
      },
      "shortName": "Spatial-SSRL",
      "keywords": [
        "Spatial-SSRL",
        "Spatial understanding",
        "Self-supervised reinforcement learning",
        "Verifiable rewards",
        "RGB-D images",
        "Relative depth",
        "3D position prediction",
        "Spatial pretext tasks"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251027606",
        "year": 2026,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2026 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Liu_Spatial-SSRL_Enhancing_Spatial_Understanding_via_Self-Supervised_Reinforcement_Learning_CVPR_2026_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 9570,
        "lastPage": 9581,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2026/papers/Liu_Spatial-SSRL_Enhancing_Spatial_Understanding_via_Self-Supervised_Reinforcement_Learning_CVPR_2026_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2510.27606v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2510.27606v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2510.27606v2"
          }
        },
        "abstract": {
          "text": "Spatial understanding remains a weakness of Large Vision-Language Models (LVLMs). Existing supervised fine-tuning (SFT) and recent reinforcement learning with verifiable rewards (RLVR) pipelines depend on costly supervision, specialized tools, or constrained environments that limit scale. We introduce Spatial-SSRL, a self-supervised RL paradigm that derives verifiable signals directly from ordinary RGB or RGB-D images. Spatial-SSRL automatically formulates five pretext tasks that capture 2D and 3D spatial structure: shuffled patch reordering, flipped patch recognition, cropped patch inpainting, regional depth ordering, and relative 3D position prediction. These tasks provide ground-truth answers that are easy to verify and require no human or LVLM annotation. Training on our tasks substantially improves spatial reasoning while preserving general visual capabilities. On seven spatial understanding benchmarks in both image and video settings, Spatial-SSRL delivers average accuracy gains of 4.63% (3B) and 3.89% (7B) over the Qwen2.5-VL baselines. Our results show that simple, intrinsic supervision enables RLVR at scale and provides a practical route to stronger spatial intelligence in LVLMs.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Spatial-SSRL derives verifiable spatial rewards from ordinary RGB or RGB-D images, improving spatial understanding without human or LVLM annotations for its pretext tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Spatial reinforcement learning often depends on expensive labels, tools or constrained environments. Spatial-SSRL constructs self-supervised tasks whose answers follow directly from image transformations and depth information.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Defines patch reordering, flip recognition, crop inpainting, depth ordering and relative 3D-position tasks.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses these intrinsic verification signals to scale spatial RLVR training.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Spatial-SSRL-3B improves the seven-benchmark average from the non-reasoning Qwen2.5-VL-3B baseline’s 45.91 to 50.54; the 7B model improves from 52.69 to 56.58. The reasoning-prompted baselines score 44.85 and 49.58, respectively.",
            "fragment": "S4.T1",
            "locator": "Table 1 · seven spatial benchmarks · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Depth-only training reaches 63.48 on the 3DSR-Height subset versus 58.91 for all five tasks, while all-task training reaches 74.11 on 3DSR-Location versus 72.94 for depth only. The task mixture improves breadth but does not dominate every specialized subset.",
            "fragment": "S4.T4",
            "locator": "Table 4 · task ablation, 7B backbone · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Annotation- or tool-dependent spatial RL",
              "design": "Obtains supervision from labeled spatial data, external tools or constrained task environments."
            },
            {
              "method": "Spatial-SSRL",
              "design": "Derives verifiable pretext-task rewards from RGB or RGB-D inputs without human or LVLM annotations."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2512.01248",
      "title": "TRivia: Self-supervised Fine-tuning of Vision-Language Models for Table Recognition",
      "authors": [
        {
          "name": "Junyuan Zhang"
        },
        {
          "name": "Bin Wang"
        },
        {
          "name": "Qintong Zhang"
        },
        {
          "name": "Fan Wu"
        },
        {
          "name": "Zichen Wen"
        },
        {
          "name": "Jialin Lu"
        },
        {
          "name": "Junjie Shan"
        },
        {
          "name": "Ziqi Zhao"
        },
        {
          "name": "Shuya Yang"
        },
        {
          "name": "Ziling Wang"
        },
        {
          "name": "Ziyang Miao"
        },
        {
          "name": "Huaping Zhong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Ka-Ho Chow"
        },
        {
          "name": "Conghui He"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-12-01",
        "arxivLastUpdated": "2026-03-24"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2512.01248",
        "googleScholar": "hW23VKIAAAAJ:t6usbXjVLHcC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2512.01248"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:t6usbXjVLHcC"
        },
        {
          "type": "code",
          "url": "https://github.com/opendatalab/TRivia",
          "repository": "opendatalab/TRivia"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 540,
        "order": 15
      },
      "shortName": "TRivia",
      "keywords": [
        "TRivia",
        "Table recognition",
        "Document parsing",
        "Self-supervised fine-tuning",
        "Unlabeled table images",
        "Question-answering rewards",
        "GRPO",
        "Structured document extraction"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251201248",
        "year": 2026,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2026 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2026/html/Zhang_TRivia_Self-supervised_Fine-tuning_of_Vision-Language_Models_for_Table_Recognition_CVPR_2026_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 33196,
        "lastPage": 33206,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2026/papers/Zhang_TRivia_Self-supervised_Fine-tuning_of_Vision-Language_Models_for_Table_Recognition_CVPR_2026_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2512.01248v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2512.01248v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2512.01248v2"
          }
        },
        "abstract": {
          "text": "Table recognition (TR) aims to transform table images into semi-structured representations such as HTML or Markdown. As a core component of document parsing, TR has long relied on supervised learning, with recent efforts dominated by fine-tuning vision-language models (VLMs) using labeled data. While VLMs have brought TR to the next level, pushing performance further demands large-scale labeled data that is costly to obtain. Consequently, although proprietary models have continuously pushed the performance boundary, open-source models, often trained with limited resources and, in practice, the only viable option for many due to privacy regulations, still lag far behind. To bridge this gap, we introduce TRivia, a self-supervised fine-tuning method that enables pretrained VLMs to learn TR directly from unlabeled table images in the wild. Built upon Group Relative Policy Optimization, TRivia automatically identifies unlabeled samples that most effectively facilitate learning and eliminates the need for human annotations through a question-answering-based reward mechanism. An attention-guided module generates diverse questions for each table image, and the ability to interpret the recognition results and answer them correctly provides feedback to optimize the TR model. This closed-loop process allows the TR model to autonomously learn to recognize, structure, and reason over tables without labeled data. Leveraging this pipeline, we present TRivia-3B, an open-sourced, compact, and state-of-the-art TR model that surpasses existing systems (e.g., Gemini 2.5 Pro, MinerU2.5) on three popular benchmarks. Model and code are released at: https://github.com/HKU-TASR/TRivia",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "TRivia improves table recognition from unlabeled table images by rewarding whether recognized table content can correctly answer automatically generated questions.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "High-quality table recognition normally requires costly labeled images. TRivia creates a self-supervised loop that selects informative samples, generates attention-guided questions and uses question-answering feedback to train the recognizer.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Adapts GRPO to learn table structure and content from unlabeled images.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces TRivia-3B as a compact table-recognition model trained with this feedback loop.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "TRivia-3B obtains TEDS scores of 91.60 on OmniDocBench, 84.90 on CC-OCR and 90.76 on OCRBench, versus MinerU2.5’s 90.85, 79.76 and 87.13. The reported overall TEDS across these three benchmarks is 89.88.",
            "fragment": "S4.T1",
            "locator": "Table 1 · table recognition · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Starting from Stage 2’s 88.57 overall TEDS, SFT with Qwen2.5-VL-72B pseudo-labels reduces performance to 80.02 and GRPO with those labels yields 83.65. TRivia’s table-QA approach reaches 89.88, supporting its use of question-answer consistency in place of direct pseudo-label supervision.",
            "fragment": "S5.T2",
            "locator": "Table 2 · pseudo-label ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Supervised table recognition",
              "design": "Learns table structure and content from labeled table images."
            },
            {
              "method": "TRivia",
              "design": "Uses unlabeled images and automatically generated question-answering feedback to fine-tune a table recognizer."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2508.00819",
      "title": "Beyond Fixed: Training-Free Variable-Length Denoising for Diffusion Large Language Models",
      "authors": [
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-08-01",
        "arxivLastUpdated": "2025-08-18"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2508.00819",
        "googleScholar": "hW23VKIAAAAJ:SdhP9T11ey4C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2508.00819"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:SdhP9T11ey4C"
        },
        {
          "type": "code",
          "url": "https://github.com/Li-Jinsong/DAEDAL",
          "repository": "Li-Jinsong/DAEDAL"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 561,
        "order": 16
      },
      "shortName": "DAEDAL",
      "keywords": [
        "DAEDAL",
        "Diffusion language models",
        "Variable-length generation",
        "Adaptive denoising",
        "Mask-token insertion",
        "Training-free inference",
        "Generation efficiency",
        "Dynamic length expansion"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250800819",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/940380c12da75e64351cceff4c880557-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 91715,
        "lastPage": 91731,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/940380c12da75e64351cceff4c880557-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2508.00819v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2508.00819v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2508.00819v2"
          }
        },
        "abstract": {
          "text": "Diffusion Large Language Models (DLLMs) are emerging as a powerful alternative to the dominant Autoregressive Large Language Models, offering efficient parallel generation and capable global context modeling. However, the practical application of DLLMs is hindered by a critical architectural constraint: the need for a statically predefined generation length. This static length allocation leads to a problematic trade-off: insufficient lengths cripple performance on complex tasks, while excessive lengths incur significant computational overhead and sometimes result in performance degradation. While the inference framework is rigid, we observe that the model itself possesses internal signals that correlate with the optimal response length for a given task. To bridge this gap, we leverage these latent signals and introduce DAEDAL, a novel training-free denoising strategy that enables Dynamic Adaptive Length Expansion for Diffusion Large Language Models. DAEDAL operates in two phases: 1) Before the denoising process, DAEDAL starts from a short initial length and iteratively expands it to a coarse task-appropriate length, guided by a sequence completion metric. 2) During the denoising process, DAEDAL dynamically intervenes by pinpointing and expanding insufficient generation regions through mask token insertion, ensuring the final output is fully developed. Extensive experiments on DLLMs demonstrate that DAEDAL achieves performance comparable, and in some cases superior, to meticulously tuned fixed-length baselines, while simultaneously enhancing computational efficiency by achieving a higher effective token ratio. By resolving the static length constraint, DAEDAL unlocks new potential for DLLMs, bridging a critical gap with their Autoregressive counterparts and paving the way for more efficient and capable generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "DAEDAL enables training-free variable-length generation in diffusion language models by expanding the response length before and during denoising when internal signals indicate insufficient space.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "A fixed output length either truncates difficult responses or wastes computation. DAEDAL uses completion signals to choose a coarse initial length and inserts mask tokens into underdeveloped regions during generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Separates initial length selection from local expansion during denoising.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses signals already present in the model without additional training.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On GSM8K, DAEDAL reaches 85.8% accuracy with an average total length of 363 tokens, versus 83.8% at the best fixed-length baseline of 1,024 tokens. The four-task accuracy average rises from 52.05% at that fixed length to 54.75%.",
            "fragment": "S4.T1",
            "locator": "Table 1 · LLaDA-Instruct-8B · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "On GSM8K with an initial length of 64, Stage 1 alone achieves 84.1% accuracy with 311 total tokens, while both DAEDAL stages reach 85.8% with 363. This isolates the accuracy–length trade-off introduced by the second stage.",
            "fragment": "S4.T3",
            "locator": "Table 3 · adaptive-length stage ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Fixed-length diffusion generation",
              "design": "Allocates one response length, risking truncation or unused decoding capacity."
            },
            {
              "method": "DAEDAL",
              "design": "Chooses an initial length from internal signals and expands underdeveloped regions during denoising."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2508.17356",
      "title": "DiCache: Let Diffusion Model Determine Its Own Cache",
      "authors": [
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Yibin Wang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-08-24",
        "arxivLastUpdated": "2025-10-02"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2508.17356",
        "googleScholar": "hW23VKIAAAAJ:p2g8aNsByqUC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2508.17356"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:p2g8aNsByqUC"
        },
        {
          "type": "code",
          "url": "https://github.com/Bujiazi/DiCache",
          "repository": "Bujiazi/DiCache"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 582,
        "order": 17
      },
      "shortName": "DiCache",
      "keywords": [
        "DiCache",
        "Diffusion acceleration",
        "Adaptive caching",
        "Online feature probes",
        "Cache trajectory alignment",
        "Training-free inference",
        "Video generation",
        "Image generation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250817356",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/78288ef33b18a351c3cd679dc9a15c8d-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 73778,
        "lastPage": 73803,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/78288ef33b18a351c3cd679dc9a15c8d-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2508.17356v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2508.17356v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2508.17356v2"
          }
        },
        "abstract": {
          "text": "Recent years have witnessed the rapid development of acceleration techniques for diffusion models, especially caching-based acceleration methods. These studies seek to answer two fundamental questions: \"When to cache\" and \"How to use cache\", typically relying on predefined empirical laws or dataset-level priors to determine caching timings and adopting handcrafted rules for multi-step cache utilization. However, given the highly dynamic nature of the diffusion process, they often exhibit limited generalizability and fail to cope with diverse samples. In this paper, a strong sample-specific correlation is revealed between the variation patterns of the shallow-layer feature differences in the diffusion model and those of deep-layer features. Moreover, we have observed that the features from different model layers form similar trajectories. Based on these observations, we present DiCache, a novel training-free adaptive caching strategy for accelerating diffusion models at runtime, answering both when and how to cache within a unified framework. Specifically, DiCache is composed of two principal components: (1) Online Probe Profiling Scheme leverages a shallow-layer online probe to obtain an on-the-fly indicator for the caching error in real time, enabling the model to dynamically customize the caching schedule for each sample. (2) Dynamic Cache Trajectory Alignment adaptively approximates the deep-layer feature output from multi-step historical caches based on the shallow-layer feature trajectory, facilitating higher visual quality. Extensive experiments validate DiCache's capability in achieving higher efficiency and improved fidelity over state-of-the-art approaches on various leading diffusion models including WAN 2.1, HunyuanVideo and Flux.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "DiCache lets each diffusion sample determine its caching schedule and cache reuse through shallow-layer probes, reducing dependence on fixed caching rules.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Dataset-level caching schedules cannot fully capture sample-specific diffusion dynamics. DiCache exploits correlations between shallow- and deep-layer feature trajectories to estimate caching error online.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Uses Online Probe Profiling to select when cached features can be reused.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Applies Dynamic Cache Trajectory Alignment to approximate deeper features from historical caches.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "For FLUX, DiCache reduces latency from 15.11 to 4.69 seconds (3.22× speedup), with LPIPS 0.2704 relative to vanilla output. For HunyuanVideo, latency falls from 1,186.32 to 507.24 seconds (2.34×), with LPIPS 0.1492.",
            "fragment": "S4.T1",
            "locator": "Table 1 · A800 80GB inference benchmark · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "On HunyuanVideo, increasing the reuse threshold from 0.05 to 0.20 raises speedup from 1.76× to 2.90× but increases LPIPS from 0.1047 to 0.1886. The default threshold of 0.10 yields 2.34× speedup and LPIPS 0.1492.",
            "fragment": "S4.T2",
            "locator": "Table 2 · reuse-threshold ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Dataset-level cache schedules",
              "design": "Applies predetermined reuse rules that cannot fully reflect each sample's diffusion dynamics."
            },
            {
              "method": "DiCache",
              "design": "Uses online shallow-layer probes and cache-trajectory alignment to adapt reuse to the current sample."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2507.15852",
      "title": "Advancing Complex Video Object Segmentation via Progressive Concept Construction",
      "authors": [
        {
          "name": "Zhixiong Zhang"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Songxin He"
        },
        {
          "name": "Jianfan Lin"
        },
        {
          "name": "Junsong Tang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-07-21",
        "arxivLastUpdated": "2026-02-28"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2507.15852",
        "googleScholar": "hW23VKIAAAAJ:dshw04ExmUIC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2507.15852"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:dshw04ExmUIC"
        },
        {
          "type": "code",
          "url": "https://github.com/OpenIXCLab/SeC",
          "repository": "OpenIXCLab/SeC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 603,
        "order": 18
      },
      "shortName": "SeC",
      "keywords": [
        "SeC",
        "Segment Concept",
        "Video object segmentation",
        "Object-centric representations",
        "Concept construction",
        "Scene transitions",
        "SeCVOS",
        "Semantic tracking"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250715852",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/d3f48777432f64bc96b1203713f20352-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 130411,
        "lastPage": 130434,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/d3f48777432f64bc96b1203713f20352-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2507.15852v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2507.15852v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2507.15852v3"
          }
        },
        "abstract": {
          "text": "We propose Segment Concept (SeC), a concept-driven video object segmentation (VOS) framework that shifts from conventional feature matching to the progressive construction and utilization of high-level, object-centric representations. SeC employs Large Vision-Language Models (LVLMs) to integrate visual cues across diverse frames, constructing robust conceptual priors. To balance semantic reasoning with computational overhead, SeC forwards the LVLMs only when a new scene appears, injecting concept-level features at those points. To rigorously assess VOS methods in scenarios demanding high-level conceptual reasoning and robust semantic understanding, we introduce the Semantic Complex Scenarios Video Object Segmentation benchmark (SeCVOS). SeCVOS comprises 160 manually annotated multi-scenario videos designed to challenge models with substantial appearance variations and dynamic scene transformations. Empirical evaluations demonstrate that SeC substantially outperforms state-of-the-art approaches, including SAM 2 and its advanced variants, on both SeCVOS and standard VOS benchmarks. In particular, SeC achieves an 11.8-point improvement over SAM 2.1 on SeCVOS, establishing a new state-of-the-art in concept-aware VOS.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "SeC improves video object segmentation through progressively constructed object concepts, helping maintain identity across large appearance changes and scene transitions.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Feature matching alone is brittle when objects or scenes change substantially. SeC uses large vision-language models to integrate cues across frames and injects conceptual features when new scenes appear.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Builds high-level object representations while limiting expensive LVLM calls to scene changes.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Introduces SeCVOS, 160 manually annotated videos with semantically complex scenarios.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On SeCVOS, SeC reaches 70.0 J&F versus SAM2.1’s 58.2. On the multi-scene-change subset the scores are 67.5 and 52.4, respectively, a 15.1-point improvement.",
            "fragment": "S5.T4",
            "locator": "Table 4 · scene-changing video segmentation · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "On SeCVOS, pixel-level association alone raises J&F from 58.2 to 62.2, and adding concept guidance raises it to 70.0. On SA-V, the same progression is 78.6 → 82.4 → 82.7, showing that concept guidance has a larger effect on the scene-changing benchmark.",
            "fragment": "S5.T7",
            "locator": "Table 6 · component ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Feature-matching video segmentation",
              "design": "Can lose object identity through major appearance changes or scene transitions."
            },
            {
              "method": "SeC",
              "design": "Builds object concepts across frames and injects updated conceptual features at scene changes."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2509.20317",
      "title": "SIM-CoT: Supervised Implicit Chain-of-Thought",
      "authors": [
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Xiaoran Liu"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        },
        {
          "name": "Xipeng Qiu"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-09-24",
        "arxivLastUpdated": "2025-09-25"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2509.20317",
        "googleScholar": "hW23VKIAAAAJ:tOudhMTPpwUC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2509.20317"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:tOudhMTPpwUC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/SIM-CoT",
          "repository": "InternLM/SIM-CoT"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 624,
        "order": 19
      },
      "shortName": "SIM-CoT",
      "keywords": [
        "SIM-CoT",
        "Implicit chain-of-thought",
        "Latent reasoning",
        "Step-level supervision",
        "Training stability",
        "Reasoning interpretability",
        "Token efficiency",
        "Auxiliary decoder"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250920317",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/5d087955ee13fe9a7402eedec879b9c3-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 56721,
        "lastPage": 56742,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/5d087955ee13fe9a7402eedec879b9c3-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2509.20317v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2509.20317v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2509.20317v2"
          }
        },
        "abstract": {
          "text": "Implicit Chain-of-Thought (CoT) methods offer a token-efficient alternative to explicit CoT reasoning in Large Language Models (LLMs), but a persistent performance gap has limited their adoption. We identify a core latent instability issue when scaling the computational budget of implicit CoT: as the number of reasoning tokens increases, training often becomes unstable and collapses. Our analysis shows that this instability arises from latent representations becoming homogeneous and losing semantic diversity, caused by insufficient step-level supervision in current implicit CoT methods. To address this, we propose SIM-CoT, a plug-and-play training module that introduces step-level supervision to stabilize and enrich the latent reasoning space. SIM-CoT employs an auxiliary decoder during training to align each implicit token with its corresponding explicit reasoning step, ensuring latent states capture distinct and meaningful information. The auxiliary decoder is removed at inference, preserving the efficiency of implicit CoT with no added overhead. It also provides interpretability by projecting each latent token onto an explicit reasoning vocabulary, enabling per-step visualization and diagnosis. SIM-CoT significantly improves both in-domain accuracy and out-of-domain stability of implicit CoT methods, boosting Coconut by +8.2% on GPT-2 and CODI by +3.0% on LLaMA-3.1 8B. It further surpasses the explicit CoT baseline on GPT-2 by 2.1% with 2.3× greater token efficiency, while closing the performance gap on larger models like LLaMA-3.1 8B. Code: https://github.com/InternLM/SIM-CoT",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "SIM-CoT stabilizes implicit reasoning by supervising each latent reasoning step with an auxiliary decoder that is removed at inference.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Longer implicit reasoning chains can collapse when latent states become homogeneous. SIM-CoT aligns each implicit token with an explicit reasoning step to preserve semantic diversity and enable step-level inspection.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Adds a plug-in training module that provides supervision for individual latent reasoning states.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Projects latent tokens into an explicit reasoning vocabulary without adding inference-time decoder overhead.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Adding SIM-CoT to CODI raises GSM8k-Aug accuracy from 52.7% to 56.1% while the reported token count remains 13.2. The out-of-domain average rises from 55.8% to 56.8% with the same 13.4-token average.",
            "fragment": "S4.T2",
            "locator": "Table 2 · LLaMA-3.2-1B implicit reasoning · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "For the 1B model, a 1B decoder yields 56.1% GSM8k-Aug accuracy, while 3B and 8B decoders yield 50.4% and 50.0%. Larger auxiliary decoders do not improve this setting.",
            "fragment": "A3.T4",
            "locator": "Table 4a · decoder-size ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Implicit CoT without step-level supervision",
              "design": "Can develop homogeneous latent states and unstable training as the reasoning budget grows."
            },
            {
              "method": "SIM-CoT",
              "design": "Aligns each latent step with explicit reasoning through a training-only auxiliary decoder."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2506.19848",
      "title": "ScaleCap: Inference-Time Scalable Image Captioning via Dual-Modality Debiasing",
      "authors": [
        {
          "name": "Long Xing"
        },
        {
          "name": "Qidong Huang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Weiming Zhang"
        },
        {
          "name": "Nenghai Yu"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Feng Wu"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-06-24",
        "arxivLastUpdated": "2025-06-24"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2506.19848",
        "googleScholar": "hW23VKIAAAAJ:a0OBvERweLwC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2506.19848"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:a0OBvERweLwC"
        },
        {
          "type": "code",
          "url": "https://github.com/Cooperx521/ScaleCap",
          "repository": "Cooperx521/ScaleCap"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 645,
        "order": 20
      },
      "shortName": "ScaleCap",
      "keywords": [
        "ScaleCap",
        "Dense image captioning",
        "Inference-time scaling",
        "Dual-modality debiasing",
        "Hallucination reduction",
        "Contrastive sentence rating",
        "Visual question answering",
        "Caption quality"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250619848",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/37050ebbbd7096719ab96cec19a4c69f-Abstract-Conference.html"
        },
        "status": "published",
        "title": "ScaleCap: Scalable Image Captioning via Dual-Modality Debiasing",
        "firstPage": 32266,
        "lastPage": 32285,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/37050ebbbd7096719ab96cec19a4c69f-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2506.19848v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2506.19848v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2506.19848v1"
          }
        },
        "abstract": {
          "text": "This paper presents ScaleCap, an inference-time scalable image captioning strategy that generates comprehensive and detailed image captions. The key challenges of high-quality image captioning lie in the inherent biases of LVLMs: multimodal bias resulting in imbalanced descriptive granularity, offering detailed accounts of some elements while merely skimming over others; linguistic bias leading to hallucinated descriptions of non-existent objects. To address these issues, we propose a scalable debiased captioning strategy, which continuously enriches and calibrates the caption with increased inference budget. Specifically, we propose two novel components: heuristic question answering and contrastive sentence rating. The former generates content-specific questions based on the image and answers them to progressively inject relevant information into the caption. The latter employs sentence-level offline contrastive decoding to effectively identify and eliminate hallucinations caused by linguistic biases. With increased inference cost, more heuristic questions are raised by ScaleCap to progressively capture additional visual details, generating captions that are more accurate, balanced, and informative. Extensive modality alignment experiments demonstrate the effectiveness of ScaleCap. Annotating 450K images with ScaleCap and using them for LVLM pretraining leads to consistent performance gains across 11 widely used benchmarks. Furthermore, ScaleCap showcases superb richness and fidelity of generated captions with two additional tasks: replacing images with captions in VQA task, and reconstructing images from captions to assess semantic coverage. Code is available at https://github.com/Cooperx521/ScaleCap.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "ScaleCap turns additional inference budget into richer, better-grounded image captions by iteratively adding missing visual details and removing hallucinated descriptions.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Captioners can describe some image elements in detail while omitting others or inventing content. ScaleCap addresses this imbalance with heuristic questions and sentence-level contrastive checks.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Uses image-specific question answering to progressively enrich caption coverage.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Rates sentences with offline contrastive decoding to reduce language-driven hallucinations.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "For Qwen2.5-7B with Qwen2.5-ViT, ScaleCap-450k raises the 11-benchmark average to 64.7, versus 61.6 for vanilla data, 62.4 for ShareGPT4V-450k and 63.0 for DenseFusion-450k. The dataset-size-matched comparison isolates caption quality.",
            "fragment": "S3.T1",
            "locator": "Table 1 · matched 450k-caption pretraining · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "With a Qwen2-72B summarizer, increasing the vision-language captioner from 7B to 72B changes the four-benchmark average from 58.5 to 58.7. Keeping the 7B captioner but shrinking the summarizer to 7B lowers it to 46.5.",
            "fragment": "S4.T7",
            "locator": "Table 6 · captioner and summarizer scale · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Single-pass dense captioning",
              "design": "May overdescribe some image elements while omitting others or hallucinating details."
            },
            {
              "method": "ScaleCap",
              "design": "Adds missing information with targeted questions and checks sentences with contrastive rating as inference budget grows."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2510.24693",
      "title": "STAR-Bench: Probing Deep Spatio-Temporal Reasoning as Audio 4D Intelligence",
      "authors": [
        {
          "name": "Zihan Liu"
        },
        {
          "name": "Zhikang Niu"
        },
        {
          "name": "Qiuyang Xiao"
        },
        {
          "name": "Zhisheng Zheng"
        },
        {
          "name": "Ruoqi Yuan"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Jianze Liang"
        },
        {
          "name": "Xie Chen"
        },
        {
          "name": "Leilei Sun"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-10-28",
        "arxivLastUpdated": "2025-11-28"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2510.24693",
        "googleScholar": "hW23VKIAAAAJ:q3oQSFYPqjQC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2510.24693"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:q3oQSFYPqjQC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/StarBench",
          "repository": "InternLM/StarBench"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 666,
        "order": 21
      },
      "shortName": "STAR-Bench",
      "keywords": [
        "STAR-Bench",
        "Audio 4D intelligence",
        "Spatial audio",
        "Temporal audio reasoning",
        "Acoustic perception",
        "Sound localization",
        "Audio-language models",
        "Multimodal evaluation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv251024693",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/d9f8b5abc8e0926539ecbb492af7b2f1-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 134703,
        "lastPage": 134731,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/d9f8b5abc8e0926539ecbb492af7b2f1-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2510.24693v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2510.24693v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2510.24693v2"
          }
        },
        "abstract": {
          "text": "Despite rapid progress in Multi-modal Large Language Models and Large Audio-Language Models, existing audio benchmarks largely test semantics that can be recovered from text captions, masking deficits in fine-grained perceptual reasoning. We formalize audio 4D intelligence that is defined as reasoning over sound dynamics in time and 3D space, and introduce STAR-Bench to measure it. STAR-Bench combines a Foundational Acoustic Perception setting (six attributes under absolute and relative regimes) with a Holistic Spatio-Temporal Reasoning setting that includes segment reordering for continuous and discrete processes and spatial tasks spanning static localization, multi-source relations, and dynamic trajectories. Our data curation pipeline uses two methods to ensure high-quality samples. For foundational tasks, we use procedurally synthesized and physics-simulated audio. For holistic data, we follow a four-stage process that includes human annotation and final selection based on human performance. Unlike prior benchmarks where caption-only answering reduces accuracy slightly, STAR-Bench induces far larger drops (-31.5% temporal, -35.2% spatial), evidencing its focus on linguistically hard-to-describe cues. Evaluating 19 models reveals substantial gaps compared with humans and a capability hierarchy: closed-source models are bottlenecked by fine-grained perception, while open-source models lag across perception, knowledge, and reasoning. Our STAR-Bench provides critical insights and a clear path forward for developing future models with a more robust understanding of the physical world.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "STAR-Bench probes audio reasoning that text captions cannot adequately replace, exposing weaknesses in understanding sound dynamics across time and three-dimensional space.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Semantic audio benchmarks can conceal perceptual deficits because captions already contain enough information to answer. STAR-Bench targets acoustic attributes, temporal ordering and spatial relations that demand listening.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Combines foundational acoustic perception with holistic temporal and spatial reasoning tasks.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses procedural synthesis, physical simulation and human-validated curation for complementary task families.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Gemini-2.5-Pro achieves 49.59% macro accuracy on STAR-Bench versus the human reference of 79.11%. Its temporal overall accuracy is 58.52% and spatial overall accuracy is 43.62%, compared with 88.00% and 73.72% for humans.",
            "fragment": "S4.T2",
            "locator": "Table 2 · audio reasoning evaluation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "For temporal reasoning, Gemini-2.5-Pro’s average accuracy is 58.52%, but its all-correct rate across repeated runs is 34.89%. The two metrics distinguish average correctness from reliable answers on the same examples.",
            "fragment": "A5.T5",
            "locator": "Table 5 · repeated-run consistency · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Caption-solvable audio evaluation",
              "design": "Emphasizes semantic information that can often be recovered without listening to the original sound."
            },
            {
              "method": "STAR-Bench",
              "design": "Tests acoustic attributes and temporal-spatial relations that are difficult to convey fully through captions."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2602.16455",
      "title": "Visual Self-Refine: A Pixel-Guided Paradigm for Accurate Chart Parsing",
      "authors": [
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2026-02-18",
        "arxivLastUpdated": "2026-02-18"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2602.16455"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2602.16455"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 687,
        "order": 22
      },
      "shortName": "ChartVSR",
      "keywords": [
        "ChartVSR",
        "Visual Self-Refine",
        "Chart parsing",
        "Pixel-level localization",
        "Visual feedback",
        "Perceptual self-correction",
        "ChartP-Bench",
        "Structured data extraction"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv260216455",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/89b0e466b46292ce0bfe53618aadd3de-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 85499,
        "lastPage": 85522,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/89b0e466b46292ce0bfe53618aadd3de-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2602.16455v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2602.16455v1",
            "encodingFormat": "application/pdf"
          },
          "html": {
            "label": "ChartVSR paper · arXiv v1",
            "url": "https://arxiv.org/html/2602.16455v1"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2602.16455v1"
          }
        },
        "abstract": {
          "text": "While Large Vision-Language Models (LVLMs) have demonstrated remarkable capabilities for reasoning and self-correction at the textual level, these strengths provide minimal benefits for complex tasks centered on visual perception, such as Chart Parsing. Existing models often struggle with visually dense charts, leading to errors like data omission, misalignment, and hallucination. Inspired by the human strategy of using a finger as a ``visual anchor'' to ensure accuracy when reading complex charts, we propose a new paradigm named Visual Self-Refine (VSR). The core idea of VSR is to enable a model to generate pixel-level localization outputs, visualize them, and then feed these visualizations back to itself, allowing it to intuitively inspect and correct its own potential visual perception errors. We instantiate the VSR paradigm in the domain of Chart Parsing by proposing ChartVSR. This model decomposes the parsing process into two stages: a Refine Stage, where it iteratively uses visual feedback to ensure the accuracy of all data points' Pixel-level Localizations, and a Decode Stage, where it uses these verified localizations as precise visual anchors to parse the final structured data. To address the limitations of existing benchmarks, we also construct ChartP-Bench, a new and highly challenging benchmark for chart parsing. Our work also highlights VSR as a general-purpose visual feedback mechanism, offering a promising new direction for enhancing accuracy on a wide range of vision-centric tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "ChartVSR refines pixel-level data-point locations through visual feedback before decoding chart values, giving chart parsing an explicit visual self-correction process.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Textual self-correction is insufficient for dense charts with omitted or misaligned data points. Visual Self-Refine renders localization predictions back to the model so that it can inspect and correct perceptual errors.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Separates iterative visual localization refinement from final structured-data decoding.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Introduces ChartP-Bench to evaluate difficult chart-parsing cases.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "ChartVSR-3B achieves an overall average SCRM AP of 38.41, compared with 34.07 for Gemini-2.5-Pro. The score averages the benchmark’s Strict, Slight and High matching criteria.",
            "fragment": "S4.T3",
            "locator": "Table 3 · ChartP-Bench evaluation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Removing visual self-refinement lowers overall average SCRM AP from 38.41 to 36.54. On the Hard subset, which contains charts with more than 18 data points, the score falls from 37.66 to 35.14.",
            "fragment": "S4.T4",
            "locator": "Table 4 · visual-refinement ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Textual self-correction for chart parsing",
              "design": "Revises text predictions without explicitly checking their pixel-level alignment to chart marks."
            },
            {
              "method": "ChartVSR",
              "design": "Renders point-localization predictions for visual inspection and refinement before decoding structured values."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2509.22647",
      "title": "CapRL: Stimulating Dense Image Caption Capabilities via Reinforcement Learning",
      "authors": [
        {
          "name": "Long Xing"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jianze Liang"
        },
        {
          "name": "Qidong Huang"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Feng Wu"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2025-09-26",
        "arxivLastUpdated": "2025-09-26"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2509.22647",
        "googleScholar": "hW23VKIAAAAJ:sSrBHYA8nusC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2509.22647"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:sSrBHYA8nusC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/CapRL",
          "repository": "InternLM/CapRL"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 700,
        "order": 23
      },
      "shortName": "CapRL",
      "keywords": [
        "CapRL",
        "Dense captioning",
        "Verifiable rewards",
        "Reinforcement learning",
        "Caption utility",
        "Vision-free evaluation",
        "Multimodal pretraining",
        "CapRL-5M"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250922647",
        "year": 2026,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2026 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2026/hash/15f1dbc086bfd94d8c32557b573cbe18-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 13066,
        "lastPage": 13093,
        "volume": "2026",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2026/file/15f1dbc086bfd94d8c32557b573cbe18-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2509.22647v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2509.22647v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2509.22647v1"
          }
        },
        "abstract": {
          "text": "Image captioning is a fundamental task that bridges the visual and linguistic domains, playing a critical role in pre-training Large Vision-Language Models (LVLMs). Current state-of-the-art captioning models are typically trained with Supervised Fine-Tuning (SFT), a paradigm that relies on expensive, non-scalable data annotated by humans or proprietary models. This approach often leads to models that memorize specific ground-truth answers, limiting their generality and ability to generate diverse, creative descriptions. To overcome the limitation of SFT, we propose applying the Reinforcement Learning with Verifiable Rewards (RLVR) paradigm to the open-ended task of image captioning. A primary challenge, however, is designing an objective reward function for the inherently subjective nature of what constitutes a \"good\" caption. We introduce Captioning Reinforcement Learning (CapRL), a novel training framework that redefines caption quality through its utility: a high-quality caption should enable a non-visual language model to accurately answer questions about the corresponding image. CapRL employs a decoupled two-stage pipeline where an LVLM generates a caption, and the objective reward is derived from the accuracy of a separate, vision-free LLM answering Multiple-Choice Questions based solely on that caption. As the first study to apply RLVR to the subjective image captioning task, we demonstrate that CapRL significantly enhances multiple settings. Pretraining on the CapRL-5M caption dataset annotated by CapRL-3B results in substantial gains across 12 benchmarks. Moreover, within the Prism Framework for caption quality evaluation, CapRL achieves performance comparable to Qwen2.5-VL-72B, while exceeding the baseline by an average margin of 8.4%. Code is available here: https://github.com/InternLM/CapRL.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "CapRL trains dense captioners with verifiable question-answering rewards: a useful caption should let a vision-free language model answer questions about the image.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Open-ended caption quality is hard to score objectively, while supervised captioning depends on costly annotations. CapRL evaluates captions through the accuracy of a separate text-only model answering image-related multiple-choice questions.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Decouples visual caption generation from vision-free reward evaluation.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Releases a captioning model and the CapRL-5M corpus for multimodal pretraining.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "For Qwen2.5-3B with Qwen2.5-ViT, CapRL-1M reaches a 12-benchmark average of 59.7 versus 56.7 for ShareGPT4V-1M and 57.1 for DenseFusion-1M. Scaling CapRL to 5M captions raises the average to 62.0.",
            "fragment": "S4.T1",
            "locator": "Table 1 · caption pretraining at 1M and 5M scale · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Recaptioning the same ShareGPT4V-1M images with CapRL raises the 3B model’s average from 56.7 to 58.7; recaptioning DenseFusion-1M raises it from 57.1 to 59.9. This controls image selection when measuring the benefit of CapRL captions.",
            "fragment": "S4.T2",
            "locator": "Table 2 · controlled image-source ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Supervised dense captioning",
              "design": "Depends on reference captions whose collection and quality assessment can be expensive."
            },
            {
              "method": "CapRL",
              "design": "Scores captions by whether a separate text-only model can answer image-related questions from their content."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2403.13805",
      "title": "RAR: Retrieving And Ranking Augmented MLLMs for Visual Recognition",
      "authors": [
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2026,
        "venueGroup": "TIP",
        "citationText": "IEEE Transactions on Image Processing (TIP), 2026"
      },
      "dates": {
        "arxivFirstPosted": "2024-03-20",
        "arxivLastUpdated": "2026-05-15"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2403.13805",
        "googleScholar": "hW23VKIAAAAJ:LkGwnXOMwfcC",
        "doi": "10.1109/tip.2025.3644175"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2403.13805"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:LkGwnXOMwfcC"
        },
        {
          "type": "code",
          "url": "https://github.com/Liuziyu77/RAR",
          "repository": "Liuziyu77/RAR"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 721,
        "order": 24
      },
      "shortName": "RAR",
      "keywords": [
        "RAR",
        "Retrieval-augmented recognition",
        "Multimodal reranking",
        "Fine-grained classification",
        "CLIP",
        "Few-shot recognition",
        "Zero-shot recognition",
        "Category memory"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv240313805",
        "year": 2026,
        "journal": "IEEE Transactions on Image Processing",
        "publisher": "Institute of Electrical and Electronics Engineers (IEEE)",
        "source": {
          "label": "TIP publisher record",
          "url": "https://doi.org/10.1109/tip.2025.3644175"
        },
        "status": "published",
        "title": "RAR: Retrieving and Ranking Augmented MLLMs for Visual Recognition",
        "volume": "35",
        "firstPage": 388,
        "lastPage": 401,
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2403.13805v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2403.13805v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2403.13805v2"
          }
        },
        "abstract": {
          "text": "CLIP (Contrastive Language-Image Pre-training) uses contrastive learning from noise image-text pairs to excel at recognizing a wide array of candidates, yet its focus on broad associations hinders the precision in distinguishing subtle differences among fine-grained items. Conversely, Multimodal Large Language Models (MLLMs) excel at classifying fine-grained categories, thanks to their substantial knowledge from pre-training on web-level corpora. However, the performance of MLLMs declines with an increase in category numbers, primarily due to growing complexity and constraints of limited context window size. To synergize the strengths of both approaches and enhance the few-shot/zero-shot recognition abilities for datasets characterized by extensive and fine-grained vocabularies, this paper introduces RAR, a Retrieving And Ranking augmented method for MLLMs. We initially establish a multi-modal retriever based on CLIP to create and store explicit memory for different categories beyond the immediate context window. During inference, RAR retrieves the top-k similar results from the memory and uses MLLMs to rank and make the final predictions. Our proposed approach not only addresses the inherent limitations in fine-grained recognition but also preserves the model's comprehensive knowledge base, significantly boosting accuracy across a range of vision-language recognition tasks. Notably, our approach demonstrates a significant improvement in performance on 5 fine-grained visual recognition benchmarks, 11 few-shot image recognition datasets, and the 2 object detection datasets under the zero-shot recognition setting.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "RAR combines CLIP retrieval with MLLM reranking to recognize fine-grained categories without placing an entire large vocabulary into the language model's context.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "CLIP scales to many categories but can miss subtle distinctions, whereas MLLMs struggle with large candidate sets. RAR retrieves a small set of relevant category memories and lets the MLLM rank them.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Builds an explicit multimodal category memory outside the immediate context window.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Separates scalable candidate retrieval from fine-grained semantic classification.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Across five fine-grained datasets, RAR achieves 58.5% average clustering accuracy and 65.3% semantic-similarity accuracy, compared with FineR’s 57.0% and 64.3%.",
            "fragment": "S4.T2",
            "locator": "Table II · fine-grained recognition · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Using CLIP ViT-L/14@336 features and LLaVA-1.5 reranking, RAR raises average top-1 accuracy across 11 datasets from 65.0% to 69.9% in the 4-shot setting and from 70.8% to 75.5% in the 8-shot setting.",
            "fragment": "S4.T4",
            "locator": "Table IV · CLIP ViT-L/14@336 retrieval · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "CLIP category matching",
              "design": "Scales to large candidate vocabularies but can miss fine-grained distinctions."
            },
            {
              "method": "Direct MLLM classification",
              "design": "Offers semantic reasoning but is constrained by large candidate sets in the context window."
            },
            {
              "method": "RAR",
              "design": "Uses CLIP to retrieve category memories and an MLLM to rerank the selected candidates."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2505.03318",
      "title": "Unified Multimodal Chain-of-Thought Reward Model through Reinforcement Fine-Tuning",
      "authors": [
        {
          "name": "Yibin Wang"
        },
        {
          "name": "Zhimin Li"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Chunyu Wang"
        },
        {
          "name": "Qinglin Lu"
        },
        {
          "name": "Cheng Jin",
          "corresponding": true
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-05-06",
        "arxivLastUpdated": "2025-10-29"
      },
      "topics": [
        "Reinforcement Learning from Human Feedback"
      ],
      "identifiers": {
        "arxiv": "2505.03318",
        "googleScholar": "hW23VKIAAAAJ:35N4QoGY0k4C",
        "doi": "10.52202/085713-5315"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2505.03318"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:35N4QoGY0k4C"
        },
        {
          "type": "code",
          "url": "https://github.com/CodeGoat24/UnifiedReward",
          "repository": "CodeGoat24/UnifiedReward"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/CodeGoat24/UnifiedReward-Think-qwen-7b",
          "label": "Models",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": true,
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 742,
        "order": 25
      },
      "shortName": "UnifiedReward-Think",
      "keywords": [
        "UnifiedReward-Think",
        "Multimodal reward models",
        "Chain-of-thought",
        "Reinforcement fine-tuning",
        "GRPO",
        "Preference learning",
        "Visual understanding",
        "Visual generation",
        "Rejection sampling"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250503318",
        "year": 2025,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2025 proceedings record",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/e95e9f0c127aa1cfa2628adb2f3cb107-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 159130,
        "lastPage": 159157,
        "volume": "38",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2025/file/e95e9f0c127aa1cfa2628adb2f3cb107-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12",
        "verificationNote": "Zhimin Li is spelled and ordered as in the formal PDF, p. 1; the proceedings metadata incorrectly exports this name as zhimin, li."
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2505.03318v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2505.03318v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2505.03318v3"
          }
        },
        "abstract": {
          "text": "Recent advances in multimodal Reward Models (RMs) have shown significant promise in delivering reward signals to align vision models with human preferences. However, current RMs are generally restricted to providing direct responses or engaging in shallow reasoning processes with limited depth, often leading to inaccurate reward signals. We posit that incorporating explicit long chains of thought (CoT) into the reward reasoning process can significantly strengthen their reliability and robustness. Furthermore, we believe that once RMs internalize CoT reasoning, their direct response accuracy can also be improved through implicit reasoning capabilities. To this end, this paper proposes UnifiedReward-Think, the first unified multimodal CoT-based reward model, capable of multi-dimensional, step-by-step long-chain reasoning for both visual understanding and generation reward tasks. Specifically, we adopt an exploration-driven reinforcement fine-tuning approach to elicit and incentivize the model's latent complex reasoning ability: (1) We first use a small amount of image generation preference data to distill the reasoning process of GPT-4o, which is then used for the model's cold start to learn the format and structure of CoT reasoning. (2) Subsequently, by leveraging the model's prior knowledge and generalization capabilities, we prepare large-scale unified multimodal preference data to elicit the model's reasoning process across various vision tasks. During this phase, correct reasoning outputs are retained for rejection sampling to refine the model (3) while incorrect predicted samples are finally used for Group Relative Policy Optimization (GRPO) based reinforcement fine-tuning, enabling the model to explore diverse reasoning paths and optimize for correct and robust solutions. Extensive experiments across various vision reward tasks demonstrate the superiority of our model.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "UnifiedReward-Think strengthens multimodal reward judgments through explicit multi-step reasoning and reinforcement fine-tuning across visual understanding and generation tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Direct or shallow reward judgments can be inaccurate. UnifiedReward-Think develops longer, multidimensional reasoning chains and uses exploration to improve the reliability of reward predictions.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Uses a small distilled reasoning set for cold start, then unified multimodal preference data for rejection sampling.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Applies GRPO to incorrectly predicted samples so the model can explore better reasoning paths.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "UnifiedReward-Think reaches 73.8% overall accuracy versus UnifiedReward’s 67.5%; macro accuracy rises from 66.6% to 72.3%. The trained model’s direct-response mode without explicit CoT reaches 73.1% overall accuracy.",
            "fragment": "S4.T1",
            "locator": "Table 1 · VLRewardBench, LLaVA-OneVision-7B · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "On VLRewardBench, cold-start training alone scores 66.9% overall, rejection sampling raises it to 72.1%, and subsequent GRPO reaches 73.8%. Direct GRPO without CoT training reaches 69.0%, supporting the staged reasoning pipeline.",
            "fragment": "S4.T3",
            "locator": "Table 3 · training-stage ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Direct or shallow reward judgments",
              "design": "Produces preference decisions with limited explicit reasoning."
            },
            {
              "method": "UnifiedReward-Think",
              "design": "Builds multidimensional reasoning through cold start, rejection sampling and reinforcement fine-tuning."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2504.06232",
      "title": "HiFlow: Training-free High-Resolution Image Generation with Flow-Aligned Guidance",
      "authors": [
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Pan Zhang",
          "corresponding": true
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-04-08",
        "arxivLastUpdated": "2025-05-16"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2504.06232",
        "googleScholar": "hW23VKIAAAAJ:J_g5lzvAfSwC",
        "doi": "10.52202/085713-4831"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2504.06232"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:J_g5lzvAfSwC"
        },
        {
          "type": "code",
          "url": "https://github.com/Bujiazi/HiFlow",
          "repository": "Bujiazi/HiFlow"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": true
      },
      "source": {
        "file": "research.html",
        "line": 766,
        "order": 26
      },
      "shortName": "HiFlow",
      "keywords": [
        "HiFlow",
        "High-resolution image generation",
        "Flow-aligned guidance",
        "Training-free generation",
        "Flow matching",
        "Resolution extrapolation",
        "Structure preservation",
        "Text-to-image models"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250406232",
        "year": 2025,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2025 proceedings record",
          "url": "https://proceedings.neurips.cc/paper_files/paper/2025/hash/d4ecde5d18d36150c529d1b9c5f0d727-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 144317,
        "lastPage": 144351,
        "volume": "38",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2025/file/d4ecde5d18d36150c529d1b9c5f0d727-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2504.06232v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2504.06232v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2504.06232v2"
          }
        },
        "abstract": {
          "text": "Text-to-image (T2I) diffusion/flow models have drawn considerable attention recently due to their remarkable ability to deliver flexible visual creations. Still, high-resolution image synthesis presents formidable challenges due to the scarcity and complexity of high-resolution content. Recent approaches have investigated training-free strategies to enable high-resolution image synthesis with pre-trained models. However, these techniques often struggle with generating high-quality visuals and tend to exhibit artifacts or low-fidelity details, as they typically rely solely on the endpoint of the low-resolution sampling trajectory while neglecting intermediate states that are critical for preserving structure and synthesizing finer detail. To this end, we present HiFlow, a training-free and model-agnostic framework to unlock the resolution potential of pre-trained flow models. Specifically, HiFlow establishes a virtual reference flow within the high-resolution space that effectively captures the characteristics of low-resolution flow information, offering guidance for high-resolution generation through three key aspects: initialization alignment for low-frequency consistency, direction alignment for structure preservation, and acceleration alignment for detail fidelity. By leveraging such flow-aligned guidance, HiFlow substantially elevates the quality of high-resolution image synthesis of T2I models and demonstrates versatility across their personalized variants. Extensive experiments validate HiFlow's capability in achieving superior high-resolution image quality over state-of-the-art methods.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "HiFlow improves training-free high-resolution generation by aligning initialization, direction and acceleration with a reference flow derived from lower-resolution generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Using only the endpoint of a low-resolution trajectory can lose structure and fine detail. HiFlow transfers information from intermediate flow states into a virtual high-resolution reference flow.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Aligns low-frequency initialization, structural direction and detail-related acceleration.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Provides a model-agnostic approach that also applies to personalized flow-model variants.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "At 4K resolution, HiFlow obtains FID 52.55 and patch FID 45.01, compared with I-Max’s 53.27 and 52.93. Lower values indicate better distributional image quality in this evaluation.",
            "fragment": "S4.T1",
            "locator": "Table 1 · 4096×4096 image generation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "The reported 4K latency is 379 seconds for HiFlow versus 735 for I-Max. Removing acceleration alignment raises HiFlow’s 4K patch FID from 45.01 to 51.79, isolating its contribution to fine detail.",
            "fragment": "S4.T3",
            "locator": "Tables 2–3 · latency and alignment ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Endpoint-only low-resolution guidance",
              "design": "Uses the final lower-resolution result while discarding intermediate generation-trajectory information."
            },
            {
              "method": "HiFlow",
              "design": "Transfers intermediate flow information through aligned initialization, structural direction and detail-related acceleration."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2503.01785",
      "title": "Visual-RFT: Visual Reinforcement Fine-Tuning",
      "authors": [
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-03-03",
        "arxivLastUpdated": "2025-03-03"
      },
      "topics": [
        "Reinforcement Learning from Human Feedback"
      ],
      "identifiers": {
        "arxiv": "2503.01785",
        "googleScholar": "hW23VKIAAAAJ:O3NaXMp0MMsC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2503.01785"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:O3NaXMp0MMsC"
        },
        {
          "type": "code",
          "url": "https://github.com/Liuziyu77/Visual-RFT",
          "repository": "Liuziyu77/Visual-RFT"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/collections/laolao77/virft-datasets-67bc271b6f2833eccc0651df",
          "label": "Dataset",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": true,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 787,
        "order": 27
      },
      "citation": {
        "type": "inproceedings",
        "key": "Liu_2025_ICCV",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 2034,
        "lastPage": 2044,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "proceedings": {
            "label": "ICCV 2025 proceedings record",
            "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.html"
          },
          "paper": {
            "label": "ICCV 2025 paper",
            "encodingFormat": "application/pdf",
            "url": "https://openaccess.thecvf.com/content/ICCV2025/papers/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.pdf"
          }
        },
        "abstract": {
          "source": "proceedings",
          "text": "Reinforcement Fine-Tuning (RFT) in Large Reasoning Models like OpenAI o1 learns from feedback on its answers, which is especially useful in applications when fine-tuning data is scarce. Recent open-source work like DeepSeek-R1 demonstrates that reinforcement learning with verifiable reward is possibly one key direction in reproducing o1. While the R1-style model has demonstrated success in language models, its application in multi-modal domains remains under-explored. This work introduces Visual Reinforcement Fine-Tuning (Visual-RFT), which further extends the application areas of RFT on visual tasks. Specifically, Visual-RFT first uses Large Vision-Language Models (LVLMs) to generate multiple responses containing reasoning tokens and final answers for each input, and then uses our proposed visual perception verifiable reward functions to update the model via the policy optimization algorithm such as Group Relative Policy Optimization (GRPO). We design different verifiable reward functions for different perception tasks, such as the Intersection over Union (IoU) reward for object detection. Experimental results on fine-grained image classification, few-shot object detection, reasoning grounding, as well as open-vocabulary object detection benchmarks show the competitive performance and advanced generalization ability of Visual-RFT compared with Supervised Fine-tuning (SFT). For example, Visual-RFT improves accuracy by 24.3% over the baseline in one-shot fine-grained image classification with around 100 samples. In few-shot object detection, Visual-RFT also exceeds the baseline by 21.0 on COCO’s 4-shot setting and 15.4 on LVIS. Our Visual-RFT represents a paradigm shift in fine-tuning LVLMs, offering a data-efficient, reward-driven approach that enhances reasoning and adaptability for domain-specific tasks."
        },
        "summary": {
          "text": "Visual-RFT studies how to adapt large vision-language models to visual perception tasks with limited labeled examples. It trains with GRPO and task-specific, rule-based rewards, using class correctness for classification and box overlap, confidence, and output format for detection. The experiments compare reinforcement fine-tuning with supervised fine-tuning on classification, detection, and grounding.",
          "source": "paper",
          "locator": "Sections 1 and 3",
          "page": 4
        },
        "contributions": [
          {
            "text": "Applies reinforcement learning with verifiable rewards to visual perception, including fine-grained classification, few-shot detection, reasoning grounding, and open-vocabulary detection.",
            "source": "paper",
            "locator": "Section 1",
            "page": 3
          },
          {
            "text": "Defines task-specific rewards from labels and bounding boxes, allowing the model to learn from multiple sampled responses without a separately trained preference reward model.",
            "source": "paper",
            "locator": "Section 3.2",
            "page": 5
          },
          {
            "text": "Evaluates learning from limited examples and transfer to novel categories, comparing Visual-RFT with the base model and supervised fine-tuning.",
            "source": "paper",
            "locator": "Sections 4.2–4.5",
            "page": 7
          }
        ],
        "results": [
          {
            "setting": "Fine-grained classification · 1-shot per class; average over Flower102, Pets37, Aircraft, and Cars196",
            "metric": "Accuracy (%)",
            "baseline": 56,
            "sft": 51.7,
            "result": 80.3,
            "source": "paper",
            "locator": "Table 2, p. 2039",
            "page": 6
          },
          {
            "setting": "Few-shot detection · COCO, 8 selected categories, 4-shot per category",
            "metric": "mAP",
            "baseline": 19.6,
            "sft": 25.2,
            "result": 40.6,
            "source": "paper",
            "locator": "Table 3, p. 2039",
            "page": 6
          },
          {
            "setting": "Few-shot detection · LVIS, 6 selected rare categories, approximately 10-shot (1–10 images per category)",
            "metric": "mAP",
            "baseline": 4,
            "sft": 10,
            "result": 19.4,
            "source": "paper",
            "locator": "Table 4, p. 2039; Section 4.3",
            "page": 6
          }
        ],
        "takeaway": {
          "text": "Visual-RFT shows that GRPO with task-specific verifiable rewards improves classification, detection, and grounding in large vision-language models trained with limited labeled data.",
          "source": "paper",
          "locator": "Sections 3.2 and 4; Tables 2–5",
          "page": 6
        },
        "methodComparison": {
          "source": "paper",
          "locator": "Sections 1, 3.1 and 3.2",
          "page": 4,
          "rows": [
            {
              "method": "Supervised fine-tuning (SFT)",
              "signal": "Target outputs supplied with each training example.",
              "optimization": "Increases the likelihood of the demonstrated answers.",
              "data": "Labeled image–instruction–answer examples."
            },
            {
              "method": "RLHF with a learned reward model",
              "signal": "Scores predicted by a reward model fitted to human preferences.",
              "optimization": "Updates the policy to increase the learned reward.",
              "data": "Preference comparisons for reward-model training, plus inputs for policy optimization."
            },
            {
              "method": "Visual-RFT (RLVR + GRPO)",
              "signal": "Rules evaluate class correctness, box overlap, confidence, and output format.",
              "optimization": "Samples multiple responses and uses their relative rewards for GRPO updates.",
              "data": "Task examples with ground-truth labels or boxes; evaluated with limited data."
            }
          ],
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "signal",
              "label": "Learning signal"
            },
            {
              "key": "optimization",
              "label": "Optimization"
            },
            {
              "key": "data",
              "label": "Training data"
            }
          ]
        },
        "relatedWork": {
          "cited": [
            {
              "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
              "firstPosted": "2024-02-05",
              "url": "https://arxiv.org/abs/2402.03300",
              "relationship": "Optimization foundation: introduces GRPO, the policy optimization algorithm used by Visual-RFT.",
              "evidence": {
                "url": "https://openaccess.thecvf.com/content/ICCV2025/papers/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.pdf#page=10",
                "label": "Visual-RFT reference 33; Section 3.1"
              },
              "kind": "paper"
            },
            {
              "title": "Qwen2-VL: Enhancing Vision-Language Model’s Perception of the World at Any Resolution",
              "firstPosted": "2024-09-18",
              "url": "https://arxiv.org/abs/2409.12191",
              "relationship": "Model foundation: Visual-RFT uses Qwen2-VL models for its visual perception experiments.",
              "evidence": {
                "url": "https://openaccess.thecvf.com/content/ICCV2025/papers/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.pdf#page=10",
                "label": "Visual-RFT reference 40; Section 4"
              },
              "kind": "paper"
            },
            {
              "title": "DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning",
              "firstPosted": "2025-01-22",
              "url": "https://arxiv.org/abs/2501.12948",
              "relationship": "Training motivation: Visual-RFT adapts the use of verifiable rewards and GRPO from language reasoning to visual perception.",
              "evidence": {
                "url": "https://openaccess.thecvf.com/content/ICCV2025/papers/Liu_Visual-RFT_Visual_Reinforcement_Fine-Tuning_ICCV_2025_paper.pdf#page=9",
                "label": "Visual-RFT reference 4; Sections 1 and 3.1"
              },
              "kind": "paper"
            }
          ],
          "citing": [
            {
              "kind": "post",
              "title": "Nathan Lambert on Visual-RFT",
              "attribution": "Nathan Lambert · Personal commentary",
              "firstPosted": "2025-03-10",
              "url": "https://x.com/natolambert/status/1899124771725648081",
              "quote": "One of the first papers I've seen with RLVR / reinforcement finetuning of vision language models",
              "relationship": "Highlights Visual-RFT as an early example of RLVR for vision-language models, while noting that many details remain to be explored.",
              "evidence": {
                "url": "https://x.com/natolambert/status/1899124771725648081",
                "label": "Read the original post"
              }
            },
            {
              "kind": "paper",
              "title": "VideoRFT: Incentivizing Video Reasoning Capability in MLLMs via Reinforced Fine-Tuning",
              "attribution": "Qi Wang et al. · Beijing Institute of Technology; Shenzhen University",
              "firstPosted": "2025-05-18",
              "url": "https://arxiv.org/abs/2505.12434",
              "quote": "Pioneering efforts such as Visual-RFT",
              "relationship": "Identifies Visual-RFT among pioneering methods adapting rule-based RL to image perception; explores reinforcement fine-tuning for video reasoning.",
              "evidence": {
                "url": "https://arxiv.org/html/2505.12434v4#S5.SS1",
                "label": "Section 5.1 · Reference 26"
              }
            },
            {
              "kind": "paper",
              "title": "DeepEyes: Incentivizing “Thinking with Images” via Reinforcement Learning",
              "attribution": "Ziwei Zheng et al. · Xiaohongshu; Xi’an Jiaotong University",
              "firstPosted": "2025-05-20",
              "url": "https://arxiv.org/abs/2505.14362",
              "relationship": "Cites Visual-RFT as prior work extending RL-based reasoning to object recognition; develops interleaved visual–textual reasoning through active perception.",
              "evidence": {
                "url": "https://arxiv.org/html/2505.14362v3#S2",
                "label": "Section 2 · Liu et al. (2025b)"
              }
            }
          ]
        },
        "resultsCaption": "Qwen2-VL-2B · ICCV 2025 results. Gains are absolute points over the base model.",
        "resultsNote": "Accuracy gains are percentage points; mAP gains are mAP points.",
        "resultColumns": [
          {
            "key": "baseline",
            "label": "Base model"
          },
          {
            "key": "sft",
            "label": "SFT"
          },
          {
            "key": "result",
            "label": "Visual-RFT"
          }
        ]
      },
      "keywords": [
        "Visual Reinforcement Fine-Tuning (Visual-RFT)",
        "Reinforcement Learning with Verifiable Rewards (RLVR)",
        "Group Relative Policy Optimization (GRPO)",
        "Large Vision-Language Models (LVLMs)",
        "Multimodal reinforcement learning",
        "Few-shot learning",
        "Fine-grained image classification",
        "Few-shot object detection",
        "Open-vocabulary object detection",
        "Reasoning grounding"
      ],
      "shortName": "Visual-RFT"
    },
    {
      "id": "arxiv:2504.07957",
      "title": "MM-IFEngine: Towards Multimodal Instruction Following",
      "authors": [
        {
          "name": "Shengyuan Ding"
        },
        {
          "name": "Shenxi Wu"
        },
        {
          "name": "Xiangyu Zhao"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-04-10",
        "arxivLastUpdated": "2025-04-27"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2504.07957",
        "googleScholar": "hW23VKIAAAAJ:RYcK_YlVTxYC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2504.07957"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:RYcK_YlVTxYC"
        },
        {
          "type": "code",
          "url": "https://github.com/SYuan03/MM-IFEngine",
          "repository": "SYuan03/MM-IFEngine"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/datasets/ChrisDing1105/MMIF-23k",
          "label": "Dataset",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 808,
        "order": 28
      },
      "shortName": "MM-IFEngine",
      "keywords": [
        "MM-IFEngine",
        "Multimodal instruction following",
        "Output constraints",
        "MM-IFEval",
        "Instruction tuning",
        "Direct preference optimization",
        "Perception constraints",
        "Synthetic training data"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250407957",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Ding_MM-IFEngine_Towards_Multimodal_Instruction_Following_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 1099,
        "lastPage": 1109,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Ding_MM-IFEngine_Towards_Multimodal_Instruction_Following_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2504.07957v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2504.07957v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2504.07957v2"
          }
        },
        "abstract": {
          "text": "The Instruction Following (IF) ability measures how well Multi-modal Large Language Models (MLLMs) understand exactly what users are telling them and whether they are doing it right. Existing multimodal instruction following training data is scarce, the benchmarks are simple with atomic instructions, and the evaluation strategies are imprecise for tasks demanding exact output constraints. To address this, we present MM-IFEngine, an effective pipeline to generate high-quality image-instruction pairs. Our MM-IFEngine pipeline yields large-scale, diverse, and high-quality training data MM-IFInstruct-23k, which is suitable for Supervised Fine-Tuning (SFT) and extended as MM-IFDPO-23k for Direct Preference Optimization (DPO). We further introduce MM-IFEval, a challenging and diverse multi-modal instruction-following benchmark that includes (1) both compose-level constraints for output responses and perception-level constraints tied to the input images, and (2) a comprehensive evaluation pipeline incorporating both rule-based assessment and judge model. We conduct SFT and DPO experiments and demonstrate that fine-tuning MLLMs on MM-IFInstruct-23k and MM-IFDPO-23k achieves notable gains on various IF benchmarks, such as MM-IFEval (+10.2%), MIA (+7.6%), and IFEval (+12.3%). We have fully open-sourced the datasets (both SFT and DPO), evaluation code and training scripts at https://github.com/SYuan03/MM-IFEngine.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "MM-IFEngine improves multimodal instruction following through constraint-rich training data and evaluation that checks both requested output form and image-grounded requirements.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Existing instruction-following datasets and benchmarks lack varied, exact multimodal constraints. MM-IFEngine generates image-instruction pairs and evaluates responses using rules alongside a judge model.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Creates MM-IFInstruct-23k for SFT and MM-IFDPO-23k for preference optimization.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces MM-IFEval with composition-level and perception-level constraints.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Fine-tuning Qwen2-VL-7B-Instruct on MM-IFDPO-23k raises MM-IFEval accuracy from 42.0% to 52.2%, MIA-Bench from 80.5% to 88.1%, and the reported IFEval average from 47.4% to 59.7%.",
            "fragment": "S4.T1",
            "locator": "Table 1 · instruction-following evaluation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "For Qwen2-VL-7B-Instruct, DPO with all constraints removed when generating negatives reaches a three-benchmark average of 66.7, versus 63.4 when negatives are generated without images. Removing 33% or 66% of constraints yields 65.8 and 65.9.",
            "fragment": "S5.T4",
            "locator": "Table 4 · negative-response construction · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Atomic instruction-following evaluation",
              "design": "Uses simpler constraints and can miss exact multimodal requirements."
            },
            {
              "method": "MM-IFEngine",
              "design": "Generates diverse constrained instructions and evaluates composition and perception requirements with rules and a judge model."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2412.01824",
      "title": "X-Prompt: Towards Universal In-Context Image Generation in Auto-Regressive Vision Language Foundation Models",
      "authors": [
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Ziyang Chu"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-12-02",
        "arxivLastUpdated": "2025-08-27"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2412.01824",
        "googleScholar": "hW23VKIAAAAJ:7PzlFSSx8tAC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2412.01824"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:7PzlFSSx8tAC"
        },
        {
          "type": "code",
          "url": "https://github.com/SunzeY/X-Prompt",
          "repository": "SunzeY/X-Prompt"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 829,
        "order": 29
      },
      "shortName": "X-Prompt",
      "keywords": [
        "X-Prompt",
        "In-context image generation",
        "Autoregressive vision-language models",
        "Visual prompting",
        "Task generalization",
        "Context compression",
        "Text-image prediction",
        "Few-shot generation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241201824",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Sun_X-Prompt_Generalizable_Auto-Regressive_Visual_Learning_with_In-Context_Prompting_ICCV_2025_paper.html"
        },
        "status": "published",
        "title": "X-Prompt: Generalizable Auto-Regressive Visual Learning with In-Context Prompting",
        "authors": [
          {
            "name": "Zeyi Sun"
          },
          {
            "name": "Ziyang Chu"
          },
          {
            "name": "Pan Zhang"
          },
          {
            "name": "Tong Wu"
          },
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Xiaoyi Dong"
          },
          {
            "name": "Yuanjun Xiong"
          },
          {
            "name": "Dahua Lin"
          },
          {
            "name": "Jiaqi Wang"
          }
        ],
        "month": 10,
        "firstPage": 17268,
        "lastPage": 17280,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Sun_X-Prompt_Generalizable_Auto-Regressive_Visual_Learning_with_In-Context_Prompting_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2412.01824v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2412.01824v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2412.01824v2"
          }
        },
        "abstract": {
          "text": "In-context generation is a key component of large language models' (LLMs) open-task generalization capability. By leveraging a few examples as context, LLMs can perform both in-domain and out-of-domain tasks. Recent advancements in auto-regressive vision-language models (VLMs) built upon LLMs have showcased impressive performance in text-to-image generation. However, the potential of in-context learning for general image generation tasks remains largely unexplored. To address this, we introduce X-Prompt, a purely auto-regressive large-vision language model designed to deliver competitive performance across a wide range of both seen and unseen image generation tasks, all within a unified in-context learning framework. X-Prompt incorporates a specialized design that efficiently compresses valuable features from in-context examples, supporting longer in-context token sequences and improving its ability to generalize to unseen tasks. A unified training task for both text and image prediction enables X-Prompt to handle general image generation with enhanced task awareness from in-context examples. Extensive experiments validate the model's performance across diverse seen image generation tasks and its capacity to generalize to previously unseen tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "X-Prompt extends autoregressive vision-language modeling to in-context image generation, allowing example prompts to specify both familiar and previously unseen generation tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Strong text-to-image generation does not by itself provide general in-context visual learning. X-Prompt uses example context to infer generation tasks within a unified autoregressive framework.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Compresses useful example features to support longer in-context sequences.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Trains text and image prediction jointly to improve task awareness from examples.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "In novel-task in-context evaluation, adding X-Prompt tokens improves NYU-v2 depth RMSE from 0.390 to 0.352 and Rain100H deraining PSNR from 18.10 to 18.91 dB, compared with in-context prompting without those tokens.",
            "fragment": "S4.T4",
            "locator": "Table 4 · tasks held out from training · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Adding text prediction to the image-generation training objective raises GenEval overall performance from 0.49 to 0.57. Position and color scores rise from 0.14 to 0.26 and from 0.71 to 0.85, respectively.",
            "fragment": "S3.T1",
            "locator": "Table 1 · joint text-prediction ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Text-to-image generation alone",
              "design": "Provides generation capability without necessarily learning new visual tasks from contextual examples."
            },
            {
              "method": "X-Prompt",
              "design": "Jointly predicts text and images while compressing useful example context for in-context task inference."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2406.00093",
      "title": "Bootstrap3D: Improving Multi-view Diffusion Model with Synthetic Data",
      "authors": [
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-05-31",
        "arxivLastUpdated": "2024-10-03"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2406.00093",
        "googleScholar": "hW23VKIAAAAJ:MXK_kJrjxJIC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2406.00093"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:MXK_kJrjxJIC"
        },
        {
          "type": "code",
          "url": "https://github.com/SunzeY/Bootstrap3D",
          "repository": "SunzeY/Bootstrap3D"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 849,
        "order": 30
      },
      "shortName": "Bootstrap3D",
      "keywords": [
        "Bootstrap3D",
        "Multi-view diffusion",
        "Synthetic 3D training data",
        "MV-LLaVA",
        "Dense captions",
        "View consistency",
        "Training Timestep Reschedule",
        "3D content generation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240600093",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Sun_Bootstrap3D_Improving_Multi-view_Diffusion_Model_with_Synthetic_Data_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 15714,
        "lastPage": 15726,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Sun_Bootstrap3D_Improving_Multi-view_Diffusion_Model_with_Synthetic_Data_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2406.00093v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2406.00093v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2406.00093v2"
          }
        },
        "abstract": {
          "text": "Recent years have witnessed remarkable progress in multi-view diffusion models for 3D content creation. However, there remains a significant gap in image quality and prompt-following ability compared to 2D diffusion models. A critical bottleneck is the scarcity of high-quality 3D objects with detailed captions. To address this challenge, we propose Bootstrap3D, a novel framework that automatically generates an arbitrary quantity of multi-view images to assist in training multi-view diffusion models. Specifically, we introduce a data generation pipeline that employs (1) 2D and video diffusion models to generate multi-view images based on constructed text prompts, and (2) our fine-tuned 3D-aware MV-LLaVA for filtering high-quality data and rewriting inaccurate captions. Leveraging this pipeline, we have generated 1 million high-quality synthetic multi-view images with dense descriptive captions to address the shortage of high-quality 3D data. Furthermore, we present a Training Timestep Reschedule (TTR) strategy that leverages the denoising process to learn multi-view consistency while maintaining the original 2D diffusion prior. Extensive experiments demonstrate that Bootstrap3D can generate high-quality multi-view images with superior aesthetic quality, image-text alignment, and maintained view consistency.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Bootstrap3D uses filtered and recaptioned synthetic multi-view images to improve 3D generation while preserving the strengths of pretrained 2D diffusion models.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Multi-view diffusion is limited by scarce, well-captioned 3D data. Bootstrap3D creates multi-view images using image and video generators, then filters and recaptions them with a 3D-aware vision-language model.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Generates one million synthetic multi-view images with dense captions using MV-LLaVA-based quality control.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces Training Timestep Reschedule to balance view consistency with the 2D diffusion prior.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On four-view generation, Bootstrap3D improves CLIP-L/14 retrieval score from MVDream’s 84.8 to 88.8 and lowers FID against the PixArt reference distribution from 59.2 to 31.0. CLIP evaluation uses 110 GPTeval3D prompts; FID uses 30k object-centric reference images.",
            "fragment": "S4.T1",
            "locator": "Table 1 · four-view generation · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "With 100k synthetic images and Cap3D captions, enabling TTR lowers multi-view FID from 92.0 to 60.8 and generated-object FID from 134.6 to 70.6. Adding dense recaptioning further lowers these values to 50.2 and 50.9.",
            "fragment": "S4.T3",
            "locator": "Table 3 · synthetic-data ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Multi-view learning from scarce 3D data",
              "design": "Faces limited coverage and caption quality in available multi-view supervision."
            },
            {
              "method": "Bootstrap3D",
              "design": "Adds filtered, recaptioned synthetic multi-view images and reschedules training timesteps to retain the 2D prior."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2507.02859",
      "title": "Bootstrapping Grounded Chain-of-Thought in Multimodal LLMs for Data-Efficient Model Adaptation",
      "authors": [
        {
          "name": "Jiaer Xia"
        },
        {
          "name": "Bingkui Tong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Rui Shao"
        },
        {
          "name": "Kaiyang Zhou"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-07-03",
        "arxivLastUpdated": "2025-07-03"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2507.02859",
        "googleScholar": "hW23VKIAAAAJ:abG-DnoFyZgC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2507.02859"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:abG-DnoFyZgC"
        },
        {
          "type": "code",
          "url": "https://github.com/maifoundations/GCoT",
          "repository": "maifoundations/GCoT"
        }
      ],
      "display": {
        "new": false,
        "badges": [
          "Highlight"
        ],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 868,
        "order": 31
      },
      "shortName": "GCoT",
      "keywords": [
        "Grounded Chain-of-Thought",
        "GCoT",
        "Data-efficient adaptation",
        "Visual grounding",
        "Reasoning distillation",
        "Chart understanding",
        "Document understanding",
        "Bounding-box supervision"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250702859",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Xia_Bootstrapping_Grounded_Chain-of-Thought_in_Multimodal_LLMs_for_Data-Efficient_Model_Adaptation_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 208,
        "lastPage": 217,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Xia_Bootstrapping_Grounded_Chain-of-Thought_in_Multimodal_LLMs_for_Data-Efficient_Model_Adaptation_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2507.02859v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2507.02859v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2507.02859v1"
          }
        },
        "abstract": {
          "text": "Multimodal Large Language Models (MLLMs) have demonstrated remarkable capabilities in interpreting images using natural language. However, without using large-scale datasets for retraining, these models are difficult to adapt to specialized vision tasks, e.g., chart understanding. This problem is caused by a mismatch between pre-training and downstream datasets: pre-training datasets primarily concentrate on scenes and objects but contain limited information about specialized, non-object images, such as charts and tables. In this paper, we share an interesting finding that training an MLLM with Chain-of-Thought (CoT) reasoning data can facilitate model adaptation in specialized vision tasks, especially under data-limited regimes. However, we identify a critical issue within CoT data distilled from pre-trained MLLMs, i.e., the data often contains multiple factual errors in the reasoning steps. To address the problem, we propose Grounded Chain-of-Thought (GCoT), a simple bootstrapping-based approach that aims to inject grounding information (i.e., bounding boxes) into CoT data, essentially making the reasoning steps more faithful to input images. We evaluate our approach on five specialized vision tasks, which cover a variety of visual formats including charts, tables, receipts, and reports. The results demonstrate that under data-limited regimes our approach significantly improves upon fine-tuning and distillation.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "Grounded Chain-of-Thought improves data-efficient adaptation by anchoring distilled reasoning steps to image regions, reducing factual errors in specialized visual tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Scene-centric pretraining does not cover charts, tables and documents well, and distilled reasoning can contain factual mistakes. GCoT bootstraps reasoning data with bounding-box grounding to make it more faithful to the image.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Injects visual grounding into chain-of-thought data through a bootstrapping procedure.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Studies model adaptation on specialized visual formats under limited-data conditions.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "In the 128-sample TabMWP setting, GCoT scores 33.93 ± 0.54, compared with 31.57 ± 0.47 without augmentation and 23.57 ± 2.21 without box verification. Checking only the final answer therefore leaves a substantial gap in this experiment.",
            "fragment": "S4.T2",
            "locator": "Table 2 · TabMWP component ablation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Across five tested distillation sources, the reported average increases from 22.78 to 26.54 with GCoT. Improvements range from 1.54 points for GPT-4o-sourced data to 5.62 for Claude-3.5-sourced data.",
            "fragment": "S4.T3",
            "locator": "Table 3 · distillation-source comparison · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Ungrounded reasoning distillation",
              "design": "Can transfer factual errors when adapting to specialized visual formats."
            },
            {
              "method": "GCoT",
              "design": "Bootstraps reasoning steps anchored to image regions to improve fidelity under limited-data adaptation."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2502.08590",
      "title": "Light-A-Video: Training-free Video Relighting via Progressive Light Fusion",
      "authors": [
        {
          "name": "Yujie Zhou"
        },
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Qidong Huang"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Anyi Rao"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Li Niu"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-02-12",
        "arxivLastUpdated": "2025-03-12"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2502.08590",
        "googleScholar": "hW23VKIAAAAJ:maZDTaKrznsC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2502.08590"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:maZDTaKrznsC"
        },
        {
          "type": "code",
          "url": "https://github.com/bcmi/Light-A-Video",
          "repository": "bcmi/Light-A-Video"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 887,
        "order": 32
      },
      "shortName": "Light-A-Video",
      "keywords": [
        "Light-A-Video",
        "Video relighting",
        "Training-free editing",
        "Temporal consistency",
        "Consistent Light Attention",
        "Progressive Light Fusion",
        "Diffusion models",
        "Illumination editing"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250208590",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Zhou_Light-A-Video_Training-free_Video_Relighting_via_Progressive_Light_Fusion_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 13315,
        "lastPage": 13325,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Zhou_Light-A-Video_Training-free_Video_Relighting_via_Progressive_Light_Fusion_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2502.08590v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2502.08590v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2502.08590v2"
          }
        },
        "abstract": {
          "text": "Recent advancements in image relighting models, driven by large-scale datasets and pre-trained diffusion models, have enabled the imposition of consistent lighting. However, video relighting still lags, primarily due to the excessive training costs and the scarcity of diverse, high-quality video relighting datasets. A simple application of image relighting models on a frame-by-frame basis leads to several issues: lighting source inconsistency and relighted appearance inconsistency, resulting in flickers in the generated videos. In this work, we propose Light-A-Video, a training-free approach to achieve temporally smooth video relighting. Adapted from image relighting models, Light-A-Video introduces two key techniques to enhance lighting consistency. First, we design a Consistent Light Attention (CLA) module, which enhances cross-frame interactions within the self-attention layers of the image relight model to stabilize the generation of the background lighting source. Second, leveraging the physical principle of light transport independence, we apply linear blending between the source video's appearance and the relighted appearance, using a Progressive Light Fusion (PLF) strategy to ensure smooth temporal transitions in illumination. Experiments show that Light-A-Video improves the temporal consistency of relighted video while maintaining the relighted image quality, ensuring coherent lighting transitions across frames. Project page: https://bujiazi.github.io/light-a-video.github.io/.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Light-A-Video adapts image relighting models to temporally coherent video relighting without additional training by combining cross-frame attention and progressive light blending.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Applying image relighting independently to frames produces changing light sources and flickering appearances. Light-A-Video coordinates background illumination across frames and blends original and relighted appearances progressively.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Uses Consistent Light Attention to stabilize lighting through cross-frame interaction.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces Progressive Light Fusion based on the independence of light transport.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Light-A-Video achieves a motion-preservation error of 1.833 versus 5.969 for framewise IC-Light, while CLIP score rises from 0.9040 to 0.9667. Lower motion error indicates better preservation in the paper’s evaluation.",
            "fragment": "S5.T1",
            "locator": "Table 1 · video relighting quality and motion · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Compared with IC-Light + SDEdit-0.2, Light-A-Video improves the reported video-stability preference score from 2.752 to 3.502, but FID rises from 13.79 to 29.63. The benefit is strongest in temporal consistency and user preference rather than every image-quality metric.",
            "fragment": "S5.T1",
            "locator": "Table 1 · relighting trade-offs and user study · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Independent frame relighting",
              "design": "Can produce changing illumination and inconsistent appearance between frames."
            },
            {
              "method": "Light-A-Video",
              "design": "Combines cross-frame light attention with progressive blending to coordinate relighting without further training."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2410.07167",
      "title": "Deciphering Cross-Modal Alignment in Large Vision-Language Models with Modality Integration Rate",
      "authors": [
        {
          "name": "Qidong Huang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Weiming Zhang"
        },
        {
          "name": "Nenghai Yu"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-10-09",
        "arxivLastUpdated": "2024-10-16"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2410.07167",
        "googleScholar": "hW23VKIAAAAJ:4DMP91E08xMC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2410.07167"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:4DMP91E08xMC"
        },
        {
          "type": "code",
          "url": "https://github.com/shikiw/Modality-Integration-Rate",
          "repository": "shikiw/Modality-Integration-Rate"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 906,
        "order": 33
      },
      "shortName": "MIR",
      "keywords": [
        "Modality Integration Rate",
        "MIR",
        "Cross-modal alignment",
        "LVLM pretraining",
        "Distribution distance",
        "Training quality evaluation",
        "Multimodal representation",
        "Data selection"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241007167",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Huang_Deciphering_Cross-Modal_Alignment_in_Large_Vision-Language_Models_via_Modality_Integration_ICCV_2025_paper.html"
        },
        "status": "published",
        "title": "Deciphering Cross-Modal Alignment in Large Vision-Language Models via Modality Integration Rate",
        "authors": [
          {
            "name": "Qidong Huang"
          },
          {
            "name": "Xiaoyi Dong"
          },
          {
            "name": "Pan Zhang"
          },
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Yuhang Cao"
          },
          {
            "name": "Jiaqi Wang"
          },
          {
            "name": "Weiming Zhang"
          },
          {
            "name": "Nenghai Yu"
          }
        ],
        "month": 10,
        "firstPage": 218,
        "lastPage": 227,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Huang_Deciphering_Cross-Modal_Alignment_in_Large_Vision-Language_Models_via_Modality_Integration_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2410.07167v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2410.07167v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2410.07167v2"
          }
        },
        "abstract": {
          "text": "We present the Modality Integration Rate (MIR), an effective, robust, and generalized metric to indicate the multi-modal pre-training quality of Large Vision Language Models (LVLMs). Large-scale pre-training plays a critical role in building capable LVLMs, while evaluating its training quality without the costly supervised fine-tuning stage is under-explored. Loss, perplexity, and in-context evaluation results are commonly used pre-training metrics for Large Language Models (LLMs), while we observed that these metrics are less indicative when aligning a well-trained LLM with a new modality. Due to the lack of proper metrics, the research of LVLMs in the critical pre-training stage is hindered greatly, including the training data choice, efficient module design, etc. In this paper, we propose evaluating the pre-training quality from the inter-modal distribution distance perspective and present MIR, the Modality Integration Rate, which is 1) Effective to represent the pre-training quality and show a positive relation with the benchmark performance after supervised fine-tuning. 2) Robust toward different training/evaluation data. 3) Generalize across training configurations and architecture choices. We conduct a series of pre-training experiments to explore the effectiveness of MIR and observe satisfactory results that MIR is indicative about training data selection, training strategy schedule, and model architecture design to get better pre-training results. We hope MIR could be a helpful metric for building capable LVLMs and inspire the following research about modality alignment in different areas. Our code is at: https://github.com/shikiw/Modality-Integration-Rate.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Modality Integration Rate estimates LVLM pretraining quality from cross-modal distribution distance, helping assess alignment without first running costly supervised fine-tuning.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Language-model metrics such as loss and perplexity can be weak indicators when aligning a pretrained language model with vision. MIR measures how well the modality distributions integrate during multimodal pretraining.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Relates inter-modal distribution distance to downstream performance after supervised fine-tuning.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Tests the metric across data choices, training schedules and model architectures.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "As average caption length increases from 15.2 to 181.2, MIR decreases from 3.588 to 3.218 while the reported post-SFT benchmark average rises from 63.8 to 64.4. This experiment connects the pretraining indicator with downstream performance.",
            "fragment": "S3.T1",
            "locator": "Table 1 · caption-detail experiment · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "With approximately 1.9M pretraining examples and the same 665k SFT set, keeping the LLM frozen yields MIR 3.001 and downstream average 64.0; unlocking all LLM layers yields MIR 2.656 and average 65.9.",
            "fragment": "S3.T3",
            "locator": "Table 3 · trainable-layer comparison · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Loss, perplexity and in-context metrics",
              "design": "Can be weak indicators of downstream quality when a pretrained language model is aligned with vision."
            },
            {
              "method": "Modality Integration Rate",
              "design": "Uses inter-modal distribution distance to assess pretraining quality before supervised fine-tuning."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2410.16268",
      "title": "SAM2Long: Enhancing SAM 2 for Long Video Segmentation with a Training-Free Memory Tree",
      "authors": [
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Rui Qian"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Yuwei Guo"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-10-21",
        "arxivLastUpdated": "2025-07-29"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2410.16268",
        "googleScholar": "hW23VKIAAAAJ:mVmsd5A6BfQC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/pdf/2410.16268"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:mVmsd5A6BfQC"
        },
        {
          "type": "code",
          "url": "https://github.com/Mark12Ding/SAM2Long",
          "repository": "Mark12Ding/SAM2Long"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 925,
        "order": 34
      },
      "shortName": "SAM2Long",
      "keywords": [
        "SAM2Long",
        "SAM 2",
        "Long video segmentation",
        "Memory tree",
        "Constrained tree search",
        "Error accumulation",
        "Occlusion handling",
        "Training-free segmentation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241016268",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2025/html/Ding_SAM2Long_Enhancing_SAM_2_for_Long_Video_Segmentation_with_a_ICCV_2025_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 13614,
        "lastPage": 13624,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2025/papers/Ding_SAM2Long_Enhancing_SAM_2_for_Long_Video_Segmentation_with_a_ICCV_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2410.16268v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2410.16268v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2410.16268v3"
          }
        },
        "abstract": {
          "text": "The Segment Anything Model 2 (SAM 2) has emerged as a powerful foundation model for object segmentation in both images and videos, paving the way for various downstream video applications. The crucial design of SAM 2 for video segmentation is its memory module, which prompts object-aware memories from previous frames for current frame prediction. However, its greedy-selection memory design suffers from the \"error accumulation\" problem, where an errored or missed mask will cascade and influence the segmentation of the subsequent frames, which limits the performance of SAM 2 toward complex long-term videos. To this end, we introduce SAM2Long, an improved training-free video object segmentation strategy, which considers the segmentation uncertainty within each frame and chooses the video-level optimal results from multiple segmentation pathways in a constrained tree search manner. In practice, we maintain a fixed number of segmentation pathways throughout the video. For each frame, multiple masks are proposed based on the existing pathways, creating various candidate branches. We then select the same fixed number of branches with higher cumulative scores as the new pathways for the next frame. After processing the final frame, the pathway with the highest cumulative score is chosen as the final segmentation result. Benefiting from its heuristic search design, SAM2Long is robust toward occlusions and object reappearances, and can effectively segment and track objects for complex long-term videos. Notably, SAM2Long achieves an average improvement of 3.0 points across all 24 head-to-head comparisons, with gains of up to 5.3 points in J&F on long-term video object segmentation benchmarks such as SA-V and LVOS. The code is released at https://github.com/Mark12Ding/SAM2Long.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "SAM2Long reduces error accumulation in SAM 2 by retaining multiple segmentation paths and selecting the best video-level trajectory through a constrained memory-tree search.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "A mistaken mask in greedy memory selection can corrupt later frames. SAM2Long maintains a fixed number of candidate paths, branches them at each frame and keeps paths with stronger cumulative scores.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Introduces uncertainty-aware pathway selection without retraining SAM 2.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Handles occlusion and object reappearance through video-level rather than purely greedy decisions.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On SA-V test, SAM2Long-L improves J&F from the reproduced SAM2-L baseline’s 75.5 to 80.8. In the same evaluation, SA-V validation rises from 76.3 to 80.8.",
            "fragment": "S3.T1",
            "locator": "Table 1 · matched SAM2-L comparison · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "Using the official LVOS evaluation code, SAM2.1Long improves LVOS-v1 validation J&F from 80.2 to 83.4 and LVOS-v2 from 84.1 to 85.9. These scores use a different evaluation implementation from Table 1 and should be cited with that distinction.",
            "fragment": "S4.T3",
            "locator": "Table 3 · official LVOS evaluation protocol · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Greedy memory-path selection",
              "design": "A mistaken segmentation can enter memory and propagate errors to later frames."
            },
            {
              "method": "SAM2Long",
              "design": "Retains multiple candidate paths in a constrained memory tree and selects using cumulative video-level scores."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2501.12368",
      "title": "InternLM-XComposer2.5-Reward: A Simple Yet Effective Multi-Modal Reward Model",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Shengyuan Ding"
        },
        {
          "name": "Shenxi Wu"
        },
        {
          "name": "Yubo Ma"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Wenwei Zhang"
        },
        {
          "name": "Kai Chen"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ACL",
        "citationText": "Findings of the Association for Computational Linguistics (ACL), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-01-21",
        "arxivLastUpdated": "2025-05-20"
      },
      "topics": [
        "Reinforcement Learning from Human Feedback"
      ],
      "identifiers": {
        "arxiv": "2501.12368",
        "googleScholar": "hW23VKIAAAAJ:TFP_iSt0sucC",
        "doi": "10.18653/v1/2025.findings-acl.340"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2501.12368"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:TFP_iSt0sucC"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/InternLM-XComposer",
          "repository": "InternLM/InternLM-XComposer"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/internlm/internlm-xcomposer2d5-7b-reward",
          "label": "IXC 2.5 Reward",
          "variant": "model"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/internlm/internlm-xcomposer2d5-7b-chat",
          "label": "IXC 2.5 Chat",
          "variant": "model"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 944,
        "order": 35
      },
      "shortName": "IXC-2.5-Reward",
      "keywords": [
        "InternLM-XComposer2.5-Reward",
        "Multimodal reward models",
        "Human preference alignment",
        "Reinforcement learning",
        "PPO",
        "Best-of-N selection",
        "Instruction data filtering",
        "Video reward modeling"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250112368",
        "year": 2025,
        "booktitle": "Findings of the Association for Computational Linguistics: ACL 2025",
        "source": {
          "label": "ACL 2025 proceedings record",
          "url": "https://aclanthology.org/2025.findings-acl.340/"
        },
        "status": "published",
        "month": 7,
        "firstPage": 6547,
        "lastPage": 6563,
        "publisher": "Association for Computational Linguistics",
        "pdfURL": "https://aclanthology.org/2025.findings-acl.340.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2501.12368v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2501.12368v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2501.12368v2"
          }
        },
        "abstract": {
          "text": "Despite the promising performance of Large Vision Language Models (LVLMs) in visual understanding, they occasionally generate incorrect outputs. While reward models (RMs) with reinforcement learning or test-time scaling offer the potential for improving generation quality, a critical gap remains: publicly available multi-modal RMs for LVLMs are scarce, and the implementation details of proprietary models are often unclear. We bridge this gap with InternLM-XComposer2.5-Reward (IXC-2.5-Reward), a simple yet effective multi-modal reward model that aligns LVLMs with human preferences. To ensure the robustness and versatility of IXC-2.5-Reward, we set up a high-quality multi-modal preference corpus spanning text, image, and video inputs across diverse domains, such as instruction following, general understanding, text-rich documents, mathematical reasoning, and video understanding. IXC-2.5-Reward achieves excellent results on the latest multi-modal reward model benchmark and shows competitive performance on text-only reward model benchmarks. We further demonstrate three key applications of IXC-2.5-Reward: (1) Providing a supervisory signal for RL training. We integrate IXC-2.5-Reward with Proximal Policy Optimization (PPO) yields IXC-2.5-Chat, which shows consistent improvements in instruction following and multi-modal open-ended dialogue; (2) Selecting the best response from candidate responses for test-time scaling; and (3) Filtering outlier or noisy samples from existing image and video instruction tuning training data. To ensure reproducibility and facilitate further research, we have open-sourced all model weights and training recipes at https://github.com/InternLM/InternLM-XComposer/tree/main/InternLM-XComposer-2.5-Reward",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "IXC-2.5-Reward provides a multimodal preference signal that supports reinforcement learning, best-response selection and instruction-data filtering across text, image and video inputs.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Public multimodal reward models are scarce and proprietary training recipes are difficult to reproduce. IXC-2.5-Reward addresses this gap with an open reward model trained on a preference corpus spanning instruction following, documents, reasoning and video understanding.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Constructs a diverse multimodal preference corpus and releases reward-model weights and training recipes.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Demonstrates one reward model in PPO training, test-time response selection and noisy-data filtering.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "IXC-2.5-Reward-7B obtains 65.8% overall accuracy and 70.0% macro accuracy on VLRewardBench. GPT-4o (2024-08-06) scores 65.8% overall and 62.4% macro accuracy; the equal overall score and different macro score reflect category balance.",
            "fragment": "S3.T3",
            "locator": "Table 3 · VLRewardBench · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "IXC-2.5-Reward scores 88.6 on text-only RewardBench, compared with 87.6 for InternLM2-7B-Reward and 80.0 for LLaVA-Critic-8B. This evaluates text reward capability separately from multimodal judgment.",
            "fragment": "S5.T4",
            "locator": "Table 4 · text-only RewardBench · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "PPO training with IXC-2.5-Reward",
              "design": "Uses reward scores as supervision to update the model's response policy."
            },
            {
              "method": "Test-time selection with IXC-2.5-Reward",
              "design": "Ranks existing candidate responses to select a preferred output without a policy update."
            },
            {
              "method": "Data filtering with IXC-2.5-Reward",
              "design": "Identifies noisy or outlying image and video instruction examples before training."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2506.04997",
      "title": "Towards Storage-Efficient Visual Document Retrieval: An Empirical Study on Reducing Patch-Level Embeddings",
      "authors": [
        {
          "name": "Yubo Ma"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaobao Wu"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Yixin Cao"
        },
        {
          "name": "Aixin Sun"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ACL",
        "citationText": "Findings of the Association for Computational Linguistics (ACL), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-06-05",
        "arxivLastUpdated": "2025-06-05"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2506.04997",
        "googleScholar": "hW23VKIAAAAJ:cFHS6HbyZ2cC",
        "doi": "10.18653/v1/2025.findings-acl.1003"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2506.04997"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:cFHS6HbyZ2cC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 967,
        "order": 36
      },
      "shortName": "Light-ColPali",
      "keywords": [
        "Light-ColPali",
        "Light-ColQwen2",
        "Visual document retrieval",
        "Multi-vector retrieval",
        "Token merging",
        "Patch embeddings",
        "Storage efficiency",
        "Token pruning"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250604997",
        "year": 2025,
        "booktitle": "Findings of the Association for Computational Linguistics: ACL 2025",
        "source": {
          "label": "ACL 2025 proceedings record",
          "url": "https://aclanthology.org/2025.findings-acl.1003/"
        },
        "status": "published",
        "month": 7,
        "firstPage": 19568,
        "lastPage": 19580,
        "publisher": "Association for Computational Linguistics",
        "pdfURL": "https://aclanthology.org/2025.findings-acl.1003.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2506.04997v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2506.04997v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2506.04997v1"
          }
        },
        "abstract": {
          "text": "Despite the strong performance of ColPali/ColQwen2 in Visualized Document Retrieval (VDR), it encodes each page into multiple patch-level embeddings and leads to excessive memory usage. This empirical study investigates methods to reduce patch embeddings per page at minimum performance degradation. We evaluate two token-reduction strategies: token pruning and token merging. Regarding token pruning, we surprisingly observe that a simple random strategy outperforms other sophisticated pruning methods, though still far from satisfactory. Further analysis reveals that pruning is inherently unsuitable for VDR as it requires removing certain page embeddings without query-specific information. Turning to token merging (more suitable for VDR), we search for the optimal combinations of merging strategy across three dimensions and develop Light-ColPali/ColQwen2. It maintains 98.2% of retrieval performance with only 11.8% of original memory usage, and preserves 94.6% effectiveness at 2.8% memory footprint. We expect our empirical findings and resulting Light-ColPali/ColQwen2 offer valuable insights and establish a competitive baseline for future research towards efficient VDR.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "Light-ColPali/ColQwen2 retains 98.2% of the original retrieval performance with 11.8% of the embedding memory by merging document patches instead of pruning them.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Multi-vector document retrieval stores many patch embeddings per page. This empirical study compares pruning and merging, finding that query-independent pruning can discard essential page information and that carefully configured merging offers a better storage trade-off.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Analyzes token pruning and token merging for visual document retrieval, including the surprising strength of random pruning among pruning methods.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Develops Light-ColPali/ColQwen2 by examining merging choices across three dimensions.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "At merging factor 25, Light-ColQwen2 retains 96.3% of ColQwen2’s average NDCG@5 (78.4 versus 81.4). Its relative memory cost is 3.0 versus 64.4, with both normalized to the same DSE-Qwen2 baseline.",
            "fragment": "S5.T2",
            "locator": "Table 2 · document retrieval, Qwen2-VL-2B · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "At merging factor 9, post-projector merging achieves an average retrieval score of 81.4, versus 81.0 after the LLM, 65.4 after the vision encoder, and 59.1 before it. Delaying merging preserves retrieval information more effectively in this comparison.",
            "fragment": "S5.T1",
            "locator": "Table 1 · merging-location ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Patch pruning",
              "design": "Removes page embeddings without query-specific knowledge, potentially discarding relevant evidence."
            },
            {
              "method": "Patch merging",
              "design": "Combines page embeddings and yields the stronger measured storage-performance trade-off in this study."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2502.05173",
      "title": "VideoRoPE: What Makes for Good Video Rotary Position Embedding?",
      "authors": [
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Xiaoran Liu"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jian Tong"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Qipeng Guo",
          "corresponding": true
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        },
        {
          "name": "Xipeng Qiu"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICML",
        "citationText": "International Conference on Machine Learning (ICML), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-02-07",
        "arxivLastUpdated": "2025-05-30"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2502.05173",
        "googleScholar": "hW23VKIAAAAJ:isC4tDSrTZIC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2502.05173"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:isC4tDSrTZIC"
        },
        {
          "type": "code",
          "url": "https://github.com/Wiselnn570/VideoRoPE",
          "repository": "Wiselnn570/VideoRoPE"
        }
      ],
      "display": {
        "new": false,
        "badges": [
          "Oral"
        ],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 982,
        "order": 37
      },
      "shortName": "VideoRoPE",
      "keywords": [
        "VideoRoPE",
        "Rotary position embedding",
        "Video language models",
        "Spatiotemporal encoding",
        "Long-video retrieval",
        "Temporal frequency allocation",
        "V-NIAH-D",
        "Position encoding"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250205173",
        "year": 2025,
        "booktitle": "Proceedings of the 42nd International Conference on Machine Learning",
        "source": {
          "label": "ICML 2025 proceedings record",
          "url": "https://proceedings.mlr.press/v267/wei25h.html"
        },
        "status": "published",
        "firstPage": 66118,
        "lastPage": 66136,
        "volume": "267",
        "publisher": "PMLR",
        "pdfURL": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/wei25h/wei25h.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2502.05173v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2502.05173v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2502.05173v3"
          }
        },
        "abstract": {
          "text": "While Rotary Position Embedding (RoPE) and its variants are widely adopted for their long-context capabilities, the extension of the 1D RoPE to video, with its complex spatio-temporal structure, remains an open challenge. This work first introduces a comprehensive analysis that identifies four key characteristics essential for the effective adaptation of RoPE to video, which have not been fully considered in prior work. As part of our analysis, we introduce a challenging V-NIAH-D (Visual Needle-In-A-Haystack with Distractors) task, which adds periodic distractors into V-NIAH. The V-NIAH-D task demonstrates that previous RoPE variants, lacking appropriate temporal dimension allocation, are easily misled by distractors. Based on our analysis, we introduce VideoRoPE, with a 3D structure designed to preserve spatio-temporal relationships. VideoRoPE features low-frequency temporal allocation to mitigate periodic oscillations, a diagonal layout to maintain spatial symmetry, and adjustable temporal spacing to decouple temporal and spatial indexing. VideoRoPE consistently surpasses previous RoPE variants, across diverse downstream tasks such as long video retrieval, video understanding, and video hallucination. Our code will be available at https://github.com/Wiselnn570/VideoRoPE.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "VideoRoPE adapts rotary position embeddings to video through low-frequency temporal allocation, a spatially symmetric diagonal layout and adjustable temporal spacing.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Extending one-dimensional position encoding to video requires preserving both temporal and spatial structure. VideoRoPE studies the design requirements and uses distractor-heavy retrieval to expose failures caused by inappropriate temporal frequency allocation.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Introduces V-NIAH-D, a visual needle-in-a-haystack task with periodic distractors.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Designs a three-dimensional rotary embedding that separates temporal spacing from spatial indexing.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "At 64k context, VideoRoPE scores 57.26 on LongVideoBench, 65.56 on MLVU and 61.33 on Video-MME, compared with M-RoPE’s 54.35, 61.10 and 59.67. The models are trained with an 8k context window.",
            "fragment": "S5.T2",
            "locator": "Table 2 · extrapolation beyond 8k training context · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "VideoRoPE achieves 91.11% on V-NIAH and 87.11% on its distractor variant V-NIAH-D, versus M-RoPE’s 78.67% and 74.67%. Accuracy is averaged across haystack lengths and needle-frame depths.",
            "fragment": "S5.T3",
            "locator": "Table 3 · distractor-controlled retrieval · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Video RoPE without suitable temporal allocation",
              "design": "Can be misled by periodic distractors in long-video retrieval."
            },
            {
              "method": "VideoRoPE",
              "design": "Combines low-frequency temporal allocation, diagonal spatial layout and adjustable temporal spacing."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2502.13128",
      "title": "SongGen: A Single Stage Auto-regressive Transformer for Text-to-Song Generation",
      "authors": [
        {
          "name": "Zihan Liu"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Zhixiong Zhang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICML",
        "citationText": "International Conference on Machine Learning (ICML), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-02-18",
        "arxivLastUpdated": "2025-05-30"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2502.13128",
        "googleScholar": "hW23VKIAAAAJ:blknAaTinKkC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2502.13128"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:blknAaTinKkC"
        },
        {
          "type": "code",
          "url": "https://github.com/LiuZH-19/SongGen",
          "repository": "LiuZH-19/SongGen"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1001,
        "order": 38
      },
      "shortName": "SongGen",
      "keywords": [
        "SongGen",
        "Text-to-song generation",
        "Autoregressive audio generation",
        "Controllable music generation",
        "Voice cloning",
        "Vocal synthesis",
        "Dual-track generation",
        "Audio tokenization"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250213128",
        "year": 2025,
        "booktitle": "Proceedings of the 42nd International Conference on Machine Learning",
        "source": {
          "label": "ICML 2025 proceedings record",
          "url": "https://proceedings.mlr.press/v267/liu25m.html"
        },
        "status": "published",
        "firstPage": 38351,
        "lastPage": 38364,
        "volume": "267",
        "publisher": "PMLR",
        "pdfURL": "https://raw.githubusercontent.com/mlresearch/v267/main/assets/liu25m/liu25m.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2502.13128v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2502.13128v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2502.13128v2"
          }
        },
        "abstract": {
          "text": "Text-to-song generation, the task of creating vocals and accompaniment from textual inputs, poses significant challenges due to domain complexity and data scarcity. Existing approaches often employ multi-stage generation procedures, leading to cumbersome training and inference pipelines, as well as suboptimal overall generation quality due to error accumulation across stages. In this paper, we propose SongGen, a fully open-source, single-stage auto-regressive transformer designed for controllable song generation. The proposed model facilitates fine-grained control over diverse musical attributes, including lyrics and textual descriptions of instrumentation, genre, mood, and timbre, while also offering an optional three-second reference clip for voice cloning. Within a unified auto-regressive framework, SongGen supports two output modes: mixed mode, which generates a mixture of vocals and accompaniment directly, and dual-track mode, which synthesizes them separately for greater flexibility in downstream applications. We explore diverse token pattern strategies for each mode, leading to notable improvements and valuable insights. Furthermore, we design an automated data preprocessing pipeline with effective quality control. To foster community engagement and future research, we will release our model weights, training code, annotated data, and preprocessing pipeline. The code is available at https://github.com/LiuZH-19/SongGen.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "SongGen generates controllable vocals and accompaniment in a single autoregressive stage, with either mixed audio output or separate vocal and accompaniment tracks.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Multi-stage song generation introduces pipeline complexity and can accumulate errors. SongGen unifies generation in one transformer while supporting lyrics, musical descriptions and an optional three-second reference for voice cloning.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Supports mixed and dual-track generation within a unified autoregressive framework.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Studies audio token patterns and develops an automated, quality-controlled preprocessing pipeline.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "SongGen’s Mixed-pro configuration obtains FAD 1.71 versus 2.18 for the multi-stage baseline, with CLAP 0.35 versus 0.29. Its phoneme error rate is 40.58 versus 38.80, so improved audio-distribution and text-alignment scores do not imply better lyric accuracy in every comparison.",
            "fragment": "S4.T1",
            "locator": "Table 1 · song-generation comparison · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Removing curriculum learning increases FAD from 1.71 to 2.35 and phoneme error rate from 40.58 to 55.71. Removing high-quality fine-tuning gives FAD 2.01 and phoneme error rate 43.68.",
            "fragment": "S4.T4",
            "locator": "Table 4 · training-scheme ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Multi-stage song generation",
              "design": "Separates generation stages, increasing pipeline complexity and the potential for error accumulation."
            },
            {
              "method": "SongGen",
              "design": "Generates mixed audio or separate vocal and accompaniment tracks in a single autoregressive stage."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2410.06241",
      "title": "ByTheWay: Boost Your Text-to-Video Generation Model to Higher Quality in a Training-free Way",
      "authors": [
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-10-08",
        "arxivLastUpdated": "2025-02-27"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2410.06241",
        "googleScholar": "hW23VKIAAAAJ:dfsIfKJdRG4C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2410.06241"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:dfsIfKJdRG4C"
        },
        {
          "type": "code",
          "url": "https://github.com/Bujiazi/ByTheWay",
          "repository": "Bujiazi/ByTheWay"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1020,
        "order": 39
      },
      "shortName": "ByTheWay",
      "keywords": [
        "ByTheWay",
        "Text-to-video generation",
        "Training-free guidance",
        "Temporal attention",
        "Temporal consistency",
        "Motion enhancement",
        "Fourier analysis",
        "Video diffusion"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241006241",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Bu_ByTheWay_Boost_Your_Text-to-Video_Generation_Model_to_Higher_Quality_in_CVPR_2025_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 12999,
        "lastPage": 13008,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2025/papers/Bu_ByTheWay_Boost_Your_Text-to-Video_Generation_Model_to_Higher_Quality_in_CVPR_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2410.06241v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2410.06241v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2410.06241v3"
          }
        },
        "abstract": {
          "text": "The text-to-video (T2V) generation models, offering convenient visual creation, have recently garnered increasing attention. Despite their substantial potential, the generated videos may present artifacts, including structural implausibility, temporal inconsistency, and a lack of motion, often resulting in near-static video. In this work, we have identified a correlation between the disparity of temporal attention maps across different blocks and the occurrence of temporal inconsistencies. Additionally, we have observed that the energy contained within the temporal attention maps is directly related to the magnitude of motion amplitude in the generated videos. Based on these observations, we present ByTheWay, a training-free method to improve the quality of text-to-video generation without introducing additional parameters, augmenting memory or sampling time. Specifically, ByTheWay is composed of two principal components: 1) Temporal Self-Guidance improves the structural plausibility and temporal consistency of generated videos by reducing the disparity between the temporal attention maps across various decoder blocks. 2) Fourier-based Motion Enhancement enhances the magnitude and richness of motion by amplifying the energy of the map. Extensive experiments demonstrate that ByTheWay significantly improves the quality of text-to-video generation with negligible additional cost.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "ByTheWay improves video consistency and motion without training by reducing disagreement between temporal attention maps and amplifying their motion-related frequency content.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Text-to-video models can produce structural artifacts, flicker and nearly static clips. ByTheWay connects cross-block attention-map disagreement with inconsistency and attention-map energy with motion amplitude, then intervenes directly during generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Introduces Temporal Self-Guidance to coordinate temporal attention across decoder blocks.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Uses Fourier-based Motion Enhancement to increase motion magnitude and richness.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Adding ByTheWay to AnimateDiff improves VBench motion smoothness from 0.9474 to 0.9786 and dynamic degree from 0.4073 to 0.5245. FreeInit reaches 0.9713 smoothness but 0.2941 dynamic degree, illustrating the importance of measuring both smoothness and motion extent.",
            "fragment": "S5.T2",
            "locator": "Table 2 · VBench, AnimateDiff backbone · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "In paired video-quality comparisons, users prefer ByTheWay outputs in 74.58% of AnimateDiff comparisons and 69.46% of VideoCrafter2 comparisons. These preference percentages are separate from the MLLM-based structure and motion assessments in the same table.",
            "fragment": "S5.T1",
            "locator": "Table 1 · human preference study · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Unmodified text-to-video sampling",
              "design": "Can exhibit disagreement between temporal attention maps and insufficient motion-related energy."
            },
            {
              "method": "ByTheWay",
              "design": "Coordinates temporal maps through self-guidance and increases motion through Fourier-based enhancement."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2501.05510",
      "title": "OVO-Bench: How Far is Your Video-LLMs from Real-World Online Video Understanding?",
      "authors": [
        {
          "name": "Yifei Li"
        },
        {
          "name": "Junbo Niu"
        },
        {
          "name": "Ziyang Miao"
        },
        {
          "name": "Chunjiang Ge"
        },
        {
          "name": "Yuanhang Zhou"
        },
        {
          "name": "Qihao He"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Rui Qian"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Conghui He"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-01-09",
        "arxivLastUpdated": "2025-03-27"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2501.05510",
        "googleScholar": "hW23VKIAAAAJ:iH-uZ7U-co4C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2501.05510"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:iH-uZ7U-co4C"
        },
        {
          "type": "code",
          "url": "https://github.com/JoeLeelyf/OVO-Bench",
          "repository": "JoeLeelyf/OVO-Bench"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1039,
        "order": 40
      },
      "shortName": "OVO-Bench",
      "keywords": [
        "OVO-Bench",
        "Online video understanding",
        "Temporal awareness",
        "Streaming video evaluation",
        "Backward tracing",
        "Real-time understanding",
        "Forward active responding",
        "Video language models"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250105510",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Niu_OVO-Bench_How_Far_is_Your_Video-LLMs_from_Real-World_Online_Video_CVPR_2025_paper.html"
        },
        "status": "published",
        "authors": [
          {
            "name": "Junbo Niu"
          },
          {
            "name": "Yifei Li"
          },
          {
            "name": "Ziyang Miao"
          },
          {
            "name": "Chunjiang Ge"
          },
          {
            "name": "Yuanhang Zhou"
          },
          {
            "name": "Qihao He"
          },
          {
            "name": "Xiaoyi Dong"
          },
          {
            "name": "Haodong Duan"
          },
          {
            "name": "Shuangrui Ding"
          },
          {
            "name": "Rui Qian"
          },
          {
            "name": "Pan Zhang"
          },
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Yuhang Cao"
          },
          {
            "name": "Conghui He"
          },
          {
            "name": "Jiaqi Wang"
          }
        ],
        "month": 6,
        "firstPage": 18902,
        "lastPage": 18913,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2025/papers/Niu_OVO-Bench_How_Far_is_Your_Video-LLMs_from_Real-World_Online_Video_CVPR_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2501.05510v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2501.05510v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2501.05510v2"
          }
        },
        "abstract": {
          "text": "Temporal Awareness, the ability to reason dynamically based on the timestamp when a question is raised, is the key distinction between offline and online video LLMs. Unlike offline models, which rely on complete videos for static, post hoc analysis, online models process video streams incrementally and dynamically adapt their responses based on the timestamp at which the question is posed. Despite its significance, temporal awareness has not been adequately evaluated in existing benchmarks. To fill this gap, we present OVO-Bench (Online-VideO-Benchmark), a novel video benchmark that emphasizes the importance of timestamps for advanced online video understanding capability benchmarking. OVO-Bench evaluates the ability of video LLMs to reason and respond to events occurring at specific timestamps under three distinct scenarios: (1) Backward tracing: trace back to past events to answer the question. (2) Real-time understanding: understand and respond to events as they unfold at the current timestamp. (3) Forward active responding: delay the response until sufficient future information becomes available to answer the question accurately. OVO-Bench comprises 12 tasks, featuring 644 unique videos and approximately human-curated 2,800 fine-grained meta-annotations with precise timestamps. We combine automated generation pipelines with human curation. With these high-quality samples, we further developed an evaluation pipeline to systematically query video LLMs along the video timeline. Evaluations of nine Video-LLMs reveal that, despite advancements on traditional benchmarks, current models struggle with online video understanding, showing a significant gap compared to human agents. We hope OVO-Bench will drive progress in video LLMs and inspire future research in online video reasoning. Our benchmark and code can be accessed at https://github.com/JoeLeelyf/OVO-Bench.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "OVO-Bench evaluates whether video models respond appropriately at a specified moment, distinguishing backward tracing, real-time understanding and waiting for future evidence.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Offline video evaluation does not test the temporal awareness needed for streaming interaction. OVO-Bench queries models along a timeline and assesses how answers should change with the information available at each timestamp.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Defines three temporal response scenarios spanning 12 online video tasks.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Provides 644 videos and approximately 2,800 human-curated, timestamped meta-annotations with an evaluation pipeline.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "OVO-Bench reports an overall average of 63.00 for Gemini-1.5-Pro with 1-fps input and 59.54 for GPT-4o with 64 frames, versus 92.81 for humans. The table places EPM and ASI queries at the video’s end to increase the gap from supporting clues.",
            "fragment": "S3.T1",
            "locator": "Table 1 · online video understanding · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Gemini-1.5-Pro scores 69.32 on real-time perception, 62.54 on backward tracing and 57.15 on forward active responding. This table uses accuracy-based forward-response metrics; its aggregate scores do not measure response timing alone.",
            "fragment": "S3.T1",
            "locator": "Table 1 · capability breakdown · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Offline video evaluation",
              "design": "Permits complete-video analysis after events have occurred."
            },
            {
              "method": "OVO-Bench",
              "design": "Queries along the timeline and distinguishes past-event retrieval, present understanding and waiting for future evidence."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2501.03218",
      "title": "Dispider: Enabling Video LLMs with Active Real-Time Interaction via Disentangled Perception, Decision, and Reaction",
      "authors": [
        {
          "name": "Rui Qian"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2025-01-06",
        "arxivLastUpdated": "2025-01-06"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2501.03218",
        "googleScholar": "hW23VKIAAAAJ:r0BpntZqJG4C"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2501.03218"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:r0BpntZqJG4C"
        },
        {
          "type": "code",
          "url": "https://github.com/Mark12Ding/Dispider",
          "repository": "Mark12Ding/Dispider"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1058,
        "order": 41
      },
      "shortName": "Dispider",
      "keywords": [
        "Dispider",
        "Streaming video assistants",
        "Active real-time interaction",
        "Asynchronous inference",
        "Proactive interaction",
        "Video perception",
        "Interaction timing",
        "Video question answering"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv250103218",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Qian_Dispider_Enabling_Video_LLMs_with_Active_Real-Time_Interaction_via_Disentangled_CVPR_2025_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 24045,
        "lastPage": 24055,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2025/papers/Qian_Dispider_Enabling_Video_LLMs_with_Active_Real-Time_Interaction_via_Disentangled_CVPR_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2501.03218v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2501.03218v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2501.03218v1"
          }
        },
        "abstract": {
          "text": "Active Real-time interaction with video LLMs introduces a new paradigm for human-computer interaction, where the model not only understands user intent but also responds while continuously processing streaming video on the fly. Unlike offline video LLMs, which analyze the entire video before answering questions, active real-time interaction requires three capabilities: 1) Perception: real-time video monitoring and interaction capturing. 2) Decision: raising proactive interaction in proper situations, 3) Reaction: continuous interaction with users. However, inherent conflicts exist among the desired capabilities. The Decision and Reaction require a contrary Perception scale and grain, and the autoregressive decoding blocks the real-time Perception and Decision during the Reaction. To unify the conflicted capabilities within a harmonious system, we present Dispider, a system that disentangles Perception, Decision, and Reaction. Dispider features a lightweight proactive streaming video processing module that tracks the video stream and identifies optimal moments for interaction. Once the interaction is triggered, an asynchronous interaction module provides detailed responses, while the processing module continues to monitor the video in the meantime. Our disentangled and asynchronous design ensures timely, contextually accurate, and computationally efficient responses, making Dispider ideal for active real-time interaction for long-duration video streams. Experiments show that Dispider not only maintains strong performance in conventional video QA tasks, but also significantly surpasses previous online models in streaming scenario responses, thereby validating the effectiveness of our architecture. The code and model are released at https://github.com/Mark12Ding/Dispider.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "Dispider separates perception, decision and response so a video assistant can keep monitoring a stream while asynchronously generating an interaction.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Autoregressive response generation can block perception and prevent timely decisions in streaming video systems. Dispider uses a lightweight monitoring module to decide when to interact and a separate asynchronous module to produce detailed responses.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Disentangles three components with conflicting temporal and computational requirements.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Enables proactive interaction while maintaining continuous video monitoring during response generation.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With the question placed before the video, Dispider at 1 fps reaches temporal-grounding F1 of 36.1 and episodic-memory F1 of 15.5, versus 13.2 and 3.8 for VideoLLM-Online at 2 fps.",
            "fragment": "S4.T3",
            "locator": "Table 3 · streaming ETBench protocol · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Replacing uniform 16-frame clips with scene-based segmentation raises MLVU accuracy from 59.8 to 61.7 and Video-MME from 55.4 to 57.2. Temporal-grounding F1 rises from 34.5 to 36.1.",
            "fragment": "S4.T4",
            "locator": "Table 4 · clip-segmentation ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Coupled perception and autoregressive response",
              "design": "Response generation can block continuous perception and delay further interaction decisions."
            },
            {
              "method": "Dispider",
              "design": "Maintains lightweight monitoring while an asynchronous module generates detailed responses."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2410.17247",
      "title": "PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction",
      "authors": [
        {
          "name": "Long Xing"
        },
        {
          "name": "Qidong Huang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Jiajie Lu"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Conghui He"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Feng Wu"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-10-22",
        "arxivLastUpdated": "2025-02-27"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2410.17247",
        "googleScholar": "hW23VKIAAAAJ:4OULZ7Gr8RgC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2410.17247"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:4OULZ7Gr8RgC"
        },
        {
          "type": "code",
          "url": "https://github.com/Cooperx521/PyramidDrop",
          "repository": "Cooperx521/PyramidDrop"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1077,
        "order": 42
      },
      "shortName": "PyramidDrop",
      "keywords": [
        "PyramidDrop",
        "Visual token reduction",
        "Efficient vision-language models",
        "Training efficiency",
        "Inference acceleration",
        "Depth-dependent redundancy",
        "Token pruning",
        "High-resolution understanding"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241017247",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper.html"
        },
        "status": "published",
        "title": "Conical Visual Concentration for Efficient Large Vision-Language Models",
        "month": 6,
        "firstPage": 14593,
        "lastPage": 14603,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2025/papers/Xing_Conical_Visual_Concentration_for_Efficient_Large_Vision-Language_Models_CVPR_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2410.17247v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2410.17247v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2410.17247v2"
          }
        },
        "abstract": {
          "text": "In large vision-language models (LVLMs), images serve as inputs that carry a wealth of information. As the idiom \"A picture is worth a thousand words\" implies, representing a single image in current LVLMs can require hundreds or even thousands of tokens. This results in significant computational costs, which grow quadratically as input image resolution increases, thereby severely impacting the efficiency of both training and inference. Previous approaches have attempted to reduce the number of image tokens either before or within the early layers of LVLMs. However, these strategies inevitably result in the loss of crucial image information, ultimately diminishing model performance. To address this challenge, we conduct an empirical study revealing that all visual tokens are necessary for LVLMs in the shallow layers, and token redundancy progressively increases in the deeper layers of the model. To this end, we propose PyramidDrop, a visual redundancy reduction strategy for LVLMs to boost their efficiency in both training and inference with neglectable performance loss. Specifically, we partition the LVLM into several stages and drop part of the image tokens at the end of each stage with a pre-defined ratio, creating pyramid-like visual tokens across model layers. The dropping is based on a lightweight similarity calculation with a negligible time overhead. Extensive experiments demonstrate that PyramidDrop can achieve a 40% training time and 55% inference FLOPs acceleration of LLaVA-NeXT with comparable performance. Besides, the PyramidDrop could also serve as a plug-and-play strategy for inference acceleration without training, with better performance and lower inference cost than counterparts. Code is available at https://github.com/Cooperx521/PyramidDrop.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "PyramidDrop retains visual detail in shallow LVLM layers and progressively removes redundant tokens in deeper layers to reduce training and inference cost.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Early visual-token reduction can discard information before the model has processed it. PyramidDrop observes increasing redundancy with depth and drops a proportion of image tokens at stage boundaries using lightweight similarity scores.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Establishes a depth-dependent account of visual-token redundancy in large vision-language models.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Provides a staged token-reduction strategy for training and a plug-and-play inference setting.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Using PyramidDrop during training reduces GPU-hours from 366 to 218 (40.4%) and inference FLOPs from 20.8T to 9.46T. The eight-benchmark average changes from 67.6 to 67.5 under this five-patch setting.",
            "fragment": "S4.T3",
            "locator": "Table 3 · LLaVA-NeXT-7B, five image patches · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "At an average of 128 retained visual tokens on LLaVA-1.5-7B, PyramidDrop obtains a benchmark average of 66.4, retaining 95.6% of the uncompressed model’s 69.4. SparseVLM and FastV score 64.6 and 58.2 at the same token budget.",
            "fragment": "S3.T2",
            "locator": "Table 2 · inference-only token reduction · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Visual-token reduction before or in early layers",
              "design": "Can remove important image information before sufficient processing."
            },
            {
              "method": "PyramidDrop",
              "design": "Progressively drops tokens at later stage boundaries as visual redundancy increases."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2407.02165",
      "title": "WildAvatar: Learning In-the-wild 3D Avatars from the Web",
      "authors": [
        {
          "name": "Zihao Huang"
        },
        {
          "name": "Shoukang Hu"
        },
        {
          "name": "Guangcong Wang"
        },
        {
          "name": "Tianqi Liu"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zhiguo Cao"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Ziwei Liu"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-07-02",
        "arxivLastUpdated": "2025-03-12"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2407.02165",
        "googleScholar": "hW23VKIAAAAJ:fPk4N6BV_jEC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2407.02165"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:fPk4N6BV_jEC"
        },
        {
          "type": "code",
          "url": "https://github.com/wildavatar/WildAvatar_Toolbox",
          "repository": "wildavatar/WildAvatar_Toolbox"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1096,
        "order": 43
      },
      "shortName": "WildAvatar",
      "keywords": [
        "WildAvatar",
        "3D human avatars",
        "In-the-wild reconstruction",
        "Web video datasets",
        "Human motion annotation",
        "Avatar generalization",
        "Automatic data curation",
        "Human reconstruction"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240702165",
        "year": 2025,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2025 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2025/html/Huang_WildAvatar_Learning_In-the-wild_3D_Avatars_from_the_Web_CVPR_2025_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 15963,
        "lastPage": 15975,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2025/papers/Huang_WildAvatar_Learning_In-the-wild_3D_Avatars_from_the_Web_CVPR_2025_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v4",
            "url": "https://arxiv.org/abs/2407.02165v4"
          },
          "paper": {
            "label": "Paper · arXiv v4",
            "url": "https://arxiv.org/pdf/2407.02165v4",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v4",
            "url": "https://arxiv.org/html/2407.02165v4"
          }
        },
        "abstract": {
          "text": "Existing research on avatar creation is typically limited to laboratory datasets, which require high costs against scalability and exhibit insufficient representation of the real world. On the other hand, the web abounds with off-the-shelf real-world human videos, but these videos vary in quality and require accurate annotations for avatar creation. To this end, we propose an automatic annotating pipeline with filtering protocols to curate these humans from the web. Our pipeline surpasses state-of-the-art methods on the EMDB benchmark, and the filtering protocols boost verification metrics on web videos. We then curate WildAvatar, a web-scale in-the-wild human avatar creation dataset extracted from YouTube, with 10000+ different human subjects and scenes. WildAvatar is at least 10× richer than previous datasets for 3D human avatar creation and closer to the real world. To explore its potential, we demonstrate the quality and generalizability of avatar creation methods on WildAvatar. We will publicly release our code, data source links and annotations to push forward 3D human avatar creation and other related fields for real-world applications.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "takeaway": {
          "text": "WildAvatar scales 3D human avatar data beyond laboratory capture by automatically annotating and filtering web videos with more than 10,000 subjects and scenes.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "summary": {
          "text": "Laboratory avatar datasets are expensive and poorly represent real-world variation. WildAvatar combines automatic annotation with filtering protocols to turn diverse YouTube footage into data for in-the-wild avatar creation.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "contributions": [
          {
            "text": "Develops an annotation and quality-filtering pipeline for unconstrained human videos.",
            "source": "abstract",
            "locator": "arXiv abstract · v4"
          },
          {
            "text": "Curates a large in-the-wild dataset and evaluates avatar reconstruction quality and generalization.",
            "source": "abstract",
            "locator": "arXiv abstract · v4"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "The pipeline filters 465,801 candidate clips to 10,647 qualified clips. Across filtering and final refinement, PCK at threshold 0.1 rises from 0.282 to 0.921, while the out-of-mask SMPL overlap measure falls from 0.760 to 0.028.",
            "fragment": "S3.T3",
            "locator": "Table 3 · web-video annotation pipeline · arXiv v4",
            "source": "fullText"
          },
          {
            "text": "For the Gaussian-Human method in the novel-pose evaluation, replacing HMR2.0 annotations with WildAvatar annotations raises PSNR from 24.73 to 25.89 dB. Every one of the seven avatar methods listed in the table gains PSNR with the proposed annotations.",
            "fragment": "S5.T4",
            "locator": "Table 4 · downstream novel-pose synthesis · arXiv v4",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v4",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Laboratory avatar datasets",
              "design": "Requires expensive controlled capture and offers limited representation of unconstrained scenes."
            },
            {
              "method": "WildAvatar",
              "design": "Automatically annotates and filters diverse web videos to scale in-the-wild avatar training data."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2410.17637",
      "title": "MIA-DPO: Multi-Image Augmented Direct Preference Optimization For Large Vision-Language Models",
      "authors": [
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Conghui He"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-10-23",
        "arxivLastUpdated": "2024-10-23"
      },
      "topics": [
        "Reinforcement Learning from Human Feedback"
      ],
      "identifiers": {
        "arxiv": "2410.17637",
        "googleScholar": "hW23VKIAAAAJ:9ZlFYXVOiuMC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2410.17637"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:9ZlFYXVOiuMC"
        },
        {
          "type": "code",
          "url": "https://github.com/Liuziyu77/MIA-DPO",
          "repository": "Liuziyu77/MIA-DPO"
        },
        {
          "type": "project",
          "url": "https://liuziyu77.github.io/MIA-DPO"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1115,
        "order": 44
      },
      "shortName": "MIA-DPO",
      "keywords": [
        "MIA-DPO",
        "Multi-image alignment",
        "Direct preference optimization",
        "Attention-aware selection",
        "Preference data construction",
        "Visual instruction tuning",
        "Distractor images",
        "Vision-language models"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv241017637",
        "year": 2025,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2025 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/557a20663907ed637c2807f608d5bec2-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 34583,
        "lastPage": 34610,
        "volume": "2025",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2025/file/557a20663907ed637c2807f608d5bec2-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2410.17637v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2410.17637v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2410.17637v1"
          }
        },
        "abstract": {
          "text": "Visual preference alignment involves training Large Vision-Language Models (LVLMs) to predict human preferences between visual inputs. This is typically achieved by using labeled datasets of chosen/rejected pairs and employing optimization algorithms like direct preference optimization (DPO). Existing visual alignment methods, primarily designed for single-image scenarios, struggle to effectively handle the complexity of multi-image tasks due to the scarcity of diverse training data and the high cost of annotating chosen/rejected pairs. We present Multi-Image Augmented Direct Preference Optimization (MIA-DPO), a visual preference alignment approach that effectively handles multi-image inputs. MIA-DPO mitigates the scarcity of diverse multi-image training data by extending single-image data with unrelated images arranged in grid collages or pic-in-pic formats, significantly reducing the costs associated with multi-image data annotations. Our observation reveals that attention values of LVLMs vary considerably across different images. We use attention values to identify and filter out rejected responses the model may have mistakenly focused on. Our attention-aware selection for constructing the chosen/rejected pairs without relying on (i) human annotation, (ii) extra data, and (iii) external models or APIs. MIA-DPO is compatible with various architectures and outperforms existing methods on five multi-image benchmarks, achieving an average performance boost of 3.0% on LLaVA-v1.5 and 4.3% on the recent InternLM-XC2.5. Moreover, MIA-DPO has a minimal effect on the model's ability to understand single images.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "MIA-DPO creates multi-image preference-training pairs from single-image data and attention signals, reducing the need for new human annotations or external judge models.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Multi-image preference alignment is constrained by scarce preference pairs. MIA-DPO inserts unrelated images into collages or picture-in-picture inputs and uses attention to identify responses influenced by the wrong visual evidence.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Augments single-image training examples into multi-image preference tasks.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Constructs chosen/rejected pairs through attention-aware response selection without external model calls.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "MIA-DPO raises the five-benchmark average from 40.4 to 43.4 for LLaVA-v1.5-7B and from 53.6 to 57.9 for InternLM-XComposer2.5-7B. On the latter backbone, Mantis rises from 49.3 to 60.4.",
            "fragment": "S4.T1",
            "locator": "Table 1 · multi-image evaluation · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "For LLaVA-v1.5, post-selection raises the multi-image average from 42.3 to 43.4, versus a 40.4 baseline. BLINK improves from 38.7 without post-selection to 42.9 with it.",
            "fragment": "S4.T3.fig1",
            "locator": "Table 3 · attention-based post-selection · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Single-image preference alignment",
              "design": "Does not directly address the complexity and training-data scarcity of multi-image tasks."
            },
            {
              "method": "MIA-DPO",
              "design": "Adds distractor images and uses attention-aware selection to build multi-image preference pairs from existing data."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2406.05338",
      "title": "MotionClone: Training-Free Motion Cloning for Controllable Video Generation",
      "authors": [
        {
          "name": "Pengyang Ling"
        },
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Huaian Chen"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Yi Jin"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2024-06-08",
        "arxivLastUpdated": "2024-10-22"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2406.05338",
        "googleScholar": "hW23VKIAAAAJ:kNdYIx-mwKoC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2406.05338"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:kNdYIx-mwKoC"
        },
        {
          "type": "code",
          "url": "https://github.com/Bujiazi/MotionClone",
          "repository": "Bujiazi/MotionClone"
        },
        {
          "type": "project",
          "url": "https://bujiazi.github.io/motionclone.github.io/"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1135,
        "order": 45
      },
      "shortName": "MotionClone",
      "keywords": [
        "MotionClone",
        "Motion-controlled video generation",
        "Training-free motion transfer",
        "Temporal attention",
        "Camera motion",
        "Object motion",
        "Text-to-video",
        "Image-to-video"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240605338",
        "year": 2025,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2025 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2025/hash/bc82dbfbfa43232be85b8d9838f49c3e-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 75579,
        "lastPage": 75601,
        "volume": "2025",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2025/file/bc82dbfbfa43232be85b8d9838f49c3e-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v6",
            "url": "https://arxiv.org/abs/2406.05338v6"
          },
          "paper": {
            "label": "Paper · arXiv v6",
            "url": "https://arxiv.org/pdf/2406.05338v6",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v6",
            "url": "https://arxiv.org/html/2406.05338v6"
          }
        },
        "abstract": {
          "text": "Motion-based controllable video generation offers the potential for creating captivating visual content. Existing methods typically necessitate model training to encode particular motion cues or incorporate fine-tuning to inject certain motion patterns, resulting in limited flexibility and generalization. In this work, we propose MotionClone, a training-free framework that enables motion cloning from reference videos to versatile motion-controlled video generation, including text-to-video and image-to-video. Based on the observation that the dominant components in temporal-attention maps drive motion synthesis, while the rest mainly capture noisy or very subtle motions, MotionClone utilizes sparse temporal attention weights as motion representations for motion guidance, facilitating diverse motion transfer across varying scenarios. Meanwhile, MotionClone allows for the direct extraction of motion representation through a single denoising step, bypassing the cumbersome inversion processes and thus promoting both efficiency and flexibility. Extensive experiments demonstrate that MotionClone exhibits proficiency in both global camera motion and local object motion, with notable superiority in terms of motion fidelity, textual alignment, and temporal consistency.",
          "source": "abstract",
          "locator": "arXiv abstract · v6"
        },
        "takeaway": {
          "text": "MotionClone transfers motion from a reference video without training by using sparse temporal attention as guidance for text-to-video and image-to-video generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v6"
        },
        "summary": {
          "text": "Motion-controlled generation often requires specialized training or fine-tuning. MotionClone identifies dominant temporal-attention components as useful motion representations and extracts them in a single denoising step without a full inversion process.",
          "source": "abstract",
          "locator": "arXiv abstract · v6"
        },
        "contributions": [
          {
            "text": "Represents reference motion with sparse temporal-attention weights.",
            "source": "abstract",
            "locator": "arXiv abstract · v6"
          },
          {
            "text": "Supports motion transfer across scenarios with efficient single-step motion extraction.",
            "source": "abstract",
            "locator": "arXiv abstract · v6"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "MotionClone obtains textual alignment 0.3187 and temporal consistency 0.9621, compared with VMC’s 0.3134 and 0.9614. These automatic metrics are distinct from the human-rating rows in the table.",
            "fragment": "S4.T1",
            "locator": "Table 1 · automatic motion-cloning evaluation · arXiv v6",
            "source": "fullText"
          },
          {
            "text": "MotionClone’s user-study scores are 3.69 for motion preservation, 4.31 for appearance diversity and 4.28 for temporal consistency, compared with VMC’s 2.59, 3.51 and 2.85. These are rating scores, not percentages.",
            "fragment": "S4.T1",
            "locator": "Table 1 · human evaluation · arXiv v6",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v6",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Training- or fine-tuning-based motion control",
              "design": "Learns or injects specific motion patterns through model updates."
            },
            {
              "method": "MotionClone",
              "design": "Extracts sparse temporal attention from a reference video and guides generation without model training."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2407.11691",
      "title": "VLMEvalKit: An Open-Source Toolkit for Evaluating Large Multi-Modality Models",
      "authors": [
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Xinyu Fang"
        },
        {
          "name": "Junming Yang"
        },
        {
          "name": "Xiangyu Zhao"
        },
        {
          "name": "Zerun Ma"
        },
        {
          "name": "Yuxuan Qiao"
        },
        {
          "name": "Mo Li"
        },
        {
          "name": "Tianhao Liang"
        },
        {
          "name": "Lin Zhu"
        },
        {
          "name": "Amit Agarwal"
        },
        {
          "name": "Xiaozhe Li"
        },
        {
          "name": "Shengyuan Ding"
        },
        {
          "name": "Jiazi Bu"
        },
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Zhangyang Qi"
        },
        {
          "name": "Yifei Li"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zhe Chen"
        },
        {
          "name": "Lin Chen"
        },
        {
          "name": "Yuan Liu"
        },
        {
          "name": "Yubo Ma"
        },
        {
          "name": "Hailong Sun"
        },
        {
          "name": "Yifan Zhang"
        },
        {
          "name": "Shiyin Lu"
        },
        {
          "name": "Tack Hwa Wong"
        },
        {
          "name": "Weiyun Wang"
        },
        {
          "name": "Peiheng Zhou"
        },
        {
          "name": "Chaoyou Fu"
        },
        {
          "name": "Junbo Cui"
        },
        {
          "name": "Jixuan Chen"
        },
        {
          "name": "Enxin Song"
        },
        {
          "name": "Song Mao"
        },
        {
          "name": "Junming Lin"
        },
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Zhaowei Wang"
        },
        {
          "name": "Zicheng Zhang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Junjun He"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Kai Chen"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "ACM MM",
        "citationText": "ACM Multimedia (ACM MM), 2024 (Open Source Software Competition)"
      },
      "dates": {
        "arxivFirstPosted": "2024-07-16",
        "arxivLastUpdated": "2026-07-06"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2407.11691",
        "googleScholar": "hW23VKIAAAAJ:4TOpqqG69KYC",
        "doi": "10.1145/3664647.3685520"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2407.11691"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:4TOpqqG69KYC"
        },
        {
          "type": "code",
          "url": "https://github.com/open-compass/VLMEvalKit",
          "repository": "open-compass/VLMEvalKit"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1155,
        "order": 46
      },
      "shortName": "VLMEvalKit",
      "keywords": [
        "VLMEvalKit",
        "Multimodal evaluation",
        "Reproducible benchmarks",
        "OpenVLM Leaderboard",
        "Vision-language models",
        "Distributed inference",
        "Evaluation toolkit",
        "Cross-model comparison"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240711691",
        "year": 2024,
        "booktitle": "Proceedings of the 32nd ACM International Conference on Multimedia",
        "publisher": "ACM",
        "source": {
          "label": "ACM MM publisher record",
          "url": "https://doi.org/10.1145/3664647.3685520"
        },
        "status": "published",
        "month": 10,
        "title": "VLMEvalKit: An Open-Source ToolKit for Evaluating Large Multi-Modality Models",
        "authors": [
          {
            "name": "Haodong Duan"
          },
          {
            "name": "Junming Yang"
          },
          {
            "name": "Yuxuan Qiao"
          },
          {
            "name": "Xinyu Fang"
          },
          {
            "name": "Lin Chen"
          },
          {
            "name": "Yuan Liu"
          },
          {
            "name": "Xiaoyi Dong"
          },
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Pan Zhang"
          },
          {
            "name": "Jiaqi Wang"
          },
          {
            "name": "Dahua Lin"
          },
          {
            "name": "Kai Chen"
          }
        ],
        "firstPage": 11198,
        "lastPage": 11201,
        "pdfURL": "https://dl.acm.org/doi/pdf/10.1145/3664647.3685520",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v5",
            "url": "https://arxiv.org/abs/2407.11691v5"
          },
          "paper": {
            "label": "Paper · arXiv v5",
            "url": "https://arxiv.org/pdf/2407.11691v5",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v5",
            "url": "https://arxiv.org/html/2407.11691v5"
          }
        },
        "abstract": {
          "text": "We present VLMEvalKit: an open-source toolkit for evaluating large multi-modality models based on PyTorch. The toolkit aims to provide a user-friendly and comprehensive framework for researchers and developers to evaluate existing multi-modality models and publish reproducible evaluation results. In VLMEvalKit, we implement over 450+ large multi-modality model configurations, including both proprietary APIs and open-source models, and support 330+ benchmarks across diverse multi-modal benchmarks. By implementing a single interface, new models can be easily added to the toolkit, while the toolkit automatically handles the remaining workloads, including data preparation, distributed inference, prediction post-processing, and metric calculation. VLMEvalKit has also evolved to a broader evaluation suite spanning video/audio, document understanding, GUI grounding, spatial reasoning, safety, scientific reasoning, and multi-turn dialogue. Based on the evaluation results obtained with the toolkit, we host the OpenVLM Leaderboard, a comprehensive leaderboard to track the progress of multi-modality learning research. The toolkit is released on https://github.com/open-compass/VLMEvalKit and is actively maintained.",
          "source": "abstract",
          "locator": "arXiv abstract · v5"
        },
        "takeaway": {
          "text": "VLMEvalKit standardizes multimodal evaluation behind a common interface that handles data preparation, distributed inference, post-processing and metrics for reproducible comparisons.",
          "source": "abstract",
          "locator": "arXiv abstract · v5"
        },
        "summary": {
          "text": "Comparing multimodal models requires coordinating heterogeneous APIs, datasets and evaluation rules. VLMEvalKit supplies a shared PyTorch-based framework for proprietary and open models and publishes results through the OpenVLM Leaderboard.",
          "source": "abstract",
          "locator": "arXiv abstract · v5"
        },
        "contributions": [
          {
            "text": "Lets a new model join the evaluation pipeline through a single interface.",
            "source": "abstract",
            "locator": "arXiv abstract · v5"
          },
          {
            "text": "Extends evaluation across image, video, audio, documents, GUI grounding, spatial reasoning, safety and multi-turn dialogue.",
            "source": "abstract",
            "locator": "arXiv abstract · v5"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "The expanded report documents support for more than 335 multimodal benchmarks, including video, document parsing, spatial grounding, safety and scientific reasoning. This is the scope reported in arXiv v5, which postdates the four-page ACM Multimedia 2024 publication.",
            "fragment": "S1.T1",
            "locator": "Table 1 · expanded arXiv v5 toolkit · arXiv v5",
            "source": "fullText"
          },
          {
            "text": "The general-VQA comparison normalizes individual benchmark scores to 0–100 before averaging. In its 2025-09-17 snapshot, Gemini-2.5-Pro scores 80.1 on average and GPT-5-20250807 79.9; these date-stamped results illustrate the toolkit’s evaluation protocol rather than a live ranking.",
            "fragment": "S2.T2",
            "locator": "Table 2 · reproducible leaderboard snapshot · arXiv v5",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v5",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Model integration in VLMEvalKit",
              "design": "Implements a common interface for the model or API being evaluated."
            },
            {
              "method": "Shared evaluation pipeline",
              "design": "Handles data preparation, distributed inference, post-processing and metrics after model integration."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2407.01523",
      "title": "MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations",
      "authors": [
        {
          "name": "Yubo Ma"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Liangyu Chen"
        },
        {
          "name": "Meiqi Chen"
        },
        {
          "name": "Yizhu Jiao"
        },
        {
          "name": "Xinze Li"
        },
        {
          "name": "Xinyuan Lu"
        },
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Yan Ma"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Liangming Pan"
        },
        {
          "name": "Yu-Gang Jiang"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Yixin Cao",
          "corresponding": true
        },
        {
          "name": "Aixin Sun"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024 (Datasets and Benchmarks Track)"
      },
      "dates": {
        "arxivFirstPosted": "2024-07-01",
        "arxivLastUpdated": "2024-11-12"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2407.01523",
        "googleScholar": "hW23VKIAAAAJ:ULOm3_A8WrAC",
        "doi": "10.52202/079017-3041"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2407.01523"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:ULOm3_A8WrAC"
        },
        {
          "type": "code",
          "url": "https://github.com/mayubo2333/MMLongBench-Doc",
          "repository": "mayubo2333/MMLongBench-Doc"
        },
        {
          "type": "project",
          "url": "https://mayubo2333.github.io/MMLongBench-Doc/"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/datasets/yubo2333/MMLongBench-Doc",
          "label": "Dataset",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [
          "Spotlight"
        ],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1175,
        "order": 47
      },
      "shortName": "MMLongBench-Doc",
      "keywords": [
        "MMLongBench-Doc",
        "Long-document understanding",
        "Multimodal documents",
        "Cross-page reasoning",
        "Document question answering",
        "Evidence integration",
        "Unanswerable questions",
        "Hallucination evaluation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240701523",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/ae0e43289bffea0c1fa34633fc608e92-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        "status": "published",
        "title": "MMLONGBENCH-DOC: Benchmarking Long-context Document Understanding with Visualizations",
        "firstPage": 95963,
        "lastPage": 96010,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/ae0e43289bffea0c1fa34633fc608e92-Paper-Datasets_and_Benchmarks_Track.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2407.01523v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2407.01523v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2407.01523v3"
          }
        },
        "abstract": {
          "text": "Understanding documents with rich layouts and multi-modal components is a long-standing and practical task. Recent Large Vision-Language Models (LVLMs) have made remarkable strides in various tasks, particularly in single-page document understanding (DU). However, their abilities on long-context DU remain an open problem. This work presents MMLongBench-Doc, a long-context, multi-modal benchmark comprising 1,062 expert-annotated questions. Distinct from previous datasets, it is constructed upon 130 lengthy PDF-formatted documents with an average of 49.4 pages and 20,971 textual tokens. Towards comprehensive evaluation, answers to these questions rely on pieces of evidence from (1) different sources (text, image, chart, table, and layout structure) and (2) various locations (i.e. page number). Moreover, 33.2% of the questions are cross-page questions requiring evidence across multiple pages. 22.8% of the questions are designed to be unanswerable for detecting potential hallucinations. Experiments on 14 LVLMs demonstrate that long-context DU greatly challenges current models. Notably, the best-performing model, GPT-4o, achieves an F1 score of only 42.7%, while the second-best, GPT-4V, scores 31.4%. Furthermore, 12 LVLMs (all except GPT-4o and GPT-4V) even present worse performance than their LLM counterparts which are fed with lossy-parsed OCR documents. These results validate the necessity of future research toward more capable long-context LVLMs. Project Page: https://mayubo2333.github.io/MMLongBench-Doc",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "MMLongBench-Doc exposes a substantial gap in long-document visual understanding through questions requiring diverse evidence, cross-page reasoning and recognition of unanswerable queries.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Strong single-page document results do not establish long-document competence. MMLongBench-Doc evaluates evidence integration over lengthy PDFs containing text, images, charts, tables and layout structure.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Provides 1,062 expert-annotated questions over 130 documents averaging 49.4 pages and 20,971 text tokens.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Includes 33.2% cross-page questions and 22.8% unanswerable questions to test evidence integration and hallucination.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "MMLongBench-Doc averages 47.5 pages and 21,214.1 tokens per document. Cross-page questions account for 33.0%, and unanswerable questions for 22.5%; answer evidence is located at page 23.6 on average.",
            "fragment": "S1.T1",
            "locator": "Table 1 · long-document benchmark design · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "With Tesseract OCR, GPT-4o reaches 30.1 generalized accuracy and 30.5 F1. Its single-page accuracy is 35.4 versus 29.3 for cross-page questions, quantifying the gap under this text-extraction pipeline.",
            "fragment": "S3.T3",
            "locator": "Table 3 · OCR-plus-LLM evaluation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Single-page document evaluation",
              "design": "Tests local document understanding without requiring evidence integration across a lengthy PDF."
            },
            {
              "method": "MMLongBench-Doc",
              "design": "Combines multiple evidence types, cross-page questions and unanswerable queries over long documents."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2406.11833",
      "title": "MMDU: A Multi-Turn Multi-Image Dialog Understanding Benchmark and Instruction-Tuning Dataset for LVLMs",
      "authors": [
        {
          "name": "Ziyu Liu"
        },
        {
          "name": "Tao Chu"
        },
        {
          "name": "Yuhang Zang",
          "corresponding": true
        },
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Zijian Liang"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Yu Qiao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang",
          "corresponding": true
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024 (Datasets and Benchmarks Track)"
      },
      "dates": {
        "arxivFirstPosted": "2024-06-17",
        "arxivLastUpdated": "2024-10-29"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2406.11833",
        "googleScholar": "hW23VKIAAAAJ:Zph67rFs4hoC",
        "doi": "10.52202/079017-0278"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2406.11833"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:Zph67rFs4hoC"
        },
        {
          "type": "code",
          "url": "https://github.com/Liuziyu77/MMDU",
          "repository": "Liuziyu77/MMDU"
        },
        {
          "type": "project",
          "url": "https://liuziyu77.github.io/MMDU/"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/datasets/laolao77/MMDU",
          "label": "Dataset",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1196,
        "order": 48
      },
      "shortName": "MMDU",
      "keywords": [
        "MMDU",
        "MMDU-45k",
        "Multi-turn dialogue",
        "Multi-image understanding",
        "Conversational instruction tuning",
        "Long-context interaction",
        "Multimodal benchmarks",
        "Vision-language assistants"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240611833",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/1057053100de064a44286239724f7865-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        "status": "published",
        "firstPage": 8698,
        "lastPage": 8733,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/1057053100de064a44286239724f7865-Paper-Datasets_and_Benchmarks_Track.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2406.11833v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2406.11833v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2406.11833v2"
          }
        },
        "abstract": {
          "text": "Generating natural and meaningful responses to communicate with multi-modal human inputs is a fundamental capability of Large Vision-Language Models(LVLMs). While current open-source LVLMs demonstrate promising performance in simplified scenarios such as single-turn single-image input, they fall short in real-world conversation scenarios such as following instructions in a long context history with multi-turn and multi-images. Existing LVLM benchmarks primarily focus on single-choice questions or short-form responses, which do not adequately assess the capabilities of LVLMs in real-world human-AI interaction applications. Therefore, we introduce MMDU, a comprehensive benchmark, and MMDU-45k, a large-scale instruction tuning dataset, designed to evaluate and improve LVLMs' abilities in multi-turn and multi-image conversations. We employ the clustering algorithm to find the relevant images and textual descriptions from the open-source Wikipedia and construct the question-answer pairs by human annotators with the assistance of the GPT-4o model. MMDU has a maximum of 18k image+text tokens, 20 images, and 27 turns, which is at least 5x longer than previous benchmarks and poses challenges to current LVLMs. Our in-depth analysis of 15 representative LVLMs using MMDU reveals that open-source LVLMs lag behind closed-source counterparts due to limited conversational instruction tuning data. We demonstrate that fine-tuning open-source LVLMs on MMDU-45k significantly address this gap, generating longer and more accurate conversations, and improving scores on MMDU and existing benchmarks (MMStar: +1.1%, MathVista: +1.5%, ChartQA:+1.2%). Our contributions pave the way for bridging the gap between current LVLM models and real-world application demands. This project is available at https://github.com/Liuziyu77/MMDU.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "MMDU evaluates sustained multi-image dialogue, while MMDU-45k provides instruction data that helps open LVLMs close the conversational gap identified by the benchmark.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Single-turn or short-answer benchmarks underrepresent real multimodal conversations. MMDU uses related Wikipedia images and descriptions to construct longer dialogues with human annotation and GPT-4o assistance.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Introduces a benchmark reaching 18K image-plus-text tokens, 20 images and 27 turns.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Builds MMDU-45k for instruction tuning and evaluates 15 representative LVLMs.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Adding MMDU-45k raises InternLM-XComposer2’s MMDU average from 35.6 to 50.1, including image-relationship understanding from 35.2 to 48.7. LLaVA-1.5-7B improves from 32.2 to 37.2 overall.",
            "fragment": "S4.T2",
            "locator": "Table 2 · dialogue fine-tuning · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "For LLaVA-1.5, training with MMDU-45k raises BLINK from 37.1 to 40.1 and Mantis in the image-sequence setting from 37.8 to 44.7. This tests transfer beyond the MMDU benchmark itself.",
            "fragment": "S4.T4.fig1",
            "locator": "Table 4 · transfer to multi-image benchmarks · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Single-turn or short-response evaluation",
              "design": "Underrepresents sustained conversations with multiple images and a long interaction history."
            },
            {
              "method": "MMDU",
              "design": "Evaluates long multi-turn, multi-image dialogue and supplies MMDU-45k to improve these skills."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2406.04325",
      "title": "ShareGPT4Video: Improving Video Understanding and Generation with Better Captions",
      "authors": [
        {
          "name": "Lin Chen"
        },
        {
          "name": "Xilin Wei"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zehui Chen"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Bin Lin"
        },
        {
          "name": "Zhenyu Tang"
        },
        {
          "name": "Li Yuan"
        },
        {
          "name": "Yu Qiao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Feng Zhao"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024 (Datasets and Benchmarks Track)"
      },
      "dates": {
        "arxivFirstPosted": "2024-06-06",
        "arxivLastUpdated": "2024-06-06"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2406.04325",
        "googleScholar": "hW23VKIAAAAJ:3fE2CSJIrl8C",
        "doi": "10.52202/079017-0614"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2406.04325"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:3fE2CSJIrl8C"
        },
        {
          "type": "code",
          "url": "https://github.com/ShareGPT4Omni/ShareGPT4Video",
          "repository": "ShareGPT4Omni/ShareGPT4Video"
        },
        {
          "type": "project",
          "url": "https://sharegpt4video.github.io/"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/datasets/ShareGPT4Video/ShareGPT4Video",
          "label": "Dataset",
          "variant": "dataset"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1217,
        "order": 49
      },
      "shortName": "ShareGPT4Video",
      "keywords": [
        "ShareGPT4Video",
        "Dense video captioning",
        "Differential captioning",
        "Temporal descriptions",
        "Video instruction data",
        "ShareCaptioner-Video",
        "Video understanding",
        "Video generation supervision"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240604325",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/22a7476e4fd36818777c47e666f61a41-Abstract-Datasets_and_Benchmarks_Track.html"
        },
        "status": "published",
        "firstPage": 19472,
        "lastPage": 19495,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/22a7476e4fd36818777c47e666f61a41-Paper-Datasets_and_Benchmarks_Track.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2406.04325v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2406.04325v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2406.04325v1"
          }
        },
        "abstract": {
          "text": "We present the ShareGPT4Video series, aiming to facilitate the video understanding of large video-language models (LVLMs) and the video generation of text-to-video models (T2VMs) via dense and precise captions. The series comprises: 1) ShareGPT4Video, 40K GPT4V annotated dense captions of videos with various lengths and sources, developed through carefully designed data filtering and annotating strategy. 2) ShareCaptioner-Video, an efficient and capable captioning model for arbitrary videos, with 4.8M high-quality aesthetic videos annotated by it. 3) ShareGPT4Video-8B, a simple yet superb LVLM that reached SOTA performance on three advancing video benchmarks. To achieve this, taking aside the non-scalable costly human annotators, we find using GPT4V to caption video with a naive multi-frame or frame-concatenation input strategy leads to less detailed and sometimes temporal-confused results. We argue the challenge of designing a high-quality video captioning strategy lies in three aspects: 1) Inter-frame precise temporal change understanding. 2) Intra-frame detailed content description. 3) Frame-number scalability for arbitrary-length videos. To this end, we meticulously designed a differential video captioning strategy, which is stable, scalable, and efficient for generating captions for videos with arbitrary resolution, aspect ratios, and length. Based on it, we construct ShareGPT4Video, which contains 40K high-quality videos spanning a wide range of categories, and the resulting captions encompass rich world knowledge, object attributes, camera movements, and crucially, detailed and precise temporal descriptions of events. Based on ShareGPT4Video, we further develop ShareCaptioner-Video, a superior captioner capable of efficiently generating high-quality captions for arbitrary videos...",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "ShareGPT4Video improves video supervision with dense captions that describe both frame-level detail and precise temporal changes, then scales annotation through a dedicated video captioner.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Naive multi-frame prompting can miss details or confuse temporal order. ShareGPT4Video uses differential captioning to capture changes between frames while supporting videos with varied lengths, resolutions and aspect ratios.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Creates 40K densely captioned videos using GPT-4V and a filtering and annotation pipeline.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Develops ShareCaptioner-Video, 4.8M automatically captioned videos and the ShareGPT4Video-8B understanding model.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "ShareGPT4Video-8B reaches an overall TempCompass score of 61.5 versus 49.9 for VideoLLaVA-7B. On caption-generation action recognition, the corresponding scores are 79.8 and 50.8.",
            "fragment": "S4.T3",
            "locator": "Table 3 · temporal comprehension · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "ShareGPT4Video-8B scores 51.2 on MVBench, compared with the authors’ reproduced 43.0 for VideoLLaVA-7B and 41.3 for LLaMA-VID-7B. Model sizes and training recipes differ, so the table is a system comparison rather than a data-only ablation.",
            "fragment": "S4.T5",
            "locator": "Table 5 · MVBench evaluation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Naive multi-frame video captioning",
              "design": "Can lose fine details or confuse temporal changes when frames are supplied together."
            },
            {
              "method": "ShareGPT4Video",
              "design": "Uses differential captioning to combine frame-level detail with precise descriptions of change over time."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2405.16009",
      "title": "Streaming Long Video Understanding with Large Language Models",
      "authors": [
        {
          "name": "Rui Qian"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Shuangrui Ding"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-05-25",
        "arxivLastUpdated": "2024-05-25"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2405.16009",
        "googleScholar": "hW23VKIAAAAJ:NhqRSupF_l8C",
        "doi": "10.52202/079017-3792"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2405.16009"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:NhqRSupF_l8C"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1238,
        "order": 50
      },
      "shortName": "VideoStreaming",
      "keywords": [
        "VideoStreaming",
        "Long-video understanding",
        "Streaming encoding",
        "Memory propagation",
        "Adaptive memory selection",
        "Video question answering",
        "Efficient multimodal inference",
        "Temporal comprehension"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240516009",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/d7ce06e9293c3d8e6cb3f80b4157f875-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 119336,
        "lastPage": 119360,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/d7ce06e9293c3d8e6cb3f80b4157f875-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2405.16009v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2405.16009v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2405.16009v1"
          }
        },
        "abstract": {
          "text": "This paper presents VideoStreaming, an advanced vision-language large model (VLLM) for video understanding, that capably understands arbitrary-length video with a constant number of video tokens streamingly encoded and adaptively selected. The challenge of video understanding in the vision language area mainly lies in the significant computational burden caused by the great number of tokens extracted from long videos. Previous works rely on sparse sampling or frame compression to reduce tokens. However, such approaches either disregard temporal information in a long time span or sacrifice spatial details, resulting in flawed compression. To address these limitations, our VideoStreaming has two core designs: Memory-Propagated Streaming Encoding and Adaptive Memory Selection. The Memory-Propagated Streaming Encoding architecture segments long videos into short clips and sequentially encodes each clip with a propagated memory. In each iteration, we utilize the encoded results of the preceding clip as historical memory, which is integrated with the current clip to distill a condensed representation that encapsulates the video content up to the current timestamp. After the encoding process, the Adaptive Memory Selection strategy selects a constant number of question-related memories from all the historical memories and feeds them into the LLM to generate informative responses. The question-related selection reduces redundancy within the memories, enabling efficient and precise video understanding. Meanwhile, the disentangled video extraction and reasoning design allows the LLM to answer different questions about a video by directly selecting corresponding memories, without the need to encode the whole video for each question. Our model achieves superior performance and higher efficiency on long video benchmarks, showcasing precise temporal comprehension for detailed question answering.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "VideoStreaming combines propagated clip memory with question-conditioned memory selection so an LLM can answer about long videos using a fixed number of selected video tokens.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Sparse frame sampling loses temporal coverage, while aggressive compression can erase spatial detail. VideoStreaming encodes clips sequentially with historical memory, then selects relevant memories for each question without re-encoding the video.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Introduces Memory-Propagated Streaming Encoding to accumulate video context clip by clip.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Uses Adaptive Memory Selection to separate reusable video encoding from question-specific reasoning.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "VideoStreaming’s 7B answer model plus 1.3B streaming component scores 44.1 on the EgoSchema full test set and 66.2 on Next-QA in the zero-shot evaluation. The 7B LangRepo comparison scores 38.9 and 54.6.",
            "fragment": "S4.T4",
            "locator": "Tables 3–4 · zero-shot video QA · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "With vision-only input, VideoStreaming uses 256 LLM input tokens and 5.32 seconds per question, compared with LLaMA-VID’s 5,477 tokens and 10.47 seconds. Overview/plot/temporal scores are 2.65/3.13/1.88 versus 2.28/2.88/1.46.",
            "fragment": "S4.T6",
            "locator": "Table 6 · vision-only MovieNet-QA · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Sparse sampling or frame compression",
              "design": "Can trade away long-range temporal information or spatial details when reducing video tokens."
            },
            {
              "method": "VideoStreaming",
              "design": "Encodes clips with propagated memory and selects a fixed number of question-related memories for the LLM."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2403.20330",
      "title": "Are We on the Right Way for Evaluating Large Vision-Language Models?",
      "authors": [
        {
          "name": "Lin Chen"
        },
        {
          "name": "Jinsong Li"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zehui Chen"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Yu Qiao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Feng Zhao"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-03-29",
        "arxivLastUpdated": "2024-04-09"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2403.20330",
        "googleScholar": "hW23VKIAAAAJ:UebtZRa9Y70C",
        "doi": "10.52202/079017-0850"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2403.20330"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:UebtZRa9Y70C"
        },
        {
          "type": "code",
          "url": "https://github.com/MMStar-Benchmark/MMStar",
          "repository": "MMStar-Benchmark/MMStar"
        },
        {
          "type": "project",
          "url": "https://mmstar-benchmark.github.io/"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1253,
        "order": 51
      },
      "shortName": "MMStar",
      "keywords": [
        "MMStar",
        "Multimodal evaluation",
        "Visual necessity",
        "Data leakage",
        "Language shortcuts",
        "Vision-language benchmarks",
        "Multimodal gain",
        "Benchmark contamination"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240320330",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/2f8ee6a3d766b426d2618e555b5aeb39-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 27056,
        "lastPage": 27087,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/2f8ee6a3d766b426d2618e555b5aeb39-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2403.20330v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2403.20330v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2403.20330v2"
          }
        },
        "abstract": {
          "text": "Large vision-language models (LVLMs) have recently achieved rapid progress, sparking numerous studies to evaluate their multi-modal capabilities. However, we dig into current evaluation works and identify two primary issues: 1) Visual content is unnecessary for many samples. The answers can be directly inferred from the questions and options, or the world knowledge embedded in LLMs. This phenomenon is prevalent across current benchmarks. For instance, GeminiPro achieves 42.9% on the MMMU benchmark without any visual input, and outperforms the random choice baseline across six benchmarks over 24% on average. 2) Unintentional data leakage exists in LLM and LVLM training. LLM and LVLM could still answer some visual-necessary questions without visual content, indicating the memorizing of these samples within large-scale training data. For example, Sphinx-X-MoE gets 43.6% on MMMU without accessing images, surpassing its LLM backbone with 17.9%. Both problems lead to misjudgments of actual multi-modal gains and potentially misguide the study of LVLM. To this end, we present MMStar, an elite vision-indispensable multi-modal benchmark comprising 1,500 samples meticulously selected by humans. MMStar benchmarks 6 core capabilities and 18 detailed axes, aiming to evaluate LVLMs' multi-modal capacities with carefully balanced and purified samples. These samples are first roughly selected from current benchmarks with an automated pipeline, human review is then involved to ensure each curated sample exhibits visual dependency, minimal data leakage, and requires advanced multi-modal capabilities. Moreover, two metrics are developed to measure data leakage and actual performance gain in multi-modal training. We evaluate 16 leading LVLMs on MMStar to assess their multi-modal capabilities, and on 7 benchmarks with the proposed metrics to investigate their data leakage and actual multi-modal gain.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "MMStar tests whether multimodal gains actually depend on visual input, using visually necessary questions and metrics designed to reveal language shortcuts and possible data leakage.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Many multimodal benchmark questions can be answered without images, and training-set contamination can further inflate scores. MMStar combines automated screening with human review to evaluate visual dependence and actual gains from multimodal training.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Curates 1,500 questions covering six core capabilities and 18 detailed axes.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces metrics for data leakage and multimodal performance gain, evaluated across seven benchmarks.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On six existing multimodal benchmarks, GPT-4V scores 41.2 on average when images are removed, versus 66.0 with images and 35.3 for its GPT-4-Turbo language backbone. The control measures how much benchmark performance remains without visual input.",
            "fragment": "S3.T3",
            "locator": "Table 3 · image-removal control · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "On MMStar with two-shot text-only inference, Gemini-Pro scores 20.6% and GPT-4-Turbo 12.2%, compared with the reported random-choice baseline of 24.6%. This supports the benchmark’s requirement for visual evidence under the tested prompting protocol.",
            "fragment": "S5.T4",
            "locator": "Table 4 · MMStar text-only control · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Multimodal scores without language-only checks",
              "design": "Can conflate genuine image understanding with textual shortcuts or memorized benchmark answers."
            },
            {
              "method": "MMStar",
              "design": "Curates visually necessary questions and measures both possible data leakage and actual multimodal gains."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2404.06512",
      "title": "InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD",
      "authors": [
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Bin Wang"
        },
        {
          "name": "Linke Ouyang"
        },
        {
          "name": "Songyang Zhang"
        },
        {
          "name": "Haodong Duan"
        },
        {
          "name": "Wenwei Zhang"
        },
        {
          "name": "Yining Li"
        },
        {
          "name": "Hang Yan"
        },
        {
          "name": "Yang Gao"
        },
        {
          "name": "Zhe Chen"
        },
        {
          "name": "Xinyue Zhang"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Jingwen Li"
        },
        {
          "name": "Wenhai Wang"
        },
        {
          "name": "Kai Chen"
        },
        {
          "name": "Conghui He"
        },
        {
          "name": "Xingcheng Zhang"
        },
        {
          "name": "Jifeng Dai"
        },
        {
          "name": "Yu Qiao"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "NeurIPS",
        "citationText": "Neural Information Processing Systems (NeurIPS), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-04-09",
        "arxivLastUpdated": "2024-04-09"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2404.06512",
        "googleScholar": "hW23VKIAAAAJ:hqOjcs7Dif8C",
        "doi": "10.52202/079017-1348"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2404.06512"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:hqOjcs7Dif8C"
        },
        {
          "type": "code",
          "url": "https://github.com/InternLM/InternLM-XComposer",
          "repository": "InternLM/InternLM-XComposer"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1273,
        "order": 52
      },
      "shortName": "InternLM-XComposer2-4KHD",
      "keywords": [
        "InternLM-XComposer2-4KHD",
        "High-resolution vision-language models",
        "4K image understanding",
        "Dynamic resolution",
        "Automatic patch configuration",
        "Fine-grained perception",
        "Aspect-ratio preservation",
        "Multimodal foundation models"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240406512",
        "year": 2024,
        "booktitle": "Advances in Neural Information Processing Systems",
        "source": {
          "label": "NeurIPS 2024 proceedings record",
          "url": "https://papers.nips.cc/paper_files/paper/2024/hash/4b06cdddb1cde6624c0be1465c7b800f-Abstract-Conference.html"
        },
        "status": "published",
        "firstPage": 42566,
        "lastPage": 42592,
        "volume": "37",
        "publisher": "Curran Associates, Inc.",
        "pdfURL": "https://proceedings.neurips.cc/paper_files/paper/2024/file/4b06cdddb1cde6624c0be1465c7b800f-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2404.06512v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2404.06512v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2404.06512v1"
          }
        },
        "abstract": {
          "text": "The Large Vision-Language Model (LVLM) field has seen significant advancements, yet its progression has been hindered by challenges in comprehending fine-grained visual content due to limited resolution. Recent efforts have aimed to enhance the high-resolution understanding capabilities of LVLMs, yet they remain capped at approximately 1500 x 1500 pixels and constrained to a relatively narrow resolution range. This paper represents InternLM-XComposer2-4KHD, a groundbreaking exploration into elevating LVLM resolution capabilities up to 4K HD (3840 x 1600) and beyond. Concurrently, considering the ultra-high resolution may not be necessary in all scenarios, it supports a wide range of diverse resolutions from 336 pixels to 4K standard, significantly broadening its scope of applicability. Specifically, this research advances the patch division paradigm by introducing a novel extension: dynamic resolution with automatic patch configuration. It maintains the training image aspect ratios while automatically varying patch counts and configuring layouts based on a pre-trained Vision Transformer (ViT) (336 x 336), leading to dynamic training resolution from 336 pixels to 4K standard. Our research demonstrates that scaling training resolution up to 4K HD leads to consistent performance enhancements without hitting the ceiling of potential improvements. InternLM-XComposer2-4KHD shows superb capability that matches or even surpasses GPT-4V and Gemini Pro in 10 of the 16 benchmarks. The InternLM-XComposer2-4KHD model series with 7B parameters are publicly available at https://github.com/InternLM/InternLM-XComposer.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "InternLM-XComposer2-4KHD uses dynamic patch layouts to support inputs from 336 pixels to 4K HD, improving access to fine-grained visual details across varied resolutions.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Fixed or narrowly bounded image resolutions limit detailed visual understanding. The model adjusts patch counts and layouts while preserving image aspect ratios, reusing a pretrained 336-by-336 vision transformer.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Introduces dynamic resolution with automatic patch configuration for high-resolution LVLM training.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Studies scaling training resolution to 4K HD and releases a 7B-parameter model series.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "IXC2-4KHD raises DocVQA from IXC2-VL’s 57.7 to 90.0, ChartQA from 72.6 to 81.0 and InfoVQA from 34.4 to 68.6. These gains quantify the high-resolution model’s document and chart capabilities.",
            "fragment": "S3.T3",
            "locator": "Table 3 · high-resolution document understanding · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Compared with IXC2-VL, IXC2-4KHD raises MM-Vet from 46.7 to 54.9, while MMStar changes from 55.4 to 54.1 and MMBench-English from 80.7 to 80.2. High-resolution training improves some capabilities while largely retaining others.",
            "fragment": "S3.T4",
            "locator": "Table 4 · general-capability comparison · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Resolution-constrained LVLM input",
              "design": "Restricts access to fine-grained content and accommodates a narrower range of image resolutions."
            },
            {
              "method": "InternLM-XComposer2-4KHD",
              "design": "Dynamically configures patch counts and layouts while preserving aspect ratios from 336 pixels to 4K HD."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2405.12218",
      "title": "MVSGaussian: Fast Generalizable Gaussian Splatting Reconstruction from Multi-View Stereo",
      "authors": [
        {
          "name": "Tianqi Liu"
        },
        {
          "name": "Guangcong Wang"
        },
        {
          "name": "Shoukang Hu"
        },
        {
          "name": "Liao Shen"
        },
        {
          "name": "Xinyi Ye"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Zhiguo Cao"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Ziwei Liu"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "ECCV",
        "citationText": "European Conference on Computer Vision (ECCV), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-05-20",
        "arxivLastUpdated": "2024-07-15"
      },
      "topics": [
        "AIGC"
      ],
      "identifiers": {
        "arxiv": "2405.12218",
        "googleScholar": "hW23VKIAAAAJ:aqlVkmm33-oC",
        "doi": "10.1007/978-3-031-72649-1_3"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2405.12218"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:aqlVkmm33-oC"
        },
        {
          "type": "code",
          "url": "https://github.com/TQTQliu/MVSGaussian",
          "repository": "TQTQliu/MVSGaussian"
        },
        {
          "type": "project",
          "url": "https://mvsgaussian.github.io/"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1292,
        "order": 53
      },
      "shortName": "MVSGaussian",
      "keywords": [
        "MVSGaussian",
        "3D Gaussian Splatting",
        "Multi-view stereo",
        "Generalizable reconstruction",
        "Novel-view synthesis",
        "Real-time rendering",
        "Geometry-aware representations",
        "Per-scene optimization"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240512218",
        "year": 2025,
        "booktitle": "Computer Vision – ECCV 2024",
        "publisher": "Springer Nature Switzerland",
        "source": {
          "label": "ECCV publisher record",
          "url": "https://doi.org/10.1007/978-3-031-72649-1_3"
        },
        "status": "published",
        "firstPage": 37,
        "lastPage": 53,
        "pdfURL": "https://link.springer.com/content/pdf/10.1007/978-3-031-72649-1_3",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2405.12218v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2405.12218v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2405.12218v3"
          }
        },
        "abstract": {
          "text": "We present MVSGaussian, a new generalizable 3D Gaussian representation approach derived from Multi-View Stereo (MVS) that can efficiently reconstruct unseen scenes. Specifically, 1) we leverage MVS to encode geometry-aware Gaussian representations and decode them into Gaussian parameters. 2) To further enhance performance, we propose a hybrid Gaussian rendering that integrates an efficient volume rendering design for novel view synthesis. 3) To support fast fine-tuning for specific scenes, we introduce a multi-view geometric consistent aggregation strategy to effectively aggregate the point clouds generated by the generalizable model, serving as the initialization for per-scene optimization. Compared with previous generalizable NeRF-based methods, which typically require minutes of fine-tuning and seconds of rendering per image, MVSGaussian achieves real-time rendering with better synthesis quality for each scene. Compared with the vanilla 3D-GS, MVSGaussian achieves better view synthesis with less training computational cost. Extensive experiments on DTU, Real Forward-facing, NeRF Synthetic, and Tanks and Temples datasets validate that MVSGaussian attains state-of-the-art performance with convincing generalizability, real-time rendering speed, and fast per-scene optimization.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "MVSGaussian derives geometry-aware Gaussian representations from multi-view stereo, enabling generalizable scene reconstruction, real-time rendering and fast per-scene refinement.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "Generalizable neural rendering can still require slow rendering or lengthy scene adaptation. MVSGaussian predicts Gaussian parameters from multi-view geometry, uses hybrid rendering and aggregates consistent point clouds to initialize further optimization.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Combines multi-view stereo features with a generalizable 3D Gaussian representation and hybrid rendering.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Introduces geometrically consistent aggregation for efficient per-scene initialization.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "MVSGaussian achieves PSNR 28.21 dB, SSIM 0.963 and LPIPS 0.076 on DTU, compared with ENeRF’s 27.61 dB, 0.957 and 0.089. The reported three-view rendering speed is 21.5 fps and memory use 0.876 GB.",
            "fragment": "S5.T1",
            "locator": "Table 1 · DTU, three input views · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "With cascaded depth estimation, combining Gaussian splatting and volume rendering raises DTU PSNR from the splatting-only setting’s 27.48 to 28.21 dB and Tanks-and-Temples PSNR from 21.70 to 23.28 dB.",
            "fragment": "S5.T4",
            "locator": "Table 4 · reconstruction ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Generalizable NeRF-based rendering",
              "design": "Typically requires slower per-image rendering or substantial scene fine-tuning in the compared methods."
            },
            {
              "method": "MVSGaussian",
              "design": "Decodes stereo-derived features into Gaussians and aggregates consistent geometry for fast scene refinement."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2403.15378",
      "title": "Long-CLIP: Unlocking the Long-Text Capability of CLIP",
      "authors": [
        {
          "name": "Beichen Zhang"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Xiaoyi Dong"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "ECCV",
        "citationText": "European Conference on Computer Vision (ECCV), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-03-22",
        "arxivLastUpdated": "2024-07-22"
      },
      "topics": [
        "Vision-Language Models"
      ],
      "identifiers": {
        "arxiv": "2403.15378",
        "googleScholar": "hW23VKIAAAAJ:roLk4NBRz8UC",
        "doi": "10.1007/978-3-031-72983-6_18"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2403.15378"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:roLk4NBRz8UC"
        },
        {
          "type": "code",
          "url": "https://github.com/beichenzbc/Long-CLIP",
          "repository": "beichenzbc/Long-CLIP"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1312,
        "order": 54
      },
      "shortName": "Long-CLIP",
      "keywords": [
        "Long-CLIP",
        "Long-text image retrieval",
        "CLIP",
        "Contrastive vision-language learning",
        "Positional embedding",
        "Efficient fine-tuning",
        "Detailed text-to-image generation",
        "Zero-shot generalization"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240315378",
        "year": 2025,
        "booktitle": "Computer Vision – ECCV 2024",
        "publisher": "Springer Nature Switzerland",
        "source": {
          "label": "ECCV publisher record",
          "url": "https://doi.org/10.1007/978-3-031-72983-6_18"
        },
        "status": "published",
        "firstPage": 310,
        "lastPage": 325,
        "pdfURL": "https://link.springer.com/content/pdf/10.1007/978-3-031-72983-6_18",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v3",
            "url": "https://arxiv.org/abs/2403.15378v3"
          },
          "paper": {
            "label": "Paper · arXiv v3",
            "url": "https://arxiv.org/pdf/2403.15378v3",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v3",
            "url": "https://arxiv.org/html/2403.15378v3"
          }
        },
        "abstract": {
          "text": "Contrastive Language-Image Pre-training (CLIP) has been the cornerstone for zero-shot classification, text-image retrieval, and text-image generation by aligning image and text modalities. Despite its widespread adoption, a significant limitation of CLIP lies in the inadequate length of text input. The length of the text token is restricted to 77, and an empirical study shows the actual effective length is even less than 20. This prevents CLIP from handling detailed descriptions, limiting its applications for image retrieval and text-to-image generation with extensive prerequisites. To this end, we propose Long-CLIP as a plug-and-play alternative to CLIP that supports long-text input, retains or even surpasses its zero-shot generalizability, and aligns the CLIP latent space, making it readily replace CLIP without any further adaptation in downstream frameworks. Nevertheless, achieving this goal is far from straightforward, as simplistic fine-tuning can result in a significant degradation of CLIP's performance. Moreover, substituting the text encoder with a language model supporting longer contexts necessitates pretraining with vast amounts of data, incurring significant expenses. Accordingly, Long-CLIP introduces an efficient fine-tuning solution on CLIP with two novel strategies designed to maintain the original capabilities, including (1) a knowledge-preserved stretching of positional embedding and (2) a primary component matching of CLIP features. With leveraging just one million extra long text-image pairs, Long-CLIP has shown the superiority to CLIP for about 20% in long caption text-image retrieval and 6% in traditional text-image retrieval tasks, e.g., COCO and Flickr30k. Furthermore, Long-CLIP offers enhanced capabilities for generating images from detailed text descriptions by replacing CLIP in a plug-and-play manner.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "takeaway": {
          "text": "Long-CLIP extends CLIP to detailed text descriptions through efficient fine-tuning while preserving the embedding alignment needed for plug-and-play downstream use.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "summary": {
          "text": "CLIP's short text context restricts retrieval and generation with detailed descriptions, but naive fine-tuning can damage generalization. Long-CLIP stretches positional embeddings while preserving knowledge and matches principal feature components to retain CLIP capabilities.",
          "source": "abstract",
          "locator": "arXiv abstract · v3"
        },
        "contributions": [
          {
            "text": "Introduces knowledge-preserved positional-embedding stretching and primary component matching.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          },
          {
            "text": "Uses one million additional long text-image pairs to extend context without training a replacement encoder from scratch.",
            "source": "abstract",
            "locator": "arXiv abstract · v3"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With ViT-L/14, Long-CLIP raises Urban-200 image-to-text and text-to-image R@1 from 47.0% to 81.5%. On short-caption COCO, text-to-image R@1 rises from 35.4% to 46.3%, while direct fine-tuning lowers it to 23.1%.",
            "fragment": "S4.T3",
            "locator": "Tables 1–3 · long- and short-caption evaluation · arXiv v3",
            "source": "fullText"
          },
          {
            "text": "Combining knowledge-preserving stretching and primary-component matching gives 66.8% ImageNet accuracy versus 55.1% with neither component. COCO text-to-image R@5 rises from 43.4% to 65.8%, while Urban-200 text-to-image R@1 changes from 78.0% to 79.0%.",
            "fragment": "S4.T4",
            "locator": "Table 4 · position and feature-matching ablation · arXiv v3",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v3",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Naive CLIP fine-tuning for longer text",
              "design": "Can degrade the original model's capabilities; replacing the encoder also entails costly pretraining."
            },
            {
              "method": "Long-CLIP",
              "design": "Stretches positional embeddings and preserves feature knowledge through efficient fine-tuning on long captions."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2312.03818",
      "title": "Alpha-CLIP: A CLIP Model Focusing on Wherever You Want",
      "authors": [
        {
          "name": "Zeyi Sun"
        },
        {
          "name": "Ye Fang"
        },
        {
          "name": "Tong Wu"
        },
        {
          "name": "Pan Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Shu Kong"
        },
        {
          "name": "Yuanjun Xiong"
        },
        {
          "name": "Dahua Lin"
        },
        {
          "name": "Jiaqi Wang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2023-12-06",
        "arxivLastUpdated": "2023-12-13"
      },
      "topics": [
        "Vision-Language Models"
      ],
      "identifiers": {
        "arxiv": "2312.03818",
        "googleScholar": "hW23VKIAAAAJ:WF5omc3nYNoC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2312.03818"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:WF5omc3nYNoC"
        },
        {
          "type": "code",
          "url": "https://github.com/SunzeY/AlphaCLIP",
          "repository": "SunzeY/AlphaCLIP"
        },
        {
          "type": "project",
          "url": "https://aleafy.github.io/alpha-clip/"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/spaces/Zery/Alpha_CLIP_ImgVar",
          "label": "Space",
          "variant": "space"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1331,
        "order": 55
      },
      "shortName": "Alpha-CLIP",
      "keywords": [
        "Alpha-CLIP",
        "Region-aware CLIP",
        "Alpha channel",
        "Region-text alignment",
        "Visual prompting",
        "Open-world recognition",
        "Controllable generation",
        "Multimodal representations"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv231203818",
        "year": 2024,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2024 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2024/html/Sun_Alpha-CLIP_A_CLIP_Model_Focusing_on_Wherever_You_Want_CVPR_2024_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 13019,
        "lastPage": 13029,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2024/papers/Sun_Alpha-CLIP_A_CLIP_Model_Focusing_on_Wherever_You_Want_CVPR_2024_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2312.03818v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2312.03818v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2312.03818v2"
          }
        },
        "abstract": {
          "text": "Contrastive Language-Image Pre-training (CLIP) plays an essential role in extracting valuable content information from images across diverse tasks. It aligns textual and visual modalities to comprehend the entire image, including all the details, even those irrelevant to specific tasks. However, for a finer understanding and controlled editing of images, it becomes crucial to focus on specific regions of interest, which can be indicated as points, masks, or boxes by humans or perception models. To fulfill the requirements, we introduce Alpha-CLIP, an enhanced version of CLIP with an auxiliary alpha channel to suggest attentive regions and fine-tuned with constructed millions of RGBA region-text pairs. Alpha-CLIP not only preserves the visual recognition ability of CLIP but also enables precise control over the emphasis of image contents. It demonstrates effectiveness in various tasks, including but not limited to open-world recognition, multimodal large language models, and conditional 2D / 3D generation. It has a strong potential to serve as a versatile tool for image-related tasks.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Alpha-CLIP adds an alpha channel to focus CLIP on a selected image region while retaining whole-image recognition and supporting region-aware understanding and generation.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Whole-image embeddings can include details irrelevant to a user's intended region. Alpha-CLIP accepts region indications from points, masks or boxes and learns region-text alignment from millions of RGBA examples.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Extends CLIP's visual input with an auxiliary alpha channel for controllable regional attention.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Constructs region-text training pairs and studies transfer to recognition, multimodal language models and 2D/3D generation.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With ViT-L/14 on ImageNet-S, Alpha-CLIP with a foreground alpha mask achieves 77.41% top-1 accuracy versus original CLIP’s 73.48%. Top-5 accuracy rises from 91.60% to 94.45%.",
            "fragment": "S4.T2",
            "locator": "Table 2 · region-aware zero-shot classification · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Alpha-CLIP obtains 73.37% top-1 accuracy with a whole-image alpha map, 75.62% with a rectangular box and 77.41% with a foreground mask. The graded comparison isolates the value of increasingly precise region input.",
            "fragment": "S4.T3",
            "locator": "Table 3 · alpha-input ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Whole-image CLIP",
              "design": "Represents the complete image, including details outside a task's region of interest."
            },
            {
              "method": "Alpha-CLIP",
              "design": "Adds an alpha channel and region-text training so the representation can emphasize a specified region."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2401.15914",
      "title": "Overcoming the Pitfalls of Vision-Language Model Finetuning for OOD Generalization",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Hanlin Goh"
        },
        {
          "name": "Josh Susskind"
        },
        {
          "name": "Chen Huang"
        }
      ],
      "publication": {
        "year": 2024,
        "venueGroup": "ICLR",
        "citationText": "International Conference on Learning Representations (ICLR), 2024"
      },
      "dates": {
        "arxivFirstPosted": "2024-01-29",
        "arxivLastUpdated": "2024-04-16"
      },
      "topics": [
        "Vision-Language Models"
      ],
      "identifiers": {
        "arxiv": "2401.15914",
        "googleScholar": "hW23VKIAAAAJ:ufrVoPGSRksC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2401.15914"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:ufrVoPGSRksC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1352,
        "order": 56
      },
      "shortName": "OGEN",
      "keywords": [
        "OGEN",
        "Out-of-distribution generalization",
        "Vision-language fine-tuning",
        "Unknown-class recognition",
        "Feature synthesis",
        "Adaptive self-distillation",
        "Prompt learning",
        "Open-domain recognition"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv240115914",
        "year": 2024,
        "booktitle": "International Conference on Learning Representations",
        "source": {
          "label": "ICLR 2024 proceedings record",
          "url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/8140f43b06c9c7e14fb96953caed2665-Abstract-Conference.html"
        },
        "status": "published",
        "authors": [
          {
            "name": "Yuhang Zang"
          },
          {
            "name": "Hanlin Goh"
          },
          {
            "name": "Joshua Susskind"
          },
          {
            "name": "Chen Huang"
          }
        ],
        "firstPage": 30327,
        "lastPage": 30342,
        "volume": "2024",
        "pdfURL": "https://proceedings.iclr.cc/paper_files/paper/2024/file/8140f43b06c9c7e14fb96953caed2665-Paper-Conference.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2401.15914v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2401.15914v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2401.15914v2"
          }
        },
        "abstract": {
          "text": "Existing vision-language models exhibit strong generalization on a variety of visual domains and tasks. However, such models mainly perform zero-shot recognition in a closed-set manner, and thus struggle to handle open-domain visual concepts by design. There are recent finetuning methods, such as prompt learning, that not only study the discrimination between in-distribution (ID) and out-of-distribution (OOD) samples, but also show some improvements in both ID and OOD accuracies. In this paper, we first demonstrate that vision-language models, after long enough finetuning but without proper regularization, tend to overfit the known classes in the given dataset, with degraded performance on unknown classes. Then we propose a novel approach OGEN to address this pitfall, with the main focus on improving the OOD GENeralization of finetuned models. Specifically, a class-conditional feature generator is introduced to synthesize OOD features using just the class name of any unknown class. Such synthesized features will provide useful knowledge about unknowns and help regularize the decision boundary between ID and OOD data when optimized jointly. Equally important is our adaptive self-distillation mechanism to regularize our feature generation model during joint optimization, i.e., adaptively transferring knowledge between model states to further prevent overfitting. Experiments validate that our method yields convincing gains in OOD generalization performance in different settings. Code: https://github.com/apple/ml-ogen.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "OGEN improves out-of-distribution generalization during vision-language fine-tuning by synthesizing features for unknown classes and using adaptive self-distillation to limit overfitting.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Prolonged fine-tuning can improve known classes while degrading recognition of unknown ones. OGEN regularizes this process with a class-conditional generator that synthesizes unknown-class features from class names and transfers knowledge between model states.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Identifies overfitting to known classes as a failure mode of insufficiently regularized vision-language fine-tuning.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Combines unknown-class feature generation with adaptive self-distillation during joint optimization.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Across 11 datasets, adding OGEN to CoOp raises new-class accuracy from 63.22% to 69.54% while base-class accuracy changes from 82.69% to 83.47%. The harmonic mean rises from 71.66% to 75.87% in the main comparison.",
            "fragment": "S4.T1",
            "locator": "Table 1 · base-to-new generalization · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "In the ablation, the feature generator alone raises new-class accuracy from 63.22% to 69.02%; adding adaptive self-distillation reaches 69.54%. Joint class extrapolation scores 69.02% on new classes versus 64.08% without extrapolation.",
            "fragment": "S4.T4",
            "locator": "Tables 3–4 · feature-generation ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Long fine-tuning without proper regularization",
              "design": "Can overfit known categories and degrade unknown-class recognition."
            },
            {
              "method": "OGEN",
              "design": "Synthesizes unknown-class features and applies adaptive self-distillation during joint optimization."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2305.18279",
      "title": "Contextual Object Detection with Multimodal Large Language Models",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Jun Han"
        },
        {
          "name": "Kaiyang Zhou"
        },
        {
          "name": "Chen Change Loy"
        }
      ],
      "publication": {
        "year": 2025,
        "venueGroup": "IJCV",
        "citationText": "International Journal of Computer Vision (IJCV), 2025"
      },
      "dates": {
        "arxivFirstPosted": "2023-05-29",
        "arxivLastUpdated": "2024-08-12"
      },
      "topics": [
        "Multimodal Large Language Models"
      ],
      "identifiers": {
        "arxiv": "2305.18279",
        "googleScholar": "hW23VKIAAAAJ:xtRiw3GOFMkC",
        "doi": "10.1007/s11263-024-02214-4"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2305.18279"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:xtRiw3GOFMkC"
        },
        {
          "type": "code",
          "url": "https://github.com/yuhangzang/ContextDET",
          "repository": "yuhangzang/ContextDET"
        },
        {
          "type": "project",
          "url": "https://www.mmlab-ntu.com/project/contextdet/index.html"
        },
        {
          "type": "huggingface",
          "url": "https://huggingface.co/spaces/yuhangzang/ContextDet-Demo",
          "label": "Space",
          "variant": "space"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1367,
        "order": 57
      },
      "shortName": "ContextDET",
      "keywords": [
        "ContextDET",
        "Contextual object detection",
        "Generate-then-detect",
        "Multimodal large language models",
        "Visual grounding",
        "CODE benchmark",
        "Open-vocabulary detection",
        "Human-AI interaction"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv230518279",
        "year": 2025,
        "journal": "International Journal of Computer Vision",
        "publisher": "Springer Science and Business Media LLC",
        "source": {
          "label": "IJCV publisher record",
          "url": "https://doi.org/10.1007/s11263-024-02214-4"
        },
        "status": "published",
        "month": 2,
        "volume": "133",
        "number": "2",
        "firstPage": 825,
        "lastPage": 843,
        "pdfURL": "https://link.springer.com/content/pdf/10.1007/s11263-024-02214-4.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2305.18279v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2305.18279v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2305.18279v2"
          }
        },
        "abstract": {
          "text": "Recent Multimodal Large Language Models (MLLMs) are remarkable in vision-language tasks, such as image captioning and question answering, but lack the essential perception ability, i.e., object detection. In this work, we address this limitation by introducing a novel research problem of contextual object detection -- understanding visible objects within different human-AI interactive contexts. Three representative scenarios are investigated, including the language cloze test, visual captioning, and question answering. Moreover, we present ContextDET, a unified multimodal model that is capable of end-to-end differentiable modeling of visual-language contexts, so as to locate, identify, and associate visual objects with language inputs for human-AI interaction. Our ContextDET involves three key submodels: (i) a visual encoder for extracting visual representations, (ii) a pre-trained LLM for multimodal context decoding, and (iii) a visual decoder for predicting bounding boxes given contextual object words. The new generate-then-detect framework enables us to detect object words within human vocabulary. Extensive experiments show the advantages of ContextDET on our proposed CODE benchmark, open-vocabulary detection, and referring image segmentation. Github: https://github.com/yuhangzang/ContextDET.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "ContextDET grounds objects within language interaction by generating context-relevant object words and then detecting their image locations in an end-to-end multimodal model.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Image captioning and question answering do not by themselves provide contextual object detection. ContextDET connects a visual encoder, a pretrained language model and a bounding-box decoder to associate visual objects with the current language context.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Defines contextual object detection across language cloze tests, visual captioning and question answering.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Introduces the generate-then-detect framework and the CODE benchmark.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "ContextDET reaches 48.7% top-1 cloze accuracy and 8.1 QA AP@1, compared with LLaVA-1.5 + GLIP’s 42.9% and 6.5. Captioning AP@1 is 6.4 versus 5.5 in the same comparison.",
            "fragment": "S4.T2",
            "locator": "Table 2 · CODE validation benchmark · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Adding local visual tokens raises cloze top-1 accuracy from 30.9% to 48.7% and AP@1 from 4.0 to 10.4 in the ablation. These results support the role of local evidence in contextual object prediction.",
            "fragment": "S4.T5",
            "locator": "Table 4 · local-visual-token ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Captioning and question answering alone",
              "design": "Produces language responses without explicit contextual bounding-box predictions."
            },
            {
              "method": "ContextDET",
              "design": "Generates context-relevant object words and decodes their locations within a differentiable multimodal model."
            }
          ]
        }
      }
    },
    {
      "id": "scholar:hW23VKIAAAAJ:eQOLeE2rZwMC",
      "title": "Real-World Object Detection",
      "authors": [
        {
          "name": "Yuhang Zang"
        }
      ],
      "publication": {
        "year": 2023,
        "venueGroup": "Thesis",
        "citationText": "PhD Thesis, Nanyang Technological University, 2023"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "googleScholar": "hW23VKIAAAAJ:eQOLeE2rZwMC",
        "doi": "10.32657/10356/171489"
      },
      "links": [
        {
          "type": "paper",
          "url": "https://dr.ntu.edu.sg/server/api/core/bitstreams/170b1cb3-088b-4202-9268-2dec2f1832e0/content",
          "label": "PDF"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:eQOLeE2rZwMC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1388,
        "order": 58
      },
      "shortName": "Real-World Object Detection",
      "keywords": [
        "Real-world object detection",
        "Long-tailed recognition",
        "Open-vocabulary detection",
        "Semi-supervised learning",
        "Feature augmentation",
        "Vision-language transfer",
        "Unified Prompt Tuning",
        "Contextual object detection",
        "Doctoral thesis"
      ],
      "citation": {
        "type": "phdthesis",
        "key": "zang2023realworld",
        "year": 2023,
        "school": "Nanyang Technological University",
        "source": {
          "label": "NTU thesis record",
          "url": "https://doi.org/10.32657/10356/171489"
        },
        "status": "thesis",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "NTU institutional abstract",
            "url": "https://dr.ntu.edu.sg/server/api/core/items/8bbf3bc6-9ac5-46dd-ae97-7717bd808708"
          },
          "paper": {
            "label": "Doctoral thesis · NTU",
            "url": "https://dr.ntu.edu.sg/server/api/core/bitstreams/170b1cb3-088b-4202-9268-2dec2f1832e0/content",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Doctoral thesis · NTU",
            "url": "https://dr.ntu.edu.sg/server/api/core/bitstreams/170b1cb3-088b-4202-9268-2dec2f1832e0/content",
            "encodingFormat": "application/pdf"
          }
        },
        "abstract": {
          "text": "Object detection is a fundamental computer vision task that estimates object classification labels and location coordinates in images. Previous studies have consistently boosted the performance of object detectors. However, the real-world scenario introduces significant obstacles that hinder their overall effectiveness. In this thesis, we will concentrate specifically on two such challenges. The first challenge is the long-tailed data distribution, where real-world data often exhibits a significant imbalance in the number of images per category. Directly training an object detector on long-tailed data can introduce a bias toward head class objects, causing the omission of tail class objects. The second challenge is about generalizing to test samples from unseen classes that are not included in the training set. Detectors frequently make inaccurate classification predictions for objects from these unseen classes, including misclassifications as background or known categories. This thesis explores solutions to address the aforementioned challenges. For the long-tailed problem, we concentrate on two approaches: data augmentation (FASA) and semi-supervised learning (CascadeMatch). To enhance the detector's generalization ability, we investigate leveraging prior knowledge from Vision and Language Models (OV-DETR, UPT) or Multimodal Large Language Models (ContextDET). We first propose a simple yet effective method, Feature Augmentation and Sampling Adaptation (FASA), that addresses the long-tailed issue by augmenting the feature space, especially for rare classes. FASA does not require any elaborate loss design and removes the need for inter-class transfer learning that often involves large costs and manually-defined head/tail class groups. We show FASA is a fast, generic method that can be easily plugged into standard or long-tailed segmentation frameworks, with consistent performance gains and little added cost. Second, we propose CascadeMatch, a novel pseudo-labeling-based object detector that uses semi-supervised supervision to effectively tackle the long-tailed problem. CascadeMatch features a cascade network architecture that consists of multi-stage detection heads with incremental confidence thresholds. To avoid confirmation bias, each detection head is trained by the ensemble pseudo labels of all detection heads. To take into account the class-imbalance problem in real-world data that causes neural networks to give a higher/lower confidence to many/few-shot classes, we propose class-specific self-adaptive confidence thresholds, which are automatically tuned from labeled data with minimal human intervention. Third, to achieve generalization on unseen classes during testing, we propose a novel open-vocabulary detector called OV-DETR. Once trained, OV-DETR can detect any object given its class name or an exemplar image. For training, we choose to condition the Transformer decoder on the input embeddings obtained from a pre-trained vision-language model like CLIP, in order to enable matching for both text and image queries. With extensive experiments on LVIS and COCO datasets, we demonstrate that our OV-DETR achieves non-trivial improvements over the baseline methods. Fourth, we present a systematic study of unimodal prompt tuning methods, which serve as popular transfer learning paradigms for vision-language models like CLIP. A major finding is that none of the unimodal prompt tuning methods performs consistently well: text prompt tuning fails on data with high intra-class visual variances while visual prompt tuning cannot handle low inter-class variances. To combine the best from both worlds, we propose a conceptually simple approach called Unified Prompt Tuning (UPT), which learns a tiny neural network to jointly optimize prompts across different modalities. Extensive experiments on over 11 vision datasets show that UPT achieves a better trade-off than the unimodal counterparts on existing benchmarks. Finally, we introduce a novel research problem of contextual object detection---understanding visible objects within different human-AI interactive contexts. Three representative scenarios are investigated, including the language cloze test, visual captioning, and question answering. Moreover, we present ContextDET, a unified multimodal model that is capable of end-to-end differentiable modeling of visual-language contexts, so as to locate, identify, and associate visual objects with language inputs for human-AI interaction. Extensive experiments show the advantages of ContextDET on a series of tasks, including our proposed contextual object detection, open-vocabulary detection, and referring image segmentation.",
          "source": "abstract",
          "locator": "NTU institutional abstract"
        },
        "takeaway": {
          "text": "This thesis addresses two obstacles to real-world object detection—long-tailed data and unseen categories—through adaptive training, vision-language transfer and context-aware grounding.",
          "source": "abstract",
          "locator": "NTU institutional abstract"
        },
        "summary": {
          "text": "Detectors trained on imbalanced, closed-vocabulary data can overlook rare objects and misclassify unseen ones. The thesis develops feature augmentation and semi-supervised learning for rare categories, then uses language-model knowledge to support open-vocabulary recognition and contextual detection.",
          "source": "abstract",
          "locator": "NTU institutional abstract"
        },
        "contributions": [
          {
            "text": "Develops FASA and CascadeMatch to address rare-class data scarcity and long-tailed semi-supervised detection.",
            "source": "abstract",
            "locator": "NTU institutional abstract"
          },
          {
            "text": "Studies OV-DETR, Unified Prompt Tuning and ContextDET to transfer vision-language knowledge and connect detection with interaction context.",
            "source": "abstract",
            "locator": "NTU institutional abstract"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Chapter 3 reports that, on LVIS v1.0, feature augmentation alone raises overall mask AP from 20.8 to 22.3 and rare-class AP from 8.0 to 12.7. Adding adaptive feature sampling reaches 23.7 and 17.8, respectively.",
            "page": 60,
            "locator": "Chapter 3, Table 3.1, p. 32",
            "source": "fullText"
          },
          {
            "text": "Chapter 6 evaluates unified prompt tuning in a 16-shot setting averaged over three runs. UPT reaches 72.63% ImageNet accuracy and a 59.98% out-of-domain average, compared with 71.51% and 59.28% for CoOp.",
            "page": 125,
            "locator": "Chapter 6, Table 6.1, p. 97",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "NTU institutional abstract",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Long-tailed detection strategy",
              "design": "FASA augments rare-class features; CascadeMatch uses adaptive, ensemble-supervised pseudo-labeling."
            },
            {
              "method": "Unseen-class generalization strategy",
              "design": "OV-DETR and UPT transfer vision-language knowledge; ContextDET adds language-context-dependent localization."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2210.07225",
      "title": "Unified Vision and Language Prompt Learning",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Kaiyang Zhou"
        },
        {
          "name": "Chen Huang"
        },
        {
          "name": "Chen Change Loy"
        }
      ],
      "publication": {
        "year": 2022,
        "venueGroup": "arXiv",
        "citationText": "arXiv 2022"
      },
      "dates": {
        "arxivFirstPosted": "2022-10-13",
        "arxivLastUpdated": "2022-10-13"
      },
      "topics": [
        "Vision-Language Models"
      ],
      "identifiers": {
        "arxiv": "2210.07225",
        "googleScholar": "hW23VKIAAAAJ:Tyk-4Ss8FVUC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2210.07225"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:Tyk-4Ss8FVUC"
        },
        {
          "type": "code",
          "url": "https://github.com/yuhangzang/UPT",
          "repository": "yuhangzang/UPT"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1403,
        "order": 59
      },
      "shortName": "UPT",
      "keywords": [
        "Unified Prompt Tuning",
        "UPT",
        "Multimodal prompt learning",
        "Visual prompts",
        "Text prompts",
        "Few-shot learning",
        "Domain generalization",
        "Parameter-efficient adaptation"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv221007225",
        "year": 2022,
        "journal": "arXiv preprint arXiv:2210.07225",
        "source": {
          "label": "arXiv citation record",
          "url": "https://arxiv.org/abs/2210.07225"
        },
        "status": "preprint",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2210.07225v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2210.07225v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2210.07225v1"
          }
        },
        "abstract": {
          "text": "Prompt tuning, a parameter- and data-efficient transfer learning paradigm that tunes only a small number of parameters in a model's input space, has become a trend in the vision community since the emergence of large vision-language models like CLIP. We present a systematic study on two representative prompt tuning methods, namely text prompt tuning and visual prompt tuning. A major finding is that none of the unimodal prompt tuning methods performs consistently well: text prompt tuning fails on data with high intra-class visual variances while visual prompt tuning cannot handle low inter-class variances. To combine the best from both worlds, we propose a simple approach called Unified Prompt Tuning (UPT), which essentially learns a tiny neural network to jointly optimize prompts across different modalities. Extensive experiments on over 11 vision datasets show that UPT achieves a better trade-off than the unimodal counterparts on few-shot learning benchmarks, as well as on domain generalization benchmarks. Code and models will be released to facilitate future research.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "Unified Prompt Tuning jointly optimizes visual and textual prompts, combining complementary strengths that neither unimodal prompting method consistently provides alone.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Text prompting struggles with high within-class visual variation, while visual prompting struggles with low between-class variation. UPT learns a small network that coordinates prompts across modalities for parameter- and data-efficient transfer.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Systematically analyzes the complementary limitations of text and visual prompt tuning.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Introduces a lightweight joint optimization approach for vision-language prompts.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "Averaged over three runs, UPT reaches 72.63% ImageNet accuracy and 59.98% out-of-domain average, compared with CoOp’s 71.51% and 59.28%. The target set includes ImageNet-V2, Sketch, A and R.",
            "fragment": "S3.T1",
            "locator": "Table 1 · 16-shot domain generalization · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Across 11 datasets in the 16-shot setting, UPT achieves 81.44% average accuracy versus 78.70% for independently joint-trained visual and textual prompts, 77.88% for shared prompts and 78.73% for an MLP coupling.",
            "fragment": "S3.T2",
            "locator": "Table 2 · multimodal-prompt design ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Text prompt tuning",
              "design": "Struggles in the studied settings with high within-class visual variation."
            },
            {
              "method": "Visual prompt tuning",
              "design": "Struggles in the studied settings with low between-class variation."
            },
            {
              "method": "Unified Prompt Tuning",
              "design": "Jointly optimizes prompts across the two modalities through a small neural network."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2305.14813",
      "title": "Semi-Supervised and Long-Tailed Object Detection with CascadeMatch",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Kaiyang Zhou"
        },
        {
          "name": "Chen Huang"
        },
        {
          "name": "Chen Change Loy"
        }
      ],
      "publication": {
        "year": 2023,
        "venueGroup": "IJCV",
        "citationText": "International Journal of Computer Vision (IJCV), 2023"
      },
      "dates": {
        "arxivFirstPosted": "2023-05-24",
        "arxivLastUpdated": "2023-05-24"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2305.14813",
        "googleScholar": "hW23VKIAAAAJ:W7OEmFMy1HYC",
        "doi": "10.1007/s11263-022-01738-x"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2305.14813"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:W7OEmFMy1HYC"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1422,
        "order": 60
      },
      "shortName": "CascadeMatch",
      "keywords": [
        "CascadeMatch",
        "Semi-supervised object detection",
        "Long-tailed detection",
        "Pseudo-labeling",
        "Adaptive confidence thresholds",
        "Confirmation bias",
        "Ensemble supervision",
        "Sparse annotations"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv230514813",
        "year": 2023,
        "journal": "International Journal of Computer Vision",
        "publisher": "Springer Science and Business Media LLC",
        "source": {
          "label": "IJCV publisher record",
          "url": "https://doi.org/10.1007/s11263-022-01738-x"
        },
        "status": "published",
        "month": 4,
        "volume": "131",
        "number": "4",
        "firstPage": 987,
        "lastPage": 1001,
        "pdfURL": "https://link.springer.com/content/pdf/10.1007/s11263-022-01738-x.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2305.14813v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2305.14813v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2305.14813v1"
          }
        },
        "abstract": {
          "text": "This paper focuses on long-tailed object detection in the semi-supervised learning setting, which poses realistic challenges, but has rarely been studied in the literature. We propose a novel pseudo-labeling-based detector called CascadeMatch. Our detector features a cascade network architecture, which has multi-stage detection heads with progressive confidence thresholds. To avoid manually tuning the thresholds, we design a new adaptive pseudo-label mining mechanism to automatically identify suitable values from data. To mitigate confirmation bias, where a model is negatively reinforced by incorrect pseudo-labels produced by itself, each detection head is trained by the ensemble pseudo-labels of all detection heads. Experiments on two long-tailed datasets, i.e., LVIS and COCO-LT, demonstrate that CascadeMatch surpasses existing state-of-the-art semi-supervised approaches -- across a wide range of detection architectures -- in handling long-tailed object detection. For instance, CascadeMatch outperforms Unbiased Teacher by 1.9 AP Fix on LVIS when using a ResNet50-based Cascade R-CNN structure, and by 1.7 AP Fix when using Sparse R-CNN with a Transformer encoder. We also show that CascadeMatch can even handle the challenging sparsely annotated object detection problem.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "CascadeMatch improves semi-supervised long-tailed detection through progressive cascade thresholds, adaptive pseudo-label mining and ensemble supervision that reduces confirmation bias.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Rare categories make reliable pseudo-label selection difficult, while incorrect labels can reinforce themselves during training. CascadeMatch uses multiple detection heads, learns confidence thresholds from data and supervises each head with their ensemble predictions.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Introduces a cascade pseudo-labeling architecture with progressive, automatically adapted confidence thresholds.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Uses ensemble pseudo-labels across detection heads and studies sparsely annotated detection.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With a 12-epoch schedule, CascadeMatch raises fixed AP from the supervised baseline’s 26.3 to 30.5 and rare-class fixed AP from 19.7 to 23.1, averaged over three random seeds. Soft Teacher reaches 29.2 and 21.1 in the same setting.",
            "fragment": "S4.T7",
            "locator": "Table 7 · LVIS v1.0, Cascade R-CNN R50-FPN · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "Cascade pseudo-labeling alone gives fixed AP 30.1; adaptive pseudo-label mining alone gives 28.9; using both reaches 30.5, compared with 26.3 without unlabeled data. Rare-class fixed AP reaches 23.1 with both components.",
            "fragment": "S4.T3",
            "locator": "Table 2 · pseudo-labeling ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Self-generated pseudo-label supervision",
              "design": "Can reinforce a detector's own incorrect labels and requires suitable confidence thresholds."
            },
            {
              "method": "CascadeMatch",
              "design": "Uses progressive adaptive thresholds and supervises each detection head with ensemble pseudo-labels."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2203.11876",
      "title": "Open-Vocabulary DETR with Conditional Matching",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Wei Li"
        },
        {
          "name": "Kaiyang Zhou"
        },
        {
          "name": "Chen Huang"
        },
        {
          "name": "Chen Change Loy"
        }
      ],
      "publication": {
        "year": 2022,
        "venueGroup": "ECCV",
        "citationText": "European Conference on Computer Vision (ECCV), 2022"
      },
      "dates": {
        "arxivFirstPosted": "2022-03-22",
        "arxivLastUpdated": "2022-11-30"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2203.11876",
        "googleScholar": "hW23VKIAAAAJ:IjCSPb-OGe4C",
        "doi": "10.1007/978-3-031-20077-9_7"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2203.11876"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:IjCSPb-OGe4C"
        },
        {
          "type": "code",
          "url": "https://github.com/yuhangzang/OV-DETR",
          "repository": "yuhangzang/OV-DETR"
        },
        {
          "type": "project",
          "url": "https://www.mmlab-ntu.com/project/ovdetr/index.html"
        }
      ],
      "display": {
        "new": false,
        "badges": [
          "Oral"
        ],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1437,
        "order": 61
      },
      "shortName": "OV-DETR",
      "keywords": [
        "OV-DETR",
        "Open-vocabulary object detection",
        "Conditional matching",
        "DETR",
        "Vision-language transfer",
        "Exemplar-based detection",
        "CLIP embeddings",
        "Novel-category localization"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv220311876",
        "year": 2022,
        "booktitle": "Computer Vision – ECCV 2022",
        "publisher": "Springer Nature Switzerland",
        "source": {
          "label": "ECCV publisher record",
          "url": "https://doi.org/10.1007/978-3-031-20077-9_7"
        },
        "status": "published",
        "firstPage": 106,
        "lastPage": 122,
        "pdfURL": "https://link.springer.com/content/pdf/10.1007/978-3-031-20077-9_7",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2203.11876v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2203.11876v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2203.11876v2"
          }
        },
        "abstract": {
          "text": "Open-vocabulary object detection, which is concerned with the problem of detecting novel objects guided by natural language, has gained increasing attention from the community. Ideally, we would like to extend an open-vocabulary detector such that it can produce bounding box predictions based on user inputs in form of either natural language or exemplar image. This offers great flexibility and user experience for human-computer interaction. To this end, we propose a novel open-vocabulary detector based on DETR -- hence the name OV-DETR -- which, once trained, can detect any object given its class name or an exemplar image. The biggest challenge of turning DETR into an open-vocabulary detector is that it is impossible to calculate the classification cost matrix of novel classes without access to their labeled images. To overcome this challenge, we formulate the learning objective as a binary matching one between input queries (class name or exemplar image) and the corresponding objects, which learns useful correspondence to generalize to unseen queries during testing. For training, we choose to condition the Transformer decoder on the input embeddings obtained from a pre-trained vision-language model like CLIP, in order to enable matching for both text and image queries. With extensive experiments on LVIS and COCO datasets, we demonstrate that our OV-DETR -- the first end-to-end Transformer-based open-vocabulary detector -- achieves non-trivial improvements over current state of the arts.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "OV-DETR reformulates detection as conditional binary matching so a transformer detector can localize objects specified by either a class name or an exemplar image.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "DETR's usual class-based matching cost is unavailable for unseen categories without labeled training images. OV-DETR instead matches objects to input queries and conditions the decoder on CLIP text or image embeddings to transfer to novel queries.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Introduces binary conditional matching for end-to-end open-vocabulary transformer detection.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Supports both natural-language category queries and visual exemplars within one detector.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On OV-LVIS, OV-DETR improves novel-class mask AP from ViLD’s 16.1 to 17.4 and overall mask AP from 22.5 to 26.6. On OV-COCO, novel-class box AP50 rises from 27.6 to 29.4; the two benchmarks use different AP definitions.",
            "fragment": "S4.T3",
            "locator": "Table 3 · open-vocabulary detection · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "On OV-LVIS, adding proposals without conditional matching yields 6.3 novel-class mask AP. Adding conditional binary matching with those proposals raises it to 17.4, versus 9.5 in the baseline without either component.",
            "fragment": "S4.T2.fig1",
            "locator": "Table 2 · conditional-matching ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Class-based DETR matching",
              "design": "Cannot calculate novel-class classification costs without labeled examples of those classes."
            },
            {
              "method": "OV-DETR",
              "design": "Uses binary conditional matching with CLIP embeddings to support text and exemplar-image queries."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2102.12867",
      "title": "FASA: Feature Augmentation and Sampling Adaptation for Long-Tailed Instance Segmentation",
      "authors": [
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Chen Huang"
        },
        {
          "name": "Chen Change Loy"
        }
      ],
      "publication": {
        "year": 2021,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2021"
      },
      "dates": {
        "arxivFirstPosted": "2021-02-25",
        "arxivLastUpdated": "2021-09-30"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2102.12867",
        "googleScholar": "hW23VKIAAAAJ:qjMakFHDy7sC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2102.12867"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:qjMakFHDy7sC"
        },
        {
          "type": "code",
          "url": "https://github.com/yuhangzang/FASA",
          "repository": "yuhangzang/FASA"
        },
        {
          "type": "project",
          "url": "https://www.mmlab-ntu.com/project/fasa/index.html"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "homepageNew": false,
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1457,
        "order": 62
      },
      "shortName": "FASA",
      "keywords": [
        "FASA",
        "Long-tailed instance segmentation",
        "Feature augmentation",
        "Adaptive sampling",
        "Rare-class recognition",
        "Virtual features",
        "Class imbalance",
        "Long-tailed classification"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv210212867",
        "year": 2021,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2021 proceedings record",
          "url": "https://openaccess.thecvf.com/content/ICCV2021/html/Zang_FASA_Feature_Augmentation_and_Sampling_Adaptation_for_Long-Tailed_Instance_Segmentation_ICCV_2021_paper.html"
        },
        "status": "published",
        "month": 10,
        "firstPage": 3457,
        "lastPage": 3466,
        "pdfURL": "https://openaccess.thecvf.com/content/ICCV2021/papers/Zang_FASA_Feature_Augmentation_and_Sampling_Adaptation_for_Long-Tailed_Instance_Segmentation_ICCV_2021_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/2102.12867v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/2102.12867v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/2102.12867v2"
          }
        },
        "abstract": {
          "text": "Recent methods for long-tailed instance segmentation still struggle on rare object classes with few training data. We propose a simple yet effective method, Feature Augmentation and Sampling Adaptation (FASA), that addresses the data scarcity issue by augmenting the feature space especially for rare classes. Both the Feature Augmentation (FA) and feature sampling components are adaptive to the actual training status -- FA is informed by the feature mean and variance of observed real samples from past iterations, and we sample the generated virtual features in a loss-adapted manner to avoid over-fitting. FASA does not require any elaborate loss design, and removes the need for inter-class transfer learning that often involves large cost and manually-defined head/tail class groups. We show FASA is a fast, generic method that can be easily plugged into standard or long-tailed segmentation frameworks, with consistent performance gains and little added cost. FASA is also applicable to other tasks like long-tailed classification with state-of-the-art performance.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "FASA addresses rare-class data scarcity by generating virtual features from observed class statistics and adapting their sampling to the model's training losses.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Long-tailed instance segmentation lacks enough examples for rare categories. FASA augments feature space using running means and variances, then adjusts virtual-feature sampling to avoid overfitting without manual head/tail partitions or an elaborate new loss.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Combines adaptive feature augmentation with loss-adapted sampling of virtual examples.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Provides a generic module for standard and long-tailed segmentation frameworks and extends it to classification.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "FASA raises Mask R-CNN mask AP from 20.8 to 23.7 and rare-class AP from 8.0 to 17.8 on LVIS v1.0 validation. InstaBoost reaches 21.4 overall AP and 10.3 rare-class AP in the same comparison.",
            "fragment": "S4.T2",
            "locator": "Table 2 · LVIS v1.0 augmentation comparison · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "Feature augmentation alone gives 22.3 overall AP and 12.7 rare-class AP; adding adaptive feature sampling raises these to 23.7 and 17.8. The additional gain is concentrated in rare classes.",
            "fragment": "S4.T1",
            "locator": "Table 1 · adaptive-sampling ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Inter-class transfer for rare categories",
              "design": "Can require costly transfer procedures and manually defined head/tail class groups."
            },
            {
              "method": "FASA",
              "design": "Generates features from observed class statistics and samples them adaptively using training losses."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2008.10032",
      "title": "Seesaw Loss for Long-Tailed Instance Segmentation",
      "authors": [
        {
          "name": "Jiaqi Wang"
        },
        {
          "name": "Wenwei Zhang"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Yuhang Cao"
        },
        {
          "name": "Jiangmiao Pang"
        },
        {
          "name": "Tao Gong"
        },
        {
          "name": "Kai Chen"
        },
        {
          "name": "Ziwei Liu"
        },
        {
          "name": "Chen Change Loy"
        },
        {
          "name": "Dahua Lin"
        }
      ],
      "publication": {
        "year": 2021,
        "venueGroup": "CVPR",
        "citationText": "IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 2021"
      },
      "dates": {
        "arxivFirstPosted": "2020-08-23",
        "arxivLastUpdated": "2021-06-17"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2008.10032",
        "googleScholar": "hW23VKIAAAAJ:2osOgNQ5qMEC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2008.10032"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:2osOgNQ5qMEC"
        },
        {
          "type": "code",
          "url": "https://github.com/open-mmlab/mmdetection/tree/master/configs/seesaw_loss",
          "repository": "open-mmlab/mmdetection"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1477,
        "order": 63
      },
      "shortName": "Seesaw Loss",
      "keywords": [
        "Seesaw Loss",
        "Long-tailed instance segmentation",
        "Gradient rebalancing",
        "Class imbalance",
        "Rare-category detection",
        "Mitigation factor",
        "Compensation factor",
        "Loss function design"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv200810032",
        "year": 2021,
        "booktitle": "Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)",
        "source": {
          "label": "CVPR 2021 proceedings record",
          "url": "https://openaccess.thecvf.com/content/CVPR2021/html/Wang_Seesaw_Loss_for_Long-Tailed_Instance_Segmentation_CVPR_2021_paper.html"
        },
        "status": "published",
        "month": 6,
        "firstPage": 9695,
        "lastPage": 9704,
        "pdfURL": "https://openaccess.thecvf.com/content/CVPR2021/papers/Wang_Seesaw_Loss_for_Long-Tailed_Instance_Segmentation_CVPR_2021_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v4",
            "url": "https://arxiv.org/abs/2008.10032v4"
          },
          "paper": {
            "label": "Paper · arXiv v4",
            "url": "https://arxiv.org/pdf/2008.10032v4",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v4",
            "url": "https://arxiv.org/html/2008.10032v4"
          }
        },
        "abstract": {
          "text": "Instance segmentation has witnessed a remarkable progress on class-balanced benchmarks. However, they fail to perform as accurately in real-world scenarios, where the category distribution of objects naturally comes with a long tail. Instances of head classes dominate a long-tailed dataset and they serve as negative samples of tail categories. The overwhelming gradients of negative samples on tail classes lead to a biased learning process for classifiers. Consequently, objects of tail categories are more likely to be misclassified as backgrounds or head categories. To tackle this problem, we propose Seesaw Loss to dynamically re-balance gradients of positive and negative samples for each category, with two complementary factors, i.e., mitigation factor and compensation factor. The mitigation factor reduces punishments to tail categories w.r.t. the ratio of cumulative training instances between different categories. Meanwhile, the compensation factor increases the penalty of misclassified instances to avoid false positives of tail categories. We conduct extensive experiments on Seesaw Loss with mainstream frameworks and different data sampling strategies. With a simple end-to-end training pipeline, Seesaw Loss obtains significant gains over Cross-Entropy Loss, and achieves state-of-the-art performance on LVIS dataset without bells and whistles. Code is available at https://github.com/open-mmlab/mmdetection.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "takeaway": {
          "text": "Seesaw Loss balances rare-class learning by reducing overwhelming negative gradients while restoring penalties for misclassified examples to control false positives.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "summary": {
          "text": "Frequent categories dominate negative training signals and bias long-tailed classifiers against rare objects. Seesaw Loss dynamically adjusts positive-negative balance with a mitigation factor based on cumulative class counts and a complementary compensation factor.",
          "source": "abstract",
          "locator": "arXiv abstract · v4"
        },
        "contributions": [
          {
            "text": "Introduces complementary mitigation and compensation terms for class-dependent gradient rebalancing.",
            "source": "abstract",
            "locator": "arXiv abstract · v4"
          },
          {
            "text": "Integrates with end-to-end instance segmentation training across mainstream frameworks and sampling strategies.",
            "source": "abstract",
            "locator": "arXiv abstract · v4"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "On LVIS-v1 validation with a 2× schedule and repeat-factor sampling, Seesaw Loss raises mask AP from cross-entropy’s 25.5 to 27.6 and rare-class AP from 16.6 to 20.6. Adding normalized mask prediction reaches 28.1 overall mask AP.",
            "fragment": "S3.T1",
            "locator": "Table 1 · Mask R-CNN R101-FPN with RFS · arXiv v4",
            "source": "fullText"
          },
          {
            "text": "With Mask R-CNN and RFS, mitigation alone gives 25.1 mask AP, compensation alone 24.1, and both 25.7, versus 23.7 for the baseline. Adding normalized linear activation reaches 26.4 AP and 19.6 rare-class AP.",
            "fragment": "S4.T2",
            "locator": "Table 2 · R50-FPN loss-component ablation · arXiv v4",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v4",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Cross-entropy on long-tailed instances",
              "design": "Frequent-category negatives can overwhelm the learning signal for rare classes."
            },
            {
              "method": "Seesaw Loss",
              "design": "Reduces excessive rare-class penalties while compensating for misclassified examples to limit false positives."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:2003.07543",
      "title": "KPNet: Towards Minimal Face Detector",
      "authors": [
        {
          "name": "Guanglu Song"
        },
        {
          "name": "Yu Liu"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Xiaogang Wang"
        },
        {
          "name": "Biao Leng"
        },
        {
          "name": "Qingsheng Yuan"
        }
      ],
      "publication": {
        "year": 2020,
        "venueGroup": "AAAI",
        "citationText": "AAAI Conference on Artificial Intelligence (AAAI), 2020"
      },
      "dates": {
        "arxivFirstPosted": "2020-03-17",
        "arxivLastUpdated": "2020-03-17"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "2003.07543",
        "googleScholar": "hW23VKIAAAAJ:d1gkVwhDpl0C",
        "doi": "10.1609/aaai.v34i07.6878"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/2003.07543"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:d1gkVwhDpl0C"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1496,
        "order": 64
      },
      "shortName": "KPNet",
      "keywords": [
        "KPNet",
        "Lightweight face detection",
        "Facial landmark detection",
        "Bottom-up detection",
        "Scale-adaptive soft-argmax",
        "Compact neural networks",
        "Face alignment",
        "Real-time vision"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv200307543",
        "year": 2020,
        "journal": "Proceedings of the AAAI Conference on Artificial Intelligence",
        "publisher": "Association for the Advancement of Artificial Intelligence (AAAI)",
        "source": {
          "label": "AAAI publisher record",
          "url": "https://doi.org/10.1609/aaai.v34i07.6878"
        },
        "status": "published",
        "month": 4,
        "volume": "34",
        "number": "07",
        "firstPage": 12015,
        "lastPage": 12022,
        "pdfURL": "https://ojs.aaai.org/index.php/AAAI/article/download/6878/6732",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/2003.07543v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/2003.07543v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/2003.07543v1"
          }
        },
        "abstract": {
          "text": "The small receptive field and capacity of minimal neural networks limit their performance when using them to be the backbone of detectors. In this work, we find that the appearance feature of a generic face is discriminative enough for a tiny and shallow neural network to verify from the background. And the essential barriers behind us are 1) the vague definition of the face bounding box and 2) tricky design of anchor-boxes or receptive field. Unlike most top-down methods for joint face detection and alignment, the proposed KPNet detects small facial keypoints instead of the whole face by in a bottom-up manner. It first predicts the facial landmarks from a low-resolution image via the well-designed fine-grained scale approximation and scale adaptive soft-argmax operator. Finally, the precise face bounding boxes, no matter how we define it, can be inferred from the keypoints. Without any complex head architecture or meticulous network designing, the KPNet achieves state-of-the-art accuracy on generic face detection and alignment benchmarks with only ≈1M parameters, which runs at 1000fps on GPU and is easy to perform real-time on most modern front-end chips.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "KPNet builds a compact face detector by predicting facial keypoints first and deriving bounding boxes from them, avoiding complex anchor and detection-head designs.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Tiny detector backbones have limited receptive fields and capacity. KPNet exploits discriminative facial landmarks with fine-grained scale approximation and a scale-adaptive soft-argmax, then reconstructs face boxes in a bottom-up manner.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Unifies face localization and alignment through keypoint-first prediction.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Uses scale-aware landmark decoding to support a small, shallow network.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "At 50 false positives, scale-adaptive soft-argmax gives 69.32% recall on FDDB rotated −90° and 69.97% at +90°, versus 50.65% and 49.9% for argmax decoding. On unrotated FDDB the corresponding recalls are 91.6% and 88.98%.",
            "fragment": "Sx4.T2",
            "locator": "Table 2 · rotation robustness, DRNet backbone · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "For DRNet, adding scale-adaptive soft-argmax raises FDDB recall at one false positive from 47.3% to 81.1%; recall at 50 false positives changes from 90.7% to 91.6%. The stricter operating point highlights the reduction in localization errors.",
            "fragment": "Sx4.T3",
            "locator": "Table 3 · localization-error ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Top-down face detection and alignment",
              "design": "Detects face boxes with designs sensitive to box definitions, anchors and receptive fields."
            },
            {
              "method": "KPNet",
              "design": "Predicts landmarks first with scale-aware decoding and derives bounding boxes from those keypoints."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:1908.05900",
      "title": "Efficient and Accurate Arbitrary-Shaped Text Detection with Pixel Aggregation Network",
      "authors": [
        {
          "name": "Wenhai Wang"
        },
        {
          "name": "Enze Xie"
        },
        {
          "name": "Xiaoge Song"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Wenjia Wang"
        },
        {
          "name": "Tong Lu"
        },
        {
          "name": "Gang Yu"
        },
        {
          "name": "Chunhua Shen"
        }
      ],
      "publication": {
        "year": 2019,
        "venueGroup": "ICCV",
        "citationText": "IEEE International Conference on Computer Vision (ICCV), 2019"
      },
      "dates": {
        "arxivFirstPosted": "2019-08-16",
        "arxivLastUpdated": "2020-08-02"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "1908.05900",
        "googleScholar": "hW23VKIAAAAJ:u-x6o8ySG0sC"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/1908.05900"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:u-x6o8ySG0sC"
        },
        {
          "type": "code",
          "url": "https://github.com/open-mmlab/mmocr/tree/main/configs/textdet/panet",
          "repository": "open-mmlab/mmocr"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1511,
        "order": 65
      },
      "shortName": "PAN",
      "keywords": [
        "Pixel Aggregation Network",
        "PAN",
        "Arbitrary-shaped text detection",
        "Scene text detection",
        "Pixel aggregation",
        "Feature pyramid enhancement",
        "Real-time OCR",
        "Text instance segmentation"
      ],
      "citation": {
        "type": "inproceedings",
        "key": "arxiv190805900",
        "year": 2019,
        "booktitle": "Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)",
        "source": {
          "label": "ICCV 2019 proceedings record",
          "url": "https://openaccess.thecvf.com/content_ICCV_2019/html/Wang_Efficient_and_Accurate_Arbitrary-Shaped_Text_Detection_With_Pixel_Aggregation_Network_ICCV_2019_paper.html"
        },
        "status": "published",
        "title": "Efficient and Accurate Arbitrary-Shaped Text Detection With Pixel Aggregation Network",
        "month": 10,
        "firstPage": 8440,
        "lastPage": 8449,
        "pdfURL": "https://openaccess.thecvf.com/content_ICCV_2019/papers/Wang_Efficient_and_Accurate_Arbitrary-Shaped_Text_Detection_With_Pixel_Aggregation_Network_ICCV_2019_paper.pdf",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v2",
            "url": "https://arxiv.org/abs/1908.05900v2"
          },
          "paper": {
            "label": "Paper · arXiv v2",
            "url": "https://arxiv.org/pdf/1908.05900v2",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v2",
            "url": "https://arxiv.org/html/1908.05900v2"
          }
        },
        "abstract": {
          "text": "Scene text detection, an important step of scene text reading systems, has witnessed rapid development with convolutional neural networks. Nonetheless, two main challenges still exist and hamper its deployment to real-world applications. The first problem is the trade-off between speed and accuracy. The second one is to model the arbitrary-shaped text instance. Recently, some methods have been proposed to tackle arbitrary-shaped text detection, but they rarely take the speed of the entire pipeline into consideration, which may fall short in practical applications. In this paper, we propose an efficient and accurate arbitrary-shaped text detector, termed Pixel Aggregation Network (PAN), which is equipped with a low computational-cost segmentation head and a learnable post-processing. More specifically, the segmentation head is made up of Feature Pyramid Enhancement Module (FPEM) and Feature Fusion Module (FFM). FPEM is a cascadable U-shaped module, which can introduce multi-level information to guide the better segmentation. FFM can gather the features given by the FPEMs of different depths into a final feature for segmentation. The learnable post-processing is implemented by Pixel Aggregation (PA), which can precisely aggregate text pixels by predicted similarity vectors. Experiments on several standard benchmarks validate the superiority of the proposed PAN. It is worth noting that our method can achieve a competitive F-measure of 79.9% at 84.2 FPS on CTW1500.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "takeaway": {
          "text": "Pixel Aggregation Network detects arbitrary-shaped scene text efficiently by combining a lightweight segmentation head with learned pixel grouping.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "summary": {
          "text": "Scene text systems must recover irregular text shapes while keeping the full pipeline fast. PAN strengthens and fuses pyramid features, then groups text pixels through predicted similarity vectors rather than relying only on a costly segmentation backbone.",
          "source": "abstract",
          "locator": "arXiv abstract · v2"
        },
        "contributions": [
          {
            "text": "Introduces Feature Pyramid Enhancement and Feature Fusion modules for efficient text segmentation.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          },
          {
            "text": "Uses learnable Pixel Aggregation to assemble text instances of arbitrary shape.",
            "source": "abstract",
            "locator": "arXiv abstract · v2"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "With external training data, PAN-320 achieves 79.9% F-measure at 84.2 fps, while PAN-640 reaches 83.7% at 39.8 fps. Without external data, these F-measures are 77.1% and 81.0%, respectively.",
            "fragment": "S4.T4",
            "locator": "Table 4 · CTW1500, single-scale detection · arXiv v2",
            "source": "fullText"
          },
          {
            "text": "With ResNet18 and feature fusion fixed, enabling pixel aggregation raises CTW1500 F-measure from 79.8% to 81.0%, with throughput changing from 39.9 to 39.8 fps. On ICDAR2015, F-measure rises from 79.3% to 80.3%.",
            "fragment": "S4.T3",
            "locator": "Table 3 · pixel-aggregation ablation · arXiv v2",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v2",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Arbitrary-shape text detection without pipeline speed emphasis",
              "design": "Can recover text contours while leaving the end-to-end speed-accuracy trade-off unresolved."
            },
            {
              "method": "PAN",
              "design": "Combines efficient pyramid feature processing with learned pixel aggregation to group arbitrary-shaped text."
            }
          ]
        }
      }
    },
    {
      "id": "arxiv:1811.08605",
      "title": "Scene Text Detection with Supervised Pyramid Context Network",
      "authors": [
        {
          "name": "Enze Xie"
        },
        {
          "name": "Yuhang Zang"
        },
        {
          "name": "Shuai Shao"
        },
        {
          "name": "Gang Yu"
        },
        {
          "name": "Cong Yao"
        },
        {
          "name": "Guangyao Li"
        }
      ],
      "publication": {
        "year": 2019,
        "venueGroup": "AAAI",
        "citationText": "AAAI Conference on Artificial Intelligence (AAAI), 2019"
      },
      "dates": {
        "arxivFirstPosted": "2018-11-21",
        "arxivLastUpdated": "2018-11-21"
      },
      "topics": [
        "Image Understanding"
      ],
      "identifiers": {
        "arxiv": "1811.08605",
        "googleScholar": "hW23VKIAAAAJ:u5HHmVD_uO8C",
        "doi": "10.1609/aaai.v33i01.33019038"
      },
      "links": [
        {
          "type": "arxiv",
          "url": "https://arxiv.org/abs/1811.08605"
        },
        {
          "type": "scholar",
          "url": "https://scholar.google.com/citations?view_op=view_citation&hl=en&user=hW23VKIAAAAJ&citation_for_view=hW23VKIAAAAJ:u5HHmVD_uO8C"
        }
      ],
      "display": {
        "new": false,
        "badges": [],
        "infoWrapper": false
      },
      "source": {
        "file": "research.html",
        "line": 1530,
        "order": 66
      },
      "shortName": "SPCNET",
      "keywords": [
        "SPCNET",
        "Scene text detection",
        "Supervised pyramid context",
        "False-positive suppression",
        "Feature Pyramid Network",
        "Instance segmentation",
        "Semantic guidance",
        "OCR"
      ],
      "citation": {
        "type": "article",
        "key": "arxiv181108605",
        "year": 2019,
        "journal": "Proceedings of the AAAI Conference on Artificial Intelligence",
        "publisher": "Association for the Advancement of Artificial Intelligence (AAAI)",
        "source": {
          "label": "AAAI publisher record",
          "url": "https://doi.org/10.1609/aaai.v33i01.33019038"
        },
        "status": "published",
        "month": 7,
        "volume": "33",
        "number": "01",
        "firstPage": 9038,
        "lastPage": 9045,
        "pdfURL": "https://ojs.aaai.org/index.php/AAAI/article/download/4935/4808",
        "verifiedOn": "2026-09-12"
      },
      "content": {
        "verifiedOn": "2026-09-12",
        "sources": {
          "abstract": {
            "label": "arXiv abstract · v1",
            "url": "https://arxiv.org/abs/1811.08605v1"
          },
          "paper": {
            "label": "Paper · arXiv v1",
            "url": "https://arxiv.org/pdf/1811.08605v1",
            "encodingFormat": "application/pdf"
          },
          "fullText": {
            "label": "Full paper · arXiv v1",
            "url": "https://arxiv.org/html/1811.08605v1"
          }
        },
        "abstract": {
          "text": "Scene text detection methods based on deep learning have achieved remarkable results over the past years. However, due to the high diversity and complexity of natural scenes, previous state-of-the-art text detection methods may still produce a considerable amount of false positives, when applied to images captured in real-world environments. To tackle this issue, mainly inspired by Mask R-CNN, we propose in this paper an effective model for scene text detection, which is based on Feature Pyramid Network (FPN) and instance segmentation. We propose a supervised pyramid context network (SPCNET) to precisely locate text regions while suppressing false positives. Benefited from the guidance of semantic information and sharing FPN, SPCNET obtains significantly enhanced performance while introducing marginal extra computation. Experiments on standard datasets demonstrate that our SPCNET clearly outperforms start-of-the-art methods. Specifically, it achieves an F-measure of 92.1% on ICDAR2013, 87.2% on ICDAR2015, 74.1% on ICDAR2017 MLT and 82.9% on Total-Text.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "takeaway": {
          "text": "SPCNET uses supervised semantic context within a feature-pyramid instance-segmentation model to localize scene text and suppress false positives with little extra computation.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "summary": {
          "text": "Complex natural scenes can produce false text detections despite strong benchmark performance. SPCNET introduces semantic supervision over a shared feature pyramid to distinguish true text regions from visually similar background patterns.",
          "source": "abstract",
          "locator": "arXiv abstract · v1"
        },
        "contributions": [
          {
            "text": "Develops a supervised pyramid context network inspired by instance segmentation.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          },
          {
            "text": "Shares feature-pyramid computation while using semantic guidance to reduce false positives.",
            "source": "abstract",
            "locator": "arXiv abstract · v1"
          }
        ],
        "results": [],
        "resultNotes": [
          {
            "text": "SPCNET raises ICDAR2015 F-measure from the baseline’s 85.5% to 87.2%, with recall increasing from 83.8% to 85.8% and precision from 87.4% to 88.7%.",
            "fragment": "Sx4.T3",
            "locator": "Table 3 · ICDAR2015 text detection · arXiv v1",
            "source": "fullText"
          },
          {
            "text": "In the ICDAR2017-MLT component ablation, text-context modeling raises F-measure from 74.7% to 76.8%, and re-scoring raises it to 78.5%. Recall remains 73.4% while precision increases from 76.2% to 84.2%.",
            "fragment": "Sx4.T1",
            "locator": "Table 1 · context and re-scoring ablation · arXiv v1",
            "source": "fullText"
          }
        ],
        "methodComparison": {
          "source": "abstract",
          "locator": "arXiv abstract · v1",
          "columns": [
            {
              "key": "method",
              "label": "Approach"
            },
            {
              "key": "design",
              "label": "Key difference"
            }
          ],
          "rows": [
            {
              "method": "Scene text detection in complex backgrounds",
              "design": "Can retain false positives caused by diverse natural-scene content."
            },
            {
              "method": "SPCNET",
              "design": "Adds supervised semantic context over a shared feature pyramid to localize text and suppress false detections."
            }
          ]
        }
      }
    }
  ]
}
