<?xml version="1.0" encoding="utf-8" standalone="yes" ?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>跨模态检索 | ViLab</title>
    <link>https://vilab.team/tag/%E8%B7%A8%E6%A8%A1%E6%80%81%E6%A3%80%E7%B4%A2/</link>
      <atom:link href="https://vilab.team/tag/%E8%B7%A8%E6%A8%A1%E6%80%81%E6%A3%80%E7%B4%A2/index.xml" rel="self" type="application/rss+xml" />
    <description>跨模态检索</description>
    <generator>Hugo Blox Builder (https://hugoblox.com)</generator><language>en-us</language><lastBuildDate>Wed, 29 May 2024 00:00:00 +0000</lastBuildDate>
    <image>
      <url>https://vilab.team/media/icon_hu2896232876136423579.png</url>
      <title>跨模态检索</title>
      <link>https://vilab.team/tag/%E8%B7%A8%E6%A8%A1%E6%80%81%E6%A3%80%E7%B4%A2/</link>
    </image>
    
    <item>
      <title>Multi-modal generative embedding model</title>
      <link>https://vilab.team/publication/multi-modal-generative-embedding-model/</link>
      <pubDate>Wed, 29 May 2024 00:00:00 +0000</pubDate>
      <guid>https://vilab.team/publication/multi-modal-generative-embedding-model/</guid>
      <description>&lt;p&gt;本文提出多模态生成嵌入模型MM-GEM，将生成与嵌入两种目标统一于单个大语言模型中，实现每个模态仅需一个模型。通过引入PoolAggregator提升效率并支持细粒度嵌入与生成。实验表明，生成与嵌入目标并不显著冲突，模型在跨模态检索、零样本分类和图像描述等任务上表现优异，同时具备区域级描述生成与检索能力，并在长文本图像检索中取得显著提升。&lt;/p&gt;
</description>
    </item>
    
  </channel>
</rss>
