<?xml version="1.0" encoding="UTF-8"?>
<ArticleSet>
  <Article>
    <Journal>
      <PublisherName></PublisherName>
      <JournalTitle>Journal of Artificial Intelligence, Applications and Innovations</JournalTitle>
      <Issn>3060-7124</Issn>
      <Volume>1</Volume>
      <Issue>Journal of Artificial Intelligence, Application and Inovations </Issue>
      <PubDate PubStatus="epublish">
        <Year>2024</Year>
        <Month>01</Month>
        <Day>01</Day>
      </PubDate>
    </Journal>
    <ArticleTitle>Revolutionizing Short Video Recommendations Using the  Vision Mamba Framework</ArticleTitle>
    <VernacularTitle>Revolutionizing Short Video Recommendations Using the  Vision Mamba Framework</VernacularTitle>
    <FirstPage>1</FirstPage>
    <LastPage>11</LastPage>
    <ELocationID EIdType="doi">10.61838/jaiai.1.1.1</ELocationID>
    <Language>EN</Language>
    <AuthorList>
      <Author>
        <FirstName></FirstName>
        <LastName></LastName>
        <Affiliation></Affiliation>
      </Author>
      <Author>
        <FirstName></FirstName>
        <LastName></LastName>
        <Affiliation></Affiliation>
      </Author>
      <Author>
        <FirstName></FirstName>
        <LastName></LastName>
        <Affiliation></Affiliation>
      </Author>
    </AuthorList>
    <PublicationType>Journal Article</PublicationType>
    <History>
      <PubDate PubStatus="received">
        <Year>2023</Year>
        <Month>08</Month>
        <Day>20</Day>
      </PubDate>
    </History>
    <Abstract>&lt;p&gt;The rapid proliferation of short-form video content on platforms such as TikTok, Instagram, and YouTube Shorts has introduced significant challenges for recommendation systems, as traditional methods often struggle to keep up with the dynamic nature of user engagement and the large influx of data. In this paper, we present the Vision Mamba (Vim) framework, a cutting-edge approach in visual representation learning that employs bidirectional state space models to improve both the efficiency and accuracy of short video recommendations. The Vim framework excels by effectively capturing temporal dynamics, long-range dependencies, and the contextual relevance within video sequences, addressing computational limitations in a resource-efficient manner. Furthermore, it supports real-time personalization and scalable deployment across modern content platforms. Experimental evaluations conducted on the MicroLens dataset demonstrate that the Vision Mamba framework significantly outperforms existing traditional models, setting a new benchmark in video recommendation systems and offering enhanced user experiences with more contextually relevant and personalized content delivery.&lt;/p&gt;</Abstract>
    <ObjectList>
      <Object Type="keyword">
        <Param Name="value">short video recommendation</Param>
      </Object>
      <Object Type="keyword">
        <Param Name="value">state space models</Param>
      </Object>
      <Object Type="keyword">
        <Param Name="value">visual representation learning</Param>
      </Object>
      <Object Type="keyword">
        <Param Name="value">personalized recommendations</Param>
      </Object>
      <Object Type="keyword">
        <Param Name="value">Vision Mamba</Param>
      </Object>
    </ObjectList>
    <ArchiveCopySource DocType="pdf">https://www.journalaiai.com/index.php/aiai/article/download/1/1</ArchiveCopySource>
  </Article>
</ArticleSet>
