% Generated by scripts/update_bibliography.py.
% Custom fields: code_status, github, github_stars, github_stars_as_of, github_note.
% Star counts are snapshots; run the updater to refresh them.

@article{doretto2003dynamic,
  title = {{Dynamic Textures}},
  author = {Gianfranco Doretto and Alessandro Chiuso and Ying Nian Wu and Stefano Soatto},
  year = {2003},
  journal = {International Journal of Computer Vision},
  volume = {51},
  number = {2},
  pages = {91--109},
  publisher = {Springer Science and Business Media LLC},
  doi = {10.1023/A:1021669406132},
  url = {https://doi.org/10.1023/A:1021669406132},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@article{horn1981determining,
  title = {{Determining optical flow}},
  author = {Berthold K.P. Horn and Brian G. Schunck},
  year = {1981},
  journal = {Artificial Intelligence},
  volume = {17},
  number = {1-3},
  pages = {185--203},
  publisher = {Elsevier BV},
  doi = {10.1016/0004-3702(81)90024-2},
  url = {https://doi.org/10.1016/0004-3702(81)90024-2},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@inproceedings{schodl2000video,
  title = {{Video textures}},
  author = {Arno Schödl and Richard Szeliski and David H. Salesin and Irfan Essa},
  year = {2000},
  booktitle = {Proceedings of the 27th annual conference on Computer graphics and interactive techniques - SIGGRAPH '00},
  pages = {489--498},
  publisher = {ACM Press},
  doi = {10.1145/344779.345012},
  url = {https://doi.org/10.1145/344779.345012},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@inproceedings{lucas1981iterative,
  title = {{An Iterative Image Registration Technique with an Application to Stereo Vision}},
  author = {Bruce D. Lucas and Takeo Kanade},
  year = {1981},
  booktitle = {Proceedings of the 7th International Joint Conference on Artificial Intelligence (IJCAI)},
  pages = {674--679},
  url = {https://publications.ri.cmu.edu/storage/publications/pub_files/pub3/lucas_bruce_d_1981_1/lucas_bruce_d_1981_1.pdf},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{finn2016unsupervised,
  title = {{Unsupervised Learning for Physical Interaction through Video Prediction}},
  author = {Chelsea Finn and Ian Goodfellow and Sergey Levine},
  year = {2016},
  eprint = {1605.07157},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1605.07157},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{lotter2016deep,
  title = {{Deep Predictive Coding Networks for Video Prediction and Unsupervised Learning}},
  author = {William Lotter and Gabriel Kreiman and David Cox},
  year = {2016},
  eprint = {1605.08104},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1605.08104},
  code_status = {official-code},
  github = {https://github.com/coxlab/prednet},
  github_stars = {803},
  github_stars_as_of = {2026-08-30},
  github_note = {作者维护的 Keras 实现}
}

@misc{mathieu2015deep,
  title = {{Deep multi-scale video prediction beyond mean square error}},
  author = {Michael Mathieu and Camille Couprie and Yann LeCun},
  year = {2015},
  eprint = {1511.05440},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1511.05440},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{shi2015convolutional,
  title = {{Convolutional LSTM Network: A Machine Learning Approach for Precipitation Nowcasting}},
  author = {Xingjian Shi and Zhourong Chen and Hao Wang and Dit-Yan Yeung and Wai-kin Wong and Wang-chun Woo},
  year = {2015},
  eprint = {1506.04214},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1506.04214},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{srivastava2015unsupervised,
  title = {{Unsupervised Learning of Video Representations using LSTMs}},
  author = {Nitish Srivastava and Elman Mansimov and Ruslan Salakhutdinov},
  year = {2015},
  eprint = {1502.04681},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1502.04681},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{clark2019adversarial,
  title = {{Adversarial Video Generation on Complex Datasets}},
  author = {Aidan Clark and Jeff Donahue and Karen Simonyan},
  year = {2019},
  eprint = {1907.06571},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1907.06571},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{tulyakov2017mocogan,
  title = {{MoCoGAN: Decomposing Motion and Content for Video Generation}},
  author = {Sergey Tulyakov and Ming-Yu Liu and Xiaodong Yang and Jan Kautz},
  year = {2017},
  eprint = {1707.04993},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1707.04993},
  code_status = {official-code},
  github = {https://github.com/sergeytulyakov/mocogan},
  github_stars = {603},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{vondrick2016generating,
  title = {{Generating Videos with Scene Dynamics}},
  author = {Carl Vondrick and Hamed Pirsiavash and Antonio Torralba},
  year = {2016},
  eprint = {1609.02612},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1609.02612},
  code_status = {official-code},
  github = {https://github.com/cvondrick/videogan},
  github_stars = {706},
  github_stars_as_of = {2026-08-30},
  github_note = {论文项目页链接的作者实现}
}

@misc{oord2017neural,
  title = {{Neural Discrete Representation Learning}},
  author = {Aaron van den Oord and Oriol Vinyals and Koray Kavukcuoglu},
  year = {2017},
  eprint = {1711.00937},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1711.00937},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{villegas2022phenaki,
  title = {{Phenaki: Variable Length Video Generation From Open Domain Textual Description}},
  author = {Ruben Villegas and Mohammad Babaeizadeh and Pieter-Jan Kindermans and Hernan Moraldo and Han Zhang and Mohammad Taghi Saffar and Santiago Castro and Julius Kunze and Dumitru Erhan},
  year = {2022},
  eprint = {2210.02399},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2210.02399},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{yan2021videogpt,
  title = {{VideoGPT: Video Generation using VQ-VAE and Transformers}},
  author = {Wilson Yan and Yunzhi Zhang and Pieter Abbeel and Aravind Srinivas},
  year = {2021},
  eprint = {2104.10157},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2104.10157},
  code_status = {official-code},
  github = {https://github.com/wilson1yan/VideoGPT},
  github_stars = {1081},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{yu2022magvit,
  title = {{MAGVIT: Masked Generative Video Transformer}},
  author = {Lijun Yu and Yong Cheng and Kihyuk Sohn and José Lezama and Han Zhang and Huiwen Chang and Alexander G. Hauptmann and Ming-Hsuan Yang and Yuan Hao and Irfan Essa and Lu Jiang},
  year = {2022},
  eprint = {2212.05199},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2212.05199},
  code_status = {official-code},
  github = {https://github.com/google-research/magvit},
  github_stars = {1001},
  github_stars_as_of = {2026-08-30},
  github_note = {Google Research 官方实现；仓库已归档}
}

@misc{yu2023language,
  title = {{Language Model Beats Diffusion -- Tokenizer is Key to Visual Generation}},
  author = {Lijun Yu and José Lezama and Nitesh B. Gundavarapu and Luca Versari and Kihyuk Sohn and David Minnen and Yong Cheng and Vighnesh Birodkar and Agrim Gupta and Xiuye Gu and Alexander G. Hauptmann and Boqing Gong and Ming-Hsuan Yang and Irfan Essa and David A. Ross and Lu Jiang},
  year = {2023},
  eprint = {2310.05737},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2310.05737},
  code_status = {official-related-code},
  github = {https://github.com/google-research/magvit},
  github_stars = {1001},
  github_stars_as_of = {2026-08-30},
  github_note = {作者团队的 MAGVIT v1 仓库；README 未声明包含 MAGVIT-v2 实现；仓库已归档}
}

@misc{bartal2024lumiere,
  title = {{Lumiere: A Space-Time Diffusion Model for Video Generation}},
  author = {Omer Bar-Tal and Hila Chefer and Omer Tov and Charles Herrmann and Roni Paiss and Shiran Zada and Ariel Ephrat and Junhwa Hur and Guanghui Liu and Amit Raj and Yuanzhen Li and Michael Rubinstein and Tomer Michaeli and Oliver Wang and Deqing Sun and Tali Dekel and Inbar Mosseri},
  year = {2024},
  eprint = {2401.12945},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2401.12945},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{blattmann2023align,
  title = {{Align your Latents: High-Resolution Video Synthesis with Latent Diffusion Models}},
  author = {Andreas Blattmann and Robin Rombach and Huan Ling and Tim Dockhorn and Seung Wook Kim and Sanja Fidler and Karsten Kreis},
  year = {2023},
  eprint = {2304.08818},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2304.08818},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{blattmann2023stable,
  title = {{Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets}},
  author = {Andreas Blattmann and Tim Dockhorn and Sumith Kulal and Daniel Mendelevitch and Maciej Kilian and Dominik Lorenz and Yam Levi and Zion English and Vikram Voleti and Adam Letts and Varun Jampani and Robin Rombach},
  year = {2023},
  eprint = {2311.15127},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2311.15127},
  code_status = {official-code},
  github = {https://github.com/Stability-AI/generative-models},
  github_stars = {27276},
  github_stars_as_of = {2026-08-30},
  github_note = {Stability AI 官方模型仓库}
}

@misc{guo2023animatediff,
  title = {{AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning}},
  author = {Yuwei Guo and Ceyuan Yang and Anyi Rao and Zhengyang Liang and Yaohui Wang and Yu Qiao and Maneesh Agrawala and Dahua Lin and Bo Dai},
  year = {2023},
  eprint = {2307.04725},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2307.04725},
  code_status = {official-code},
  github = {https://github.com/guoyww/AnimateDiff},
  github_stars = {12229},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{ho2020denoising,
  title = {{Denoising Diffusion Probabilistic Models}},
  author = {Jonathan Ho and Ajay Jain and Pieter Abbeel},
  year = {2020},
  eprint = {2006.11239},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2006.11239},
  code_status = {official-code},
  github = {https://github.com/hojonathanho/diffusion},
  github_stars = {5304},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{ho2022imagen,
  title = {{Imagen Video: High Definition Video Generation with Diffusion Models}},
  author = {Jonathan Ho and William Chan and Chitwan Saharia and Jay Whang and Ruiqi Gao and Alexey Gritsenko and Diederik P. Kingma and Ben Poole and Mohammad Norouzi and David J. Fleet and Tim Salimans},
  year = {2022},
  eprint = {2210.02303},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2210.02303},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ho2022video,
  title = {{Video Diffusion Models}},
  author = {Jonathan Ho and Tim Salimans and Alexey Gritsenko and William Chan and Mohammad Norouzi and David J. Fleet},
  year = {2022},
  eprint = {2204.03458},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2204.03458},
  code_status = {community-implementation},
  github = {https://github.com/lucidrains/video-diffusion-pytorch},
  github_stars = {1383},
  github_stars_as_of = {2026-08-30},
  github_note = {第三方复现；未发现论文作者公开的官方实现}
}

@misc{singer2022make,
  title = {{Make-A-Video: Text-to-Video Generation without Text-Video Data}},
  author = {Uriel Singer and Adam Polyak and Thomas Hayes and Xi Yin and Jie An and Songyang Zhang and Qiyuan Hu and Harry Yang and Oron Ashual and Oran Gafni and Devi Parikh and Sonal Gupta and Yaniv Taigman},
  year = {2022},
  eprint = {2209.14792},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2209.14792},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{openai2024sora,
  title = {{Video Generation Models as World Simulators}},
  author = {{OpenAI}},
  year = {2024},
  howpublished = {Technical report},
  url = {https://openai.com/index/video-generation-models-as-world-simulators/},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{bommasani2021opportunities,
  title = {{On the Opportunities and Risks of Foundation Models}},
  author = {Rishi Bommasani and Drew A. Hudson and Ehsan Adeli and Russ Altman and Simran Arora and Sydney von Arx and Michael S. Bernstein and Jeannette Bohg and Antoine Bosselut and Emma Brunskill and Erik Brynjolfsson and Shyamal Buch and Dallas Card and Rodrigo Castellon and Niladri Chatterji and Annie Chen and Kathleen Creel and Jared Quincy Davis and Dora Demszky and Chris Donahue and Moussa Doumbouya and Esin Durmus and Stefano Ermon and John Etchemendy and Kawin Ethayarajh and Li Fei-Fei and Chelsea Finn and Trevor Gale and Lauren Gillespie and Karan Goel and Noah Goodman and Shelby Grossman and Neel Guha and Tatsunori Hashimoto and Peter Henderson and John Hewitt and Daniel E. Ho and Jenny Hong and Kyle Hsu and Jing Huang and Thomas Icard and Saahil Jain and Dan Jurafsky and Pratyusha Kalluri and Siddharth Karamcheti and Geoff Keeling and Fereshte Khani and Omar Khattab and Pang Wei Koh and Mark Krass and Ranjay Krishna and Rohith Kuditipudi and Ananya Kumar and Faisal Ladhak and Mina Lee and Tony Lee and Jure Leskovec and Isabelle Levent and Xiang Lisa Li and Xuechen Li and Tengyu Ma and Ali Malik and Christopher D. Manning and Suvir Mirchandani and Eric Mitchell and Zanele Munyikwa and Suraj Nair and Avanika Narayan and Deepak Narayanan and Ben Newman and Allen Nie and Juan Carlos Niebles and Hamed Nilforoshan and Julian Nyarko and Giray Ogut and Laurel Orr and Isabel Papadimitriou and Joon Sung Park and Chris Piech and Eva Portelance and Christopher Potts and Aditi Raghunathan and Rob Reich and Hongyu Ren and Frieda Rong and Yusuf Roohani and Camilo Ruiz and Jack Ryan and Christopher Ré and Dorsa Sadigh and Shiori Sagawa and Keshav Santhanam and Andy Shih and Krishnan Srinivasan and Alex Tamkin and Rohan Taori and Armin W. Thomas and Florian Tramèr and Rose E. Wang and William Wang and Bohan Wu and Jiajun Wu and Yuhuai Wu and Sang Michael Xie and Michihiro Yasunaga and Jiaxuan You and Matei Zaharia and Michael Zhang and Tianyi Zhang and Xikun Zhang and Yuhui Zhang and Lucia Zheng and Kaitlyn Zhou and Percy Liang},
  year = {2021},
  eprint = {2108.07258},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2108.07258},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{hafner2018learning,
  title = {{Learning Latent Dynamics for Planning from Pixels}},
  author = {Danijar Hafner and Timothy Lillicrap and Ian Fischer and Ruben Villegas and David Ha and Honglak Lee and James Davidson},
  year = {2018},
  eprint = {1811.04551},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1811.04551},
  code_status = {official-code},
  github = {https://github.com/google-research/planet},
  github_stars = {1260},
  github_stars_as_of = {2026-08-30},
  github_note = {Google Research 官方实现；仓库已归档}
}

@misc{hafner2019dream,
  title = {{Dream to Control: Learning Behaviors by Latent Imagination}},
  author = {Danijar Hafner and Timothy Lillicrap and Jimmy Ba and Mohammad Norouzi},
  year = {2019},
  eprint = {1912.01603},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1912.01603},
  code_status = {official-code},
  github = {https://github.com/danijar/dreamer},
  github_stars = {622},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{hafner2023mastering,
  title = {{Mastering Diverse Domains through World Models}},
  author = {Danijar Hafner and Jurgis Pasukonis and Jimmy Ba and Timothy Lillicrap},
  year = {2023},
  eprint = {2301.04104},
  archiveprefix = {arXiv},
  primaryclass = {cs.AI},
  url = {https://arxiv.org/abs/2301.04104},
  code_status = {official-code},
  github = {https://github.com/danijar/dreamerv3},
  github_stars = {3714},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{hansen2023tdmpc2,
  title = {{TD-MPC2: Scalable, Robust World Models for Continuous Control}},
  author = {Nicklas Hansen and Hao Su and Xiaolong Wang},
  year = {2023},
  eprint = {2310.16828},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2310.16828},
  code_status = {official-code},
  github = {https://github.com/nicklashansen/tdmpc2},
  github_stars = {937},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现，含多任务 checkpoint 与数据}
}

@misc{hu2023gaia,
  title = {{GAIA-1: A Generative World Model for Autonomous Driving}},
  author = {Anthony Hu and Lloyd Russell and Hudson Yeo and Zak Murez and George Fedoseev and Alex Kendall and Jamie Shotton and Gianluca Corrado},
  year = {2023},
  eprint = {2309.17080},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2309.17080},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@article{schrittwieser2020mastering,
  title = {{Mastering Atari, Go, chess and shogi by planning with a learned model}},
  author = {Julian Schrittwieser and Ioannis Antonoglou and Thomas Hubert and Karen Simonyan and Laurent Sifre and Simon Schmitt and Arthur Guez and Edward Lockhart and Demis Hassabis and Thore Graepel and Timothy Lillicrap and David Silver},
  year = {2020},
  journal = {Nature},
  volume = {588},
  number = {7839},
  pages = {604--609},
  publisher = {Springer Science and Business Media LLC},
  doi = {10.1038/s41586-020-03051-4},
  url = {https://doi.org/10.1038/s41586-020-03051-4},
  code_status = {community-implementation},
  github = {https://github.com/werner-duvaud/muzero-general},
  github_stars = {2863},
  github_stars_as_of = {2026-08-30},
  github_note = {第三方通用实现；DeepMind 仅公开伪代码，未公开完整官方训练仓库}
}

@inproceedings{oh2015actionconditional,
  title = {{Action-Conditional Video Prediction using Deep Networks in Atari Games}},
  author = {Junhyuk Oh and Xiaoxiao Guo and Honglak Lee and Richard Lewis and Satinder Singh},
  year = {2015},
  booktitle = {Advances in Neural Information Processing Systems 28},
  url = {https://papers.nips.cc/paper_files/paper/2015/hash/6ba3af5d7b2790e73f0de32e5c8c1798-Abstract.html},
  code_status = {official-code},
  github = {https://github.com/junhyukoh/nips2015-action-conditional-video-prediction},
  github_stars = {114},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{ha2018worldmodels,
  title = {{World Models}},
  author = {David Ha and Jürgen Schmidhuber},
  year = {2018},
  doi = {10.5281/zenodo.1207631},
  eprint = {1803.10122},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1803.10122},
  code_status = {official-research-artifact},
  github = {https://github.com/hardmaru/WorldModelsExperiments},
  github_stars = {734},
  github_stars_as_of = {2026-08-30},
  github_note = {作者发布的实验代码与笔记}
}

@misc{bruce2024genie,
  title = {{Genie: Generative Interactive Environments}},
  author = {Jake Bruce and Michael Dennis and Ashley Edwards and Jack Parker-Holder and Yuge Shi and Edward Hughes and Matthew Lai and Aditi Mavalankar and Richie Steigerwald and Chris Apps and Yusuf Aytar and Sarah Bechtle and Feryal Behbahani and Stephanie Chan and Nicolas Heess and Lucy Gonzalez and Simon Osindero and Sherjil Ozair and Scott Reed and Jingwei Zhang and Konrad Zolna and Jeff Clune and Nando de Freitas and Satinder Singh and Tim Rocktäschel},
  year = {2024},
  eprint = {2402.15391},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2402.15391},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{nvidia2025cosmos,
  title = {{Cosmos World Foundation Model Platform for Physical AI}},
  author = {{NVIDIA} and Niket Agarwal and Arslan Ali and Maciej Bala and Yogesh Balaji and Erik Barker and Tiffany Cai and Prithvijit Chattopadhyay and Yongxin Chen and Yin Cui and Yifan Ding and Daniel Dworakowski and Jiaojiao Fan and Michele Fenzi and Francesco Ferroni and Sanja Fidler and Dieter Fox and Songwei Ge and Yunhao Ge and Jinwei Gu and Siddharth Gururani and Ethan He and Jiahui Huang and Jacob Huffman and Pooya Jannaty and Jingyi Jin and Seung Wook Kim and Gergely Klár and Grace Lam and Shiyi Lan and Laura Leal-Taixe and Anqi Li and Zhaoshuo Li and Chen-Hsuan Lin and Tsung-Yi Lin and Huan Ling and Ming-Yu Liu and Xian Liu and Alice Luo and Qianli Ma and Hanzi Mao and Kaichun Mo and Arsalan Mousavian and Seungjun Nah and Sriharsha Niverty and David Page and Despoina Paschalidou and Zeeshan Patel and Lindsey Pavao and Morteza Ramezanali and Fitsum Reda and Xiaowei Ren and Vasanth Rao Naik Sabavat and Ed Schmerling and Stella Shi and Bartosz Stefaniak and Shitao Tang and Lyne Tchapmi and Przemek Tredak and Wei-Cheng Tseng and Jibin Varghese and Hao Wang and Haoxiang Wang and Heng Wang and Ting-Chun Wang and Fangyin Wei and Xinyue Wei and Jay Zhangjie Wu and Jiashu Xu and Wei Yang and Lin Yen-Chen and Xiaohui Zeng and Yu Zeng and Jing Zhang and Qinsheng Zhang and Yuxuan Zhang and Qingqing Zhao and Artur Zolkowski},
  year = {2025},
  eprint = {2501.03575},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2501.03575},
  code_status = {official-code},
  github = {https://github.com/nvidia-cosmos/cosmos-predict1},
  github_stars = {468},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA Cosmos Predict1 官方实现}
}

@misc{nvidia2026cosmos3,
  title = {{Cosmos 3: Omnimodal World Models for Physical AI}},
  author = {{NVIDIA}},
  year = {2026},
  eprint = {2606.02800},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.02800},
  code_status = {official-code},
  github = {https://github.com/NVIDIA/Cosmos},
  github_stars = {11671},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA 官方统一 Cosmos 仓库}
}

@misc{valevski2024diffusion,
  title = {{Diffusion Models Are Real-Time Game Engines}},
  author = {Dani Valevski and Yaniv Leviathan and Moab Arar and Shlomi Fruchter},
  year = {2024},
  eprint = {2408.14837},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2408.14837},
  code_status = {official-project-page},
  github = {https://github.com/GameNGen/GameNGen.github.io},
  github_stars = {91},
  github_stars_as_of = {2026-08-30},
  github_note = {官方项目页源码，不含模型训练或推理实现}
}

@misc{yang2023interactive,
  title = {{Learning Interactive Real-World Simulators}},
  author = {Sherry Yang and Yilun Du and Kamyar Ghasemipour and Jonathan Tompson and Leslie Kaelbling and Dale Schuurmans and Pieter Abbeel},
  year = {2023},
  eprint = {2310.06114},
  archiveprefix = {arXiv},
  primaryclass = {cs.AI},
  url = {https://arxiv.org/abs/2310.06114},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ye2026worldaction,
  title = {{World Action Models are Zero-shot Policies}},
  author = {Seonghyeon Ye and Yunhao Ge and Kaiyuan Zheng and Shenyuan Gao and Sihyun Yu and George Kurian and Suneel Indupuru and You Liang Tan and Chuning Zhu and Jiannan Xiang and Ayaan Malik and Kyungmin Lee and William Liang and Nadun Ranawaka and Jiasheng Gu and Yinzhen Xu and Guanzhi Wang and Fengyuan Hu and Avnish Narayan and Johan Bjorck and Jing Wang and Gwanghyun Kim and Dantong Niu and Ruijie Zheng and Yuqi Xie and Jimmy Wu and Qi Wang and Ryan Julian and Danfei Xu and Yilun Du and Yevgen Chebotar and Scott Reed and Jan Kautz and Yuke Zhu and Linxi "Jim" Fan and Joel Jang},
  year = {2026},
  eprint = {2602.15922},
  archiveprefix = {arXiv},
  primaryclass = {cs.RO},
  url = {https://arxiv.org/abs/2602.15922},
  code_status = {official-code},
  github = {https://github.com/dreamzero0/dreamzero},
  github_stars = {2603},
  github_stars_as_of = {2026-08-30},
  github_note = {DreamZero / World Action Model 作者实现}
}

@misc{deepmind2024genie2,
  title = {{Genie 2: A Large-Scale Foundation World Model}},
  author = {{Google DeepMind}},
  year = {2024},
  howpublished = {Official research release},
  url = {https://deepmind.google/blog/genie-2-a-large-scale-foundation-world-model/},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{deepmind2025genie3,
  title = {{Genie 3: A New Frontier for World Models}},
  author = {{Google DeepMind}},
  year = {2025},
  howpublished = {Project report},
  url = {https://deepmind.google/blog/genie-3-a-new-frontier-for-world-models/},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{nvidia2025cosmospredict2,
  title = {{Develop Custom Physical AI Foundation Models with NVIDIA Cosmos Predict-2}},
  author = {{NVIDIA}},
  year = {2025},
  howpublished = {Technical release},
  url = {https://developer.nvidia.com/blog/?p=101575},
  code_status = {official-code},
  github = {https://github.com/nvidia-cosmos/cosmos-predict2},
  github_stars = {793},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA Cosmos 官方实现}
}

@misc{runway2025gwm1,
  title = {{Introducing Runway GWM-1}},
  author = {{Runway}},
  year = {2025},
  howpublished = {Project report},
  url = {https://runway.com/research/introducing-runway-gwm-1},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{worldlabs2025marble,
  title = {{Marble: A Multimodal World Model}},
  author = {{World Labs}},
  year = {2025},
  howpublished = {Official research release},
  url = {https://www.worldlabs.ai/blog/marble-world-model},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{assran2023selfsupervised,
  title = {{Self-Supervised Learning from Images with a Joint-Embedding Predictive Architecture}},
  author = {Mahmoud Assran and Quentin Duval and Ishan Misra and Piotr Bojanowski and Pascal Vincent and Michael Rabbat and Yann LeCun and Nicolas Ballas},
  year = {2023},
  eprint = {2301.08243},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2301.08243},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/ijepa},
  github_stars = {3489},
  github_stars_as_of = {2026-08-30},
  github_note = {Meta AI Research 官方 I-JEPA 实现；仓库已归档}
}

@misc{assran2025vjepa2,
  title = {{V-JEPA 2: Self-Supervised Video Models Enable Understanding, Prediction and Planning}},
  author = {Mahmoud Assran and Adrien Bardes and David Fan and Quentin Garrido and Russell Howes and Mojtaba Komeili and Matthew Muckley and Ammar Rizvi and Claire Roberts and Koustuv Sinha and Artem Zholus and Sergio Arnaud and Abha Gejji and Ada Martin and Francois Robert Hogan and Daniel Dugas and Piotr Bojanowski and Vasil Khalidov and Patrick Labatut and Francisco Massa and Marc Szafraniec and Kapil Krishnakumar and Yong Li and Xiaodong Ma and Sarath Chandar and Franziska Meier and Yann LeCun and Michael Rabbat and Nicolas Ballas},
  year = {2025},
  eprint = {2506.09985},
  archiveprefix = {arXiv},
  primaryclass = {cs.AI},
  url = {https://arxiv.org/abs/2506.09985},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/vjepa2},
  github_stars = {4541},
  github_stars_as_of = {2026-08-30},
  github_note = {Meta AI Research 官方实现}
}

@misc{bagatella2025tdjepa,
  title = {{TD-JEPA: Latent-predictive Representations for Zero-Shot Reinforcement Learning}},
  author = {Marco Bagatella and Matteo Pirotta and Ahmed Touati and Alessandro Lazaric and Andrea Tirinzoni},
  year = {2025},
  eprint = {2510.00739},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2510.00739},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/td_jepa},
  github_stars = {61},
  github_stars_as_of = {2026-08-30},
  github_note = {论文官方零样本强化学习实现}
}

@misc{balestriero2025lejepa,
  title = {{LeJEPA: Provable and Scalable Self-Supervised Learning Without the Heuristics}},
  author = {Randall Balestriero and Yann LeCun},
  year = {2025},
  eprint = {2511.08544},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2511.08544},
  code_status = {official-code},
  github = {https://github.com/galilai-group/lejepa},
  github_stars = {1325},
  github_stars_as_of = {2026-08-30},
  github_note = {LeJEPA 与 SIGReg 官方实现}
}

@misc{bardes2023mcjepa,
  title = {{MC-JEPA: A Joint-Embedding Predictive Architecture for Self-Supervised Learning of Motion and Content Features}},
  author = {Adrien Bardes and Jean Ponce and Yann LeCun},
  year = {2023},
  eprint = {2307.12698},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2307.12698},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{bardes2024revisiting,
  title = {{Revisiting Feature Prediction for Learning Visual Representations from Video}},
  author = {Adrien Bardes and Quentin Garrido and Jean Ponce and Xinlei Chen and Michael Rabbat and Yann LeCun and Mahmoud Assran and Nicolas Ballas},
  year = {2024},
  eprint = {2404.08471},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2404.08471},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/jepa},
  github_stars = {4112},
  github_stars_as_of = {2026-08-30},
  github_note = {Meta AI Research 官方实现}
}

@misc{maes2026leworldmodel,
  title = {{LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels}},
  author = {Lucas Maes and Quentin Le Lidec and Damien Scieur and Yann LeCun and Randall Balestriero},
  year = {2026},
  eprint = {2603.19312},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2603.19312},
  code_status = {official-code},
  github = {https://github.com/lucas-maes/le-wm},
  github_stars = {4357},
  github_stars_as_of = {2026-08-30},
  github_note = {LeWorldModel 官方实现}
}

@misc{murlabadia2026vjepa21,
  title = {{V-JEPA 2.1: Unlocking Dense Features in Video Self-Supervised Learning}},
  author = {Lorenzo Mur-Labadia and Matthew Muckley and Amir Bar and Mahmoud Assran and Koustuv Sinha and Michael Rabbat and Yann LeCun and Nicolas Ballas and Adrien Bardes},
  year = {2026},
  eprint = {2603.14482},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2603.14482},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/vjepa2},
  github_stars = {4541},
  github_stars_as_of = {2026-08-30},
  github_note = {与 V-JEPA 2 / 2-AC 共用 Meta AI Research 官方仓库}
}

@misc{terver2026lightweight,
  title = {{A Lightweight Library for Energy-Based Joint-Embedding Predictive Architectures}},
  author = {Basile Terver and Randall Balestriero and Megi Dervishi and David Fan and Quentin Garrido and Tushar Nagarajan and Koustuv Sinha and Wancong Zhang and Mike Rabbat and Yann LeCun and Amir Bar},
  year = {2026},
  eprint = {2602.03604},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2602.03604},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/eb_jepa},
  github_stars = {765},
  github_stars_as_of = {2026-08-30},
  github_note = {Meta FAIR 官方轻量教学与研究库}
}

@misc{zhou2024dinowm,
  title = {{DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning}},
  author = {Gaoyue Zhou and Hengkai Pan and Yann LeCun and Lerrel Pinto},
  year = {2024},
  eprint = {2411.04983},
  archiveprefix = {arXiv},
  primaryclass = {cs.RO},
  url = {https://arxiv.org/abs/2411.04983},
  code_status = {official-code},
  github = {https://github.com/gaoyuezhou/dino_wm},
  github_stars = {558},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{lecun2022path,
  title = {{A Path Towards Autonomous Machine Intelligence}},
  author = {Yann LeCun},
  year = {2022},
  howpublished = {Position paper, version 0.9.2},
  url = {https://openreview.net/forum?id=BZ5a1r-kVsf},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{kingma2013autoencoding,
  title = {{Auto-Encoding Variational Bayes}},
  author = {Diederik P Kingma and Max Welling},
  year = {2013},
  eprint = {1312.6114},
  archiveprefix = {arXiv},
  primaryclass = {stat.ML},
  url = {https://arxiv.org/abs/1312.6114},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{huszar2015how,
  title = {{How (not) to Train your Generative Model: Scheduled Sampling, Likelihood, Adversary?}},
  author = {Ferenc Huszár},
  year = {2015},
  eprint = {1511.05101},
  archiveprefix = {arXiv},
  primaryclass = {stat.ML},
  url = {https://arxiv.org/abs/1511.05101},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{unterthiner2018towards,
  title = {{Towards Accurate Generative Models of Video: A New Metric \& Challenges}},
  author = {Thomas Unterthiner and Sjoerd van Steenkiste and Karol Kurach and Raphael Marinier and Marcin Michalski and Sylvain Gelly},
  year = {2018},
  eprint = {1812.01717},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1812.01717},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{song2020scorebased,
  title = {{Score-Based Generative Modeling through Stochastic Differential Equations}},
  author = {Yang Song and Jascha Sohl-Dickstein and Diederik P. Kingma and Abhishek Kumar and Stefano Ermon and Ben Poole},
  year = {2020},
  eprint = {2011.13456},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2011.13456},
  code_status = {official-code},
  github = {https://github.com/yang-song/score_sde},
  github_stars = {1844},
  github_stars_as_of = {2026-08-30},
  github_note = {论文作者发布的 ICLR 2021 官方实现}
}

@misc{ho2022classifierfree,
  title = {{Classifier-Free Diffusion Guidance}},
  author = {Jonathan Ho and Tim Salimans},
  year = {2022},
  eprint = {2207.12598},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2207.12598},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{lin2024sdxllightning,
  title = {{SDXL-Lightning: Progressive Adversarial Diffusion Distillation}},
  author = {Shanchuan Lin and Anran Wang and Xiao Yang},
  year = {2024},
  eprint = {2402.13929},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2402.13929},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{lin2024animatedifflightning,
  title = {{AnimateDiff-Lightning: Cross-Model Diffusion Distillation}},
  author = {Shanchuan Lin and Xiao Yang},
  year = {2024},
  eprint = {2403.12706},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2403.12706},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{fuest2025maskflow,
  title = {{MaskFlow: Discrete Flows For Flexible and Efficient Long Video Generation}},
  author = {Michael Fuest and Vincent Tao Hu and Björn Ommer},
  year = {2025},
  eprint = {2502.11234},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2502.11234},
  code_status = {official-code},
  github = {https://github.com/CompVis/maskflow},
  github_stars = {28},
  github_stars_as_of = {2026-08-30},
  github_note = {CompVis 发布的论文官方实现}
}

@misc{xie2026videorae,
  title = {{VideoRAE: Taming Video Foundation Models for Generative Modeling via Representation Autoencoders}},
  author = {Zhihao Xie and Junfeng Wu and Xinting Hu and Junchao Huang and Li Jiang},
  year = {2026},
  eprint = {2607.14088},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2607.14088},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{shutkin2026kvae,
  title = {{KVAE: Family of Tokenizers for Multimodal Generative Models}},
  author = {Andrey Shutkin and Denis Parkhomenko and Ivan Kirillov and Kirill Chernyshev and Kirill Malakhov and Ilia Vasiliev and Ilia Trushkin and Valeriya Kobenko and David Chikovani and Alexander Ivanov and Azat Saginbaev and Egor Silvestrov and Ivan Mikheev and Konstantin Zakharov},
  year = {2026},
  eprint = {2608.05798},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2608.05798},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{guo2026vrae,
  title = {{V-RAE: Rethinking Video Latent Spaces for Generation}},
  author = {Minghui Guo and Shengqiong Wu and Hao Fei},
  year = {2026},
  eprint = {2608.13556},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2608.13556},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{polyak2024moviegen,
  title = {{Movie Gen: A Cast of Media Foundation Models}},
  author = {Adam Polyak and Amit Zohar and Andrew Brown and Andros Tjandra and Animesh Sinha and Ann Lee and Apoorv Vyas and Bowen Shi and Chih-Yao Ma and Ching-Yao Chuang and David Yan and Dhruv Choudhary and Dingkang Wang and Geet Sethi and Guan Pang and Haoyu Ma and Ishan Misra and Ji Hou and Jialiang Wang and Kiran Jagadeesh and Kunpeng Li and Luxin Zhang and Mannat Singh and Mary Williamson and Matt Le and Matthew Yu and Mitesh Kumar Singh and Peizhao Zhang and Peter Vajda and Quentin Duval and Rohit Girdhar and Roshan Sumbaly and Sai Saketh Rambhatla and Sam Tsai and Samaneh Azadi and Samyak Datta and Sanyuan Chen and Sean Bell and Sharadh Ramaswamy and Shelly Sheynin and Siddharth Bhattacharya and Simran Motwani and Tao Xu and Tianhe Li and Tingbo Hou and Wei-Ning Hsu and Xi Yin and Xiaoliang Dai and Yaniv Taigman and Yaqiao Luo and Yen-Cheng Liu and Yi-Chiao Wu and Yue Zhao and Yuval Kirstain and Zecheng He and Zijian He and Albert Pumarola and Ali Thabet and Artsiom Sanakoyeu and Arun Mallya and Baishan Guo and Boris Araya and Breena Kerr and Carleigh Wood and Ce Liu and Cen Peng and Dimitry Vengertsev and Edgar Schonfeld and Elliot Blanchard and Felix Juefei-Xu and Fraylie Nord and Jeff Liang and John Hoffman and Jonas Kohler and Kaolin Fire and Karthik Sivakumar and Lawrence Chen and Licheng Yu and Luya Gao and Markos Georgopoulos and Rashel Moritz and Sara K. Sampson and Shikai Li and Simone Parmeggiani and Steve Fine and Tara Fowler and Vladan Petrovic and Yuming Du},
  year = {2024},
  eprint = {2410.13720},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2410.13720},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{kong2024hunyuanvideo,
  title = {{HunyuanVideo: A Systematic Framework For Large Video Generative Models}},
  author = {Weijie Kong and Qi Tian and Zijian Zhang and Rox Min and Zuozhuo Dai and Jin Zhou and Jiangfeng Xiong and Xin Li and Bo Wu and Jianwei Zhang and Kathrina Wu and Qin Lin and Junkun Yuan and Yanxin Long and Aladdin Wang and Andong Wang and Changlin Li and Duojun Huang and Fang Yang and Hao Tan and Hongmei Wang and Jacob Song and Jiawang Bai and Jianbing Wu and Jinbao Xue and Joey Wang and Kai Wang and Mengyang Liu and Pengyu Li and Shuai Li and Weiyan Wang and Wenqing Yu and Xinchi Deng and Yang Li and Yi Chen and Yutao Cui and Yuanbo Peng and Zhentao Yu and Zhiyu He and Zhiyong Xu and Zixiang Zhou and Zunnan Xu and Yangyu Tao and Qinglin Lu and Songtao Liu and Dax Zhou and Hongfa Wang and Yong Yang and Di Wang and Yuhong Liu and Jie Jiang and Caesar Zhong},
  year = {2024},
  eprint = {2412.03603},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2412.03603},
  code_status = {official-code},
  github = {https://github.com/Tencent-Hunyuan/HunyuanVideo},
  github_stars = {12490},
  github_stars_as_of = {2026-08-30},
  github_note = {腾讯混元官方代码与权重仓库}
}

@misc{hacohen2024ltxvideo,
  title = {{LTX-Video: Realtime Video Latent Diffusion}},
  author = {Yoav HaCohen and Nisan Chiprut and Benny Brazowski and Daniel Shalem and Dudu Moshe and Eitan Richardson and Eran Levin and Guy Shiran and Nir Zabari and Ori Gordon and Poriya Panet and Sapir Weissbuch and Victor Kulikov and Yaki Bitterman and Zeev Melumian and Ofir Bibi},
  year = {2024},
  eprint = {2501.00103},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2501.00103},
  code_status = {official-code},
  github = {https://github.com/Lightricks/LTX-Video},
  github_stars = {10916},
  github_stars_as_of = {2026-08-30},
  github_note = {Lightricks 官方代码、权重与训练工具}
}

@misc{ma2025stepvideo,
  title = {{Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model}},
  author = {Guoqing Ma and Haoyang Huang and Kun Yan and Liangyu Chen and Nan Duan and Shengming Yin and Changyi Wan and Ranchen Ming and Xiaoniu Song and Xing Chen and Yu Zhou and Deshan Sun and Deyu Zhou and Jian Zhou and Kaijun Tan and Kang An and Mei Chen and Wei Ji and Qiling Wu and Wen Sun and Xin Han and Yanan Wei and Zheng Ge and Aojie Li and Bin Wang and Bizhu Huang and Bo Wang and Brian Li and Changxing Miao and Chen Xu and Chenfei Wu and Chenguang Yu and Dapeng Shi and Dingyuan Hu and Enle Liu and Gang Yu and Ge Yang and Guanzhe Huang and Gulin Yan and Haiyang Feng and Hao Nie and Haonan Jia and Hanpeng Hu and Hanqi Chen and Haolong Yan and Heng Wang and Hongcheng Guo and Huilin Xiong and Huixin Xiong and Jiahao Gong and Jianchang Wu and Jiaoren Wu and Jie Wu and Jie Yang and Jiashuai Liu and Jiashuo Li and Jingyang Zhang and Junjing Guo and Junzhe Lin and Kaixiang Li and Lei Liu and Lei Xia and Liang Zhao and Liguo Tan and Liwen Huang and Liying Shi and Ming Li and Mingliang Li and Muhua Cheng and Na Wang and Qiaohui Chen and Qinglin He and Qiuyan Liang and Quan Sun and Ran Sun and Rui Wang and Shaoliang Pang and Shiliang Yang and Sitong Liu and Siqi Liu and Shuli Gao and Tiancheng Cao and Tianyu Wang and Weipeng Ming and Wenqing He and Xu Zhao and Xuelin Zhang and Xianfang Zeng and Xiaojia Liu and Xuan Yang and Yaqi Dai and Yanbo Yu and Yang Li and Yineng Deng and Yingming Wang and Yilei Wang and Yuanwei Lu and Yu Chen and Yu Luo and Yuchu Luo and Yuhe Yin and Yuheng Feng and Yuxiang Yang and Zecheng Tang and Zekai Zhang and Zidong Yang and Binxing Jiao and Jiansheng Chen and Jing Li and Shuchang Zhou and Xiangyu Zhang and Xinhao Zhang and Yibo Zhu and Heung-Yeung Shum and Daxin Jiang},
  year = {2025},
  eprint = {2502.10248},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2502.10248},
  code_status = {official-code},
  github = {https://github.com/stepfun-ai/Step-Video-T2V},
  github_stars = {3188},
  github_stars_as_of = {2026-08-30},
  github_note = {阶跃星辰官方代码、权重与评测入口}
}

@misc{zheng2025opensora2,
  title = {{Open-Sora 2.0: Training a Commercial-Level Video Generation Model in $200k}},
  author = {Zangwei Zheng and Xiangyu Peng and Yuxuan Lou and Chenhui Shen and Tom Young and Xinying Guo and Binluo Wang and Hang Xu and Hongxin Liu and Mingyan Jiang and Wenjun Li and Yuhui Wang and Anbang Ye and Gang Ren and Qianran Ma and Wanying Liang and Xiang Lian and Xiwen Wu and Yuting Zhong and Zhuangyan Li and Chaoyu Gong and Guojun Lei and Leijun Cheng and Limin Zhang and Minghao Li and Ruijie Zhang and Silan Hu and Shijie Huang and Xiaokang Wang and Yuanheng Zhao and Yuqi Wang and Ziang Wei and Yang You},
  year = {2025},
  eprint = {2503.09642},
  archiveprefix = {arXiv},
  primaryclass = {cs.GR},
  url = {https://arxiv.org/abs/2503.09642},
  code_status = {official-code},
  github = {https://github.com/hpcaitech/Open-Sora},
  github_stars = {29322},
  github_stars_as_of = {2026-08-30},
  github_note = {Open-Sora 2.0 官方训练代码与 checkpoints}
}

@misc{wan2025wan,
  title = {{Wan: Open and Advanced Large-Scale Video Generative Models}},
  author = {{Wan Team} and Ang Wang and Baole Ai and Bin Wen and Chaojie Mao and Chen-Wei Xie and Di Chen and Feiwu Yu and Haiming Zhao and Jianxiao Yang and Jianyuan Zeng and Jiayu Wang and Jingfeng Zhang and Jingren Zhou and Jinkai Wang and Jixuan Chen and Kai Zhu and Kang Zhao and Keyu Yan and Lianghua Huang and Mengyang Feng and Ningyi Zhang and Pandeng Li and Pingyu Wu and Ruihang Chu and Ruili Feng and Shiwei Zhang and Siyang Sun and Tao Fang and Tianxing Wang and Tianyi Gui and Tingyu Weng and Tong Shen and Wei Lin and Wei Wang and Wei Wang and Wenmeng Zhou and Wente Wang and Wenting Shen and Wenyuan Yu and Xianzhong Shi and Xiaoming Huang and Xin Xu and Yan Kou and Yangyu Lv and Yifei Li and Yijing Liu and Yiming Wang and Yingya Zhang and Yitong Huang and Yong Li and You Wu and Yu Liu and Yulin Pan and Yun Zheng and Yuntao Hong and Yupeng Shi and Yutong Feng and Zeyinzi Jiang and Zhen Han and Zhi-Fan Wu and Ziyu Liu},
  year = {2025},
  eprint = {2503.20314},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2503.20314},
  code_status = {official-code},
  github = {https://github.com/Wan-Video/Wan2.1},
  github_stars = {16908},
  github_stars_as_of = {2026-08-30},
  github_note = {Wan 2.1 官方代码、权重与模型卡}
}

@misc{chen2025skyreelsv2,
  title = {{SkyReels-V2: Infinite-length Film Generative Model}},
  author = {Guibin Chen and Dixuan Lin and Jiangping Yang and Chunze Lin and Junchen Zhu and Mingyuan Fan and Hao Zhang and Sheng Chen and Zheng Chen and Chengcheng Ma and Weiming Xiong and Wei Wang and Nuo Pang and Kang Kang and Zhiheng Xu and Yuzhe Jin and Yupeng Liang and Yubing Song and Peng Zhao and Boyuan Xu and Di Qiu and Debang Li and Zhengcong Fei and Yang Li and Yahui Zhou},
  year = {2025},
  eprint = {2504.13074},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2504.13074},
  code_status = {official-code},
  github = {https://github.com/SkyworkAI/SkyReels-V2},
  github_stars = {7472},
  github_stars_as_of = {2026-08-30},
  github_note = {Skywork AI 官方代码与权重}
}

@misc{sandai2025magi1,
  title = {{MAGI-1: Autoregressive Video Generation at Scale}},
  author = {{Sand.ai} and Hansi Teng and Hongyu Jia and Lei Sun and Lingzhi Li and Maolin Li and Mingqiu Tang and Shuai Han and Tianning Zhang and W. Q. Zhang and Weifeng Luo and Xiaoyang Kang and Yuchen Sun and Yue Cao and Yunpeng Huang and Yutong Lin and Yuxin Fang and Zewei Tao and Zheng Zhang and Zhongshu Wang and Zixun Liu and Dai Shi and Guoli Su and Hanwen Sun and Hong Pan and Jie Wang and Jiexin Sheng and Min Cui and Min Hu and Ming Yan and Shucheng Yin and Siran Zhang and Tingting Liu and Xianping Yin and Xiaoyu Yang and Xin Song and Xuan Hu and Yankai Zhang and Yuqiao Li},
  year = {2025},
  eprint = {2505.13211},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2505.13211},
  code_status = {official-code},
  github = {https://github.com/SandAI-org/MAGI-1},
  github_stars = {3774},
  github_stars_as_of = {2026-08-30},
  github_note = {Sand.ai 官方代码、权重及蒸馏与量化版本}
}

@misc{low2025ovi,
  title = {{Ovi: Twin Backbone Cross-Modal Fusion for Audio-Video Generation}},
  author = {Chetwin Low and Weimin Wang and Calder Katyal},
  year = {2025},
  eprint = {2510.01284},
  archiveprefix = {arXiv},
  primaryclass = {cs.MM},
  url = {https://arxiv.org/abs/2510.01284},
  code_status = {official-code},
  github = {https://github.com/character-ai/Ovi},
  github_stars = {1750},
  github_stars_as_of = {2026-08-30},
  github_note = {Character.AI 官方代码与权重}
}

@misc{wu2025hunyuanvideo15,
  title = {{HunyuanVideo 1.5 Technical Report}},
  author = {Bing Wu and Chang Zou and Changlin Li and Duojun Huang and Fang Yang and Hao Tan and Jack Peng and Jianbing Wu and Jiangfeng Xiong and Jie Jiang and Linus and Patrol and Peizhen Zhang and Peng Chen and Penghao Zhao and Qi Tian and Songtao Liu and Weijie Kong and Weiyan Wang and Xiao He and Xin Li and Xinchi Deng and Xuefei Zhe and Yang Li and Yanxin Long and Yuanbo Peng and Yue Wu and Yuhong Liu and Zhenyu Wang and Zuozhuo Dai and Bo Peng and Coopers Li and Gu Gong and Guojian Xiao and Jiahe Tian and Jiaxin Lin and Jie Liu and Jihong Zhang and Jiesong Lian and Kaihang Pan and Lei Wang and Lin Niu and Mingtao Chen and Mingyang Chen and Mingzhe Zheng and Miles Yang and Qiangqiang Hu and Qi Yang and Qiuyong Xiao and Runzhou Wu and Ryan Xu and Rui Yuan and Shanshan Sang and Shisheng Huang and Siruis Gong and Shuo Huang and Weiting Guo and Xiang Yuan and Xiaojia Chen and Xiawei Hu and Wenzhi Sun and Xiele Wu and Xianshun Ren and Xiaoyan Yuan and Xiaoyue Mi and Yepeng Zhang and Yifu Sun and Yiting Lu and Yitong Li and You Huang and Yu Tang and Yixuan Li and Yuhang Deng and Yuan Zhou and Zhichao Hu and Zhiguang Liu and Zhihe Yang and Zilin Yang and Zhenzhi Lu and Zixiang Zhou and Zhao Zhong},
  year = {2025},
  eprint = {2511.18870},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2511.18870},
  code_status = {official-code},
  github = {https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5},
  github_stars = {4537},
  github_stars_as_of = {2026-08-30},
  github_note = {腾讯混元 1.5 官方代码、权重与训练更新}
}

@misc{hacohen2026ltx2,
  title = {{LTX-2: Efficient Joint Audio-Visual Foundation Model}},
  author = {Yoav HaCohen and Benny Brazowski and Nisan Chiprut and Yaki Bitterman and Andrew Kvochko and Avishai Berkowitz and Daniel Shalem and Daphna Lifschitz and Dudu Moshe and Eitan Porat and Eitan Richardson and Guy Shiran and Itay Chachy and Jonathan Chetboun and Michael Finkelson and Michael Kupchick and Nir Zabari and Nitzan Guetta and Noa Kotler and Ofir Bibi and Ori Gordon and Poriya Panet and Roi Benita and Shahar Armon and Victor Kulikov and Yaron Inger and Yonatan Shiftan and Zeev Melumian and Zeev Farbman},
  year = {2026},
  eprint = {2601.03233},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2601.03233},
  code_status = {official-code},
  github = {https://github.com/Lightricks/LTX-2},
  github_stars = {9294},
  github_stars_as_of = {2026-08-30},
  github_note = {Lightricks 官方音视频推理与 LoRA 训练仓库}
}

@misc{li2026skyreelsv3,
  title = {{SkyReels-V3 Technique Report}},
  author = {Debang Li and Zhengcong Fei and Tuanhui Li and Yikun Dou and Zheng Chen and Jiangping Yang and Mingyuan Fan and Jingtao Xu and Jiahua Wang and Baoxuan Gu and Mingshan Chang and Wenjing Cai and Yuqiang Xie and Binjie Mao and Youqiang Zhang and Nuo Pang and Hao Zhang and Yuzhe Jin and Zhiheng Xu and Dixuan Lin and Guibin Chen and Yahui Zhou},
  year = {2026},
  eprint = {2601.17323},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2601.17323},
  code_status = {official-code},
  github = {https://github.com/SkyworkAI/SkyReels-V3},
  github_stars = {555},
  github_stars_as_of = {2026-08-30},
  github_note = {Skywork AI 官方推理代码与任务权重}
}

@misc{seedance2026seedance2,
  title = {{Seedance 2.0: Advancing Video Generation for World Complexity}},
  author = {{Seedance Team} and De Chen and Liyang Chen and Xin Chen and Ying Chen and Zhuo Chen and Zhuowei Chen and Feng Cheng and Tianheng Cheng and Yufeng Cheng and Mojie Chi and Xuyan Chi and Jian Cong and Qinpeng Cui and Fei Ding and Qide Dong and Yujiao Du and Haojie Duanmu and Junliang Fan and Jiarui Fang and Jing Fang and Zetao Fang and Chengjian Feng and Yu Gao and Diandian Gu and Dong Guo and Hanzhong Guo and Qiushan Guo and Boyang Hao and Hongxiang Hao and Haoxun He and Jiaao He and Qian He and Tuyen Hoang and Heng Hu and Ruoqing Hu and Yuxiang Hu and Jiancheng Huang and Weilin Huang and Zhaoyang Huang and Zhongyi Huang and Jishuo Jin and Ming Jing and Ashley Kim and Shanshan Lao and Yichong Leng and Bingchuan Li and Gen Li and Haifeng Li and Huixia Li and Jiashi Li and Ming Li and Xiaojie Li and Xingxing Li and Yameng Li and Yiying Li and Yu Li and Yueyan Li and Chao Liang and Han Liang and Jianzhong Liang and Ying Liang and Wang Liao and J. H. Lien and Shanchuan Lin and Xi Lin and Feng Ling and Yue Ling and Fangfang Liu and Jiawei Liu and Jihao Liu and Jingtuo Liu and Shu Liu and Sichao Liu and Wei Liu and Xue Liu and Zuxi Liu and Ruijie Lu and Lecheng Lyu and Jingting Ma and Tianxiang Ma and Xiaonan Nie and Jingzhe Ning and Junjie Pan and Xitong Pan and Ronggui Peng and Xueqiong Qu and Yuxi Ren and Yuchen Shen and Guang Shi and Lei Shi and Yinglong Song and Fan Sun and Li Sun and Renfei Sun and Wenjing Tang and Boyang Tao and Zirui Tao and Dongliang Wang and Feng Wang and Hulin Wang and Ke Wang and Qingyi Wang and Rui Wang and Shuai Wang and Shulei Wang and Weichen Wang and Xuanda Wang and Yanhui Wang and Yue Wang and Yuping Wang and Yuxuan Wang and Zijie Wang and Ziyu Wang and Guoqiang Wei and Meng Wei and Di Wu and Guohong Wu and Hanjie Wu and Huachao Wu and Jian Wu and Jie Wu and Ruolan Wu and Shaojin Wu and Xiaohu Wu and Xinglong Wu and Yonghui Wu and Ruiqi Xia and Xin Xia and Xuefeng Xiao and Shuang Xu and Bangbang Yang and Jiaqi Yang and Runkai Yang and Tao Yang and Yihang Yang and Zhixian Yang and Ziyan Yang and Fulong Ye and Bingqian Yi and Xing Yin and Yongbin You and Linxiao Yuan and Weihong Zeng and Xuejiao Zeng and Yan Zeng and Siyu Zhai and Zhonghua Zhai and Bowen Zhang and Chenlin Zhang and Heng Zhang and Jun Zhang and Manlin Zhang and Peiyuan Zhang and Shuo Zhang and Xiaohe Zhang and Xiaoying Zhang and Xinyan Zhang and Xinyi Zhang and Yichi Zhang and Zixiang Zhang and Haiyu Zhao and Huating Zhao and Liming Zhao and Yian Zhao and Guangcong Zheng and Jianbin Zheng and Xiaozheng Zheng and Zerong Zheng and Kuan Zhu and Feilong Zuo},
  year = {2026},
  eprint = {2604.14148},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2604.14148},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{chen2024diffusionforcing,
  title = {{Diffusion Forcing: Next-token Prediction Meets Full-Sequence Diffusion}},
  author = {Boyuan Chen and Diego Marti Monso and Yilun Du and Max Simchowitz and Russ Tedrake and Vincent Sitzmann},
  year = {2024},
  eprint = {2407.01392},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2407.01392},
  code_status = {official-code},
  github = {https://github.com/buoyancy99/diffusion-forcing},
  github_stars = {1288},
  github_stars_as_of = {2026-08-30},
  github_note = {论文作者发布的官方实现}
}

@misc{yin2024causvid,
  title = {{From Slow Bidirectional to Fast Autoregressive Video Diffusion Models}},
  author = {Tianwei Yin and Qiang Zhang and Richard Zhang and William T. Freeman and Fredo Durand and Eli Shechtman and Xun Huang},
  year = {2024},
  eprint = {2412.07772},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2412.07772},
  code_status = {official-code},
  github = {https://github.com/tianweiy/CausVid},
  github_stars = {1428},
  github_stars_as_of = {2026-08-30},
  github_note = {CausVid / CVPR 2025 官方实现}
}

@misc{huang2025selfforcing,
  title = {{Self Forcing: Bridging the Train-Test Gap in Autoregressive Video Diffusion}},
  author = {Xun Huang and Zhengqi Li and Guande He and Mingyuan Zhou and Eli Shechtman},
  year = {2025},
  eprint = {2506.08009},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2506.08009},
  code_status = {official-code},
  github = {https://github.com/guandeh17/Self-Forcing},
  github_stars = {3491},
  github_stars_as_of = {2026-08-30},
  github_note = {NeurIPS 2025 论文官方实现}
}

@misc{yang2025longlive,
  title = {{LongLive: Real-time Interactive Long Video Generation}},
  author = {Shuai Yang and Wei Huang and Ruihang Chu and Yicheng Xiao and Yuyang Zhao and Xianbang Wang and Muyang Li and Enze Xie and Yingcong Chen and Yao Lu and Song Han and Yukang Chen},
  year = {2025},
  eprint = {2509.22622},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2509.22622},
  code_status = {official-code},
  github = {https://github.com/NVlabs/LongLive},
  github_stars = {2582},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA Research 官方代码与长视频基础设施}
}

@misc{liu2025rollingforcing,
  title = {{Rolling Forcing: Autoregressive Long Video Diffusion in Real Time}},
  author = {Kunhao Liu and Wenbo Hu and Jiale Xu and Ying Shan and Shijian Lu},
  year = {2025},
  eprint = {2509.25161},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2509.25161},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{yu2025videossm,
  title = {{VideoSSM: Autoregressive Long Video Generation with Hybrid State-Space Memory}},
  author = {Yifei Yu and Xiaoshan Wu and Xinting Hu and Tao Hu and Yangtian Sun and Xiaoyang Lyu and Bo Wang and Lin Ma and Yuewen Ma and Zhongrui Wang and Xiaojuan Qi},
  year = {2025},
  eprint = {2512.04519},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2512.04519},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{samuel2026fastautoregressive,
  title = {{Fast Autoregressive Video Diffusion and World Models with Temporal Cache Compression and Sparse Attention}},
  author = {Dvir Samuel and Issar Tzachor and Matan Levy and Michael Green and Gal Chechik and Rami Ben-Ari},
  year = {2026},
  eprint = {2602.01801},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2602.01801},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{zhu2026causalforcing,
  title = {{Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation}},
  author = {Hongzhou Zhu and Min Zhao and Guande He and Hang Su and Chongxuan Li and Jun Zhu},
  year = {2026},
  eprint = {2602.02214},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2602.02214},
  code_status = {official-code},
  github = {https://github.com/thu-ml/Causal-Forcing},
  github_stars = {938},
  github_stars_as_of = {2026-08-30},
  github_note = {ICML 2026 论文官方实现}
}

@misc{xi2026quantvideogen,
  title = {{Quant VideoGen: Auto-Regressive Long Video Generation via 2-Bit KV-Cache Quantization}},
  author = {Haocheng Xi and Shuo Yang and Yilong Zhao and Muyang Li and Han Cai and Xingyang Li and Yujun Lin and Zhuoyang Zhang and Jintao Zhang and Xiuyu Li and Zhiying Xu and Jun Wu and Chenfeng Xu and Ion Stoica and Song Han and Kurt Keutzer},
  year = {2026},
  eprint = {2602.02958},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2602.02958},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{lv2026lightforcing,
  title = {{Light Forcing: Accelerating Autoregressive Video Diffusion via Sparse Attention}},
  author = {Chengtao Lv and Yumeng Shi and Yushi Huang and Ruihao Gong and Shen Ren and Wenya Wang},
  year = {2026},
  eprint = {2602.04789},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2602.04789},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{li2026rollingsink,
  title = {{Rolling Sink: Bridging Limited-Horizon Training and Open-Ended Testing in Autoregressive Video Diffusion}},
  author = {Haodong Li and Shaoteng Liu and Zhe Lin and Manmohan Chandraker},
  year = {2026},
  eprint = {2602.07775},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2602.07775},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{xu2026sparseforcing,
  title = {{Sparse Forcing: Native Trainable Sparse Attention for Real-time Autoregressive Diffusion Video Generation}},
  author = {Boxun Xu and Yuming Du and Zichang Liu and Siyu Yang and Ziyang Jiang and Siqi Yan and Rajasi Saha and Albert Pumarola and Wenchen Wang and Peng Li},
  year = {2026},
  eprint = {2604.21221},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2604.21221},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{xue2026systematic,
  title = {{A Systematic Post-Train Framework for Video Generation}},
  author = {Zeyue Xue and Siming Fu and Jie Huang and Shuai Lu and Haoran Li and Yijun Liu and Yuming Li and Xiaoxuan He and Mengzhao Chen and Haoyang Huang and Nan Duan and Ping Luo},
  year = {2026},
  eprint = {2604.25427},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2604.25427},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ji2026forcingkv,
  title = {{Forcing-KV: Hybrid KV Cache Compression for Efficient Autoregressive Video Diffusion Models}},
  author = {Yicheng Ji and Zhizhou Zhong and Jun Zhang and Qin Yang and XiTai Jin and Ying Qin and Wenhan Luo and Shuiyang Mao and Wei Liu and Huan Li},
  year = {2026},
  eprint = {2605.09681},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2605.09681},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{zhao2026causalforcingpp,
  title = {{Causal Forcing++: Scalable Few-Step Autoregressive Diffusion Distillation for Real-Time Interactive Video Generation}},
  author = {Min Zhao and Hongzhou Zhu and Kaiwen Zheng and Zihan Zhou and Bokai Yan and Xinyuan Li and Xiao Yang and Chongxuan Li and Jun Zhu},
  year = {2026},
  eprint = {2605.15141},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2605.15141},
  code_status = {official-code},
  github = {https://github.com/thu-ml/Causal-Forcing},
  github_stars = {938},
  github_stars_as_of = {2026-08-30},
  github_note = {与 Causal Forcing 共用官方仓库；README 明确覆盖 Causal Forcing++}
}

@misc{li2026attendlocally,
  title = {{Attend Locally, Remember Linearly: Linear Attention as Cross-Frame Memory for Autoregressive Video Diffusion}},
  author = {Kunyang Li and Mubarak Shah and Yuzhang Shang},
  year = {2026},
  eprint = {2605.16579},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2605.16579},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{yesiltepe2026videomla,
  title = {{VideoMLA: Low-Rank Latent KV Cache for Minute-Scale Autoregressive Video Diffusion}},
  author = {Hidir Yesiltepe and Jiazhen Hu and Tuna Han Salih Meral and Adil Kaan Akan and Kaan Oktay and Hoda Eldardiry and Pinar Yanardag},
  year = {2026},
  eprint = {2605.30351},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2605.30351},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{hu2026longliverag,
  title = {{LongLive-RAG: A General Retrieval-Augmented Framework for Long Video Generation}},
  author = {Qixin Hu and Shuai Yang and Wei Huang and Song Han and Yukang Chen},
  year = {2026},
  eprint = {2606.02553},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.02553},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{yu2026videomirai,
  title = {{Video-Mirai: Autoregressive Video Diffusion Models Need Foresight}},
  author = {Yonghao Yu and Lang Huang and Runyi Li and Zerun Wang and Toshihiko Yamasaki},
  year = {2026},
  eprint = {2606.03971},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.03971},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{li2026aad1,
  title = {{AAD-1: Asymmetric Adversarial Distillation for One-Step Autoregressive Video Generation}},
  author = {Haobo Li and Yanhong Zeng and Yunhong Lu and Jiapeng Zhu and Hao Ouyang and Qiuyu Wang and Ka Leong Cheng and Yujun Shen and Zhipeng Zhang},
  year = {2026},
  eprint = {2606.03972},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.03972},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{lu2026fademem,
  title = {{FadeMem: Distance-Aware Memory Consolidation for Autoregressive Video Diffusion}},
  author = {Yu Lu and Junjie Yang and Piotr Koniusz and YuXin Song and Yi Yang},
  year = {2026},
  eprint = {2606.10671},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.10671},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{zheng2026causalrcm,
  title = {{Causal-rCM: A Unified Teacher-Forcing and Self-Forcing Open Recipe for Autoregressive Diffusion Distillation in Streaming Video Generation and Interactive World Models}},
  author = {Kaiwen Zheng and Guande He and Min Zhao and Jintao Zhang and Huayu Chen and Jianfei Chen and Chen-Hsuan Lin and Ming-Yu Liu and Jun Zhu and Qianli Ma},
  year = {2026},
  eprint = {2606.25473},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2606.25473},
  code_status = {official-code},
  github = {https://github.com/NVlabs/rcm},
  github_stars = {794},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA Research 官方 rCM / Causal-rCM 代码与配方}
}

@misc{fiebelman2026mvforcing,
  title = {{MV-Forcing: Long Multi-View Video Generation via 4D-Grounded Spatio-Temporal Self-Forcing}},
  author = {Gal Fiebelman and Hadar Averbuch-Elor and Sagie Benaim},
  year = {2026},
  eprint = {2607.05376},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2607.05376},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{zhuang2026selfgradient,
  title = {{Self Gradient Forcing: Native Long Video Extrapolation}},
  author = {Junhao Zhuang and Shiyi Zhang and Yuxuan Bian and Yaowei Li and Yawen Luo and Yijun Liu and Weiyang Jin and Songchun Zhang and Xianglong He and Xuying Zhang and Haoran Li and Haoyang Huang and Zeyue Xue and Nan Duan},
  year = {2026},
  eprint = {2607.20368},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2607.20368},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{xiao2026joyaivideoedit,
  title = {{JoyAI-Video-Edit: Real-Time Open-Ended Video Editing with Autoregressive Diffusion}},
  author = {Yicheng Xiao and Wenxun Dai and Xinran Qin and Lin Song and Maoquan Zhang and Hang Xu and Yukang Chen and Yitong Li and Guohui Zhang and Yuan Zhang and Xuying Zhang and Tommy Zhang and Jianlong Yuan and Peihao Li and Shuai Lu and Siming Fu and Chuyang Zhao and Xin Han and Jie Huang and Wenbo Li and Guoqing Ma and Wei Huang and Xiaojuan Qi and Haoyang Huang and Nan Duan},
  year = {2026},
  eprint = {2608.03974},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2608.03974},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ban2026stream4d,
  title = {{Stream4D: 4D-Consistency for Streaming Autoregressive Diffusion Video Models}},
  author = {Yuanhao Ban and Jiaqi Feng and Hengguang Zhou and Xiaohuan Pei and Justin Cui and Cho-Jui Hsieh},
  year = {2026},
  eprint = {2608.19556},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2608.19556},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{peebles2022scalable,
  title = {{Scalable Diffusion Models with Transformers}},
  author = {William Peebles and Saining Xie},
  year = {2022},
  eprint = {2212.09748},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2212.09748},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/DiT},
  github_stars = {8689},
  github_stars_as_of = {2026-08-30},
  github_note = {作者官方 DiT 实现；仓库已归档}
}

@misc{gupta2023photorealistic,
  title = {{Photorealistic Video Generation with Diffusion Models}},
  author = {Agrim Gupta and Lijun Yu and Kihyuk Sohn and Xiuye Gu and Meera Hahn and Li Fei-Fei and Irfan Essa and Lu Jiang and José Lezama},
  year = {2023},
  eprint = {2312.06662},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2312.06662},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ma2024latte,
  title = {{Latte: Latent Diffusion Transformer for Video Generation}},
  author = {Xin Ma and Yaohui Wang and Xinyuan Chen and Gengyun Jia and Ziwei Liu and Yuan-Fang Li and Cunjian Chen and Yu Qiao},
  year = {2024},
  journal = {Transactions on Machine Learning Research},
  eprint = {2401.03048},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2401.03048},
  code_status = {official-code},
  github = {https://github.com/Vchitect/Latte},
  github_stars = {1948},
  github_stars_as_of = {2026-08-30},
  github_note = {作者官方 Latte 训练与推理实现}
}

@misc{yang2024cogvideox,
  title = {{CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer}},
  author = {Zhuoyi Yang and Jiayan Teng and Wendi Zheng and Ming Ding and Shiyu Huang and Jiazheng Xu and Yuanming Yang and Wenyi Hong and Xiaohan Zhang and Guanyu Feng and Da Yin and Yuxuan Zhang and Weihan Wang and Yean Cheng and Bin Xu and Xiaotao Gu and Yuxiao Dong and Jie Tang},
  year = {2024},
  eprint = {2408.06072},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2408.06072},
  code_status = {official-code},
  github = {https://github.com/zai-org/CogVideo},
  github_stars = {12985},
  github_stars_as_of = {2026-08-30},
  github_note = {智谱官方 CogVideoX 代码、权重与训练入口}
}

@misc{chen2025sanavideo,
  title = {{SANA-Video: Efficient Video Generation with Block Linear Diffusion Transformer}},
  author = {Junsong Chen and Yuyang Zhao and Jincheng Yu and Ruihang Chu and Junyu Chen and Shuai Yang and Xianbang Wang and Yicheng Pan and Daquan Zhou and Huan Ling and Haozhe Liu and Hongwei Yi and Hao Zhang and Muyang Li and Yukang Chen and Han Cai and Sanja Fidler and Ping Luo and Song Han and Enze Xie},
  year = {2025},
  eprint = {2509.24695},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2509.24695},
  code_status = {official-code},
  github = {https://github.com/NVlabs/Sana},
  github_stars = {8890},
  github_stars_as_of = {2026-08-30},
  github_note = {NVIDIA Research 官方 SANA 与 SANA-Video 实现和权重入口}
}

@misc{huang2025linvideo,
  title = {{LinVideo: A Post-Training Framework towards O(n) Attention in Efficient Video Generation}},
  author = {Yushi Huang and Xingtong Ge and Ruihao Gong and Chengtao Lv and Jun Zhang},
  year = {2025},
  eprint = {2510.08318},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2510.08318},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{mao2025timeripples,
  title = {{Timeripple: Accelerating vDiTs by Understanding the Spatio-Temporal Correlations in Latent Space}},
  author = {Wenxuan Miao and Yulin Sun and Aiyue Chen and Jing Lin and Yiwu Yao and Yiming Gan and Jieru Zhao and Jingwen Leng and Minyi Guo and Yu Feng},
  year = {2025},
  eprint = {2511.12035},
  archiveprefix = {arXiv},
  primaryclass = {cs.AR},
  url = {https://arxiv.org/abs/2511.12035},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{chen2026sanavideo2,
  title = {{SANA-Video 2.0: Hybrid Linear Attention with Attention Residuals for Efficient Video Generation}},
  author = {Junsong Chen and Jincheng Yu and Yitong Li and Shuchen Xue and Haozhe Liu and Jingyu Xin and Yuyang Zhao and Tian Ye and Zhangjie Wu and Zian Wang and Daquan Zhou and Ping Luo and Song Han and Enze Xie},
  year = {2026},
  eprint = {2607.21553},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2607.21553},
  code_status = {official-code},
  github = {https://github.com/NVlabs/Sana},
  github_stars = {8890},
  github_stars_as_of = {2026-08-30},
  github_note = {官方统一仓库；具体已发布 checkpoint 仍须按模型文档核对}
}

@misc{chung2015recurrent,
  title = {{A Recurrent Latent Variable Model for Sequential Data}},
  author = {Junyoung Chung and Kyle Kastner and Laurent Dinh and Kratarth Goel and Aaron Courville and Yoshua Bengio},
  year = {2015},
  eprint = {1506.02216},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/1506.02216},
  code_status = {official-code},
  github = {https://github.com/jych/nips2015_vrnn},
  github_stars = {291},
  github_stars_as_of = {2026-08-30},
  github_note = {作者发布的旧版 Theano 实现}
}

@misc{babaeizadeh2017stochastic,
  title = {{Stochastic Variational Video Prediction}},
  author = {Mohammad Babaeizadeh and Chelsea Finn and Dumitru Erhan and Roy H. Campbell and Sergey Levine},
  year = {2017},
  eprint = {1710.11252},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1710.11252},
  code_status = {official-code},
  github = {https://github.com/tensorflow/tensor2tensor},
  github_stars = {17464},
  github_stars_as_of = {2026-08-30},
  github_note = {SV2P 位于该官方仓库；仓库已归档且标记为弃用}
}

@misc{denton2018stochastic,
  title = {{Stochastic Video Generation with a Learned Prior}},
  author = {Remi Denton and Rob Fergus},
  year = {2018},
  eprint = {1802.07687},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1802.07687},
  code_status = {official-code},
  github = {https://github.com/edenton/svg},
  github_stars = {188},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现}
}

@misc{lee2018stochastic,
  title = {{Stochastic Adversarial Video Prediction}},
  author = {Alex X. Lee and Richard Zhang and Frederik Ebert and Pieter Abbeel and Chelsea Finn and Sergey Levine},
  year = {2018},
  eprint = {1804.01523},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1804.01523},
  code_status = {official-code},
  github = {https://github.com/alexlee-gk/video_prediction},
  github_stars = {304},
  github_stars_as_of = {2026-08-30},
  github_note = {作者发布的 SAVP 官方实现}
}

@misc{castrejon2019improved,
  title = {{Improved Conditional VRNNs for Video Prediction}},
  author = {Lluis Castrejon and Nicolas Ballas and Aaron Courville},
  year = {2019},
  eprint = {1904.12165},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/1904.12165},
  code_status = {official-code},
  github = {https://github.com/facebookresearch/improved_vrnn},
  github_stars = {39},
  github_stars_as_of = {2026-08-30},
  github_note = {作者配套实现；仓库已归档}
}

@misc{franceschi2020stochastic,
  title = {{Stochastic Latent Residual Video Prediction}},
  author = {Jean-Yves Franceschi and Edouard Delasalles and Mickaël Chen and Sylvain Lamprier and Patrick Gallinari},
  year = {2020},
  journal = {Thirty-seventh International Conference on Machine Learning, International Machine Learning Society, Jul 2020, Vienne, Austria. pp.89--102},
  eprint = {2002.09219},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2002.09219},
  code_status = {official-code},
  github = {https://github.com/edouardelasalles/srvp},
  github_stars = {74},
  github_stars_as_of = {2026-08-30},
  github_note = {作者实现，并提供预训练模型入口}
}

@misc{wu2021greedy,
  title = {{Greedy Hierarchical Variational Autoencoders for Large-Scale Video Prediction}},
  author = {Bohan Wu and Suraj Nair and Roberto Martin-Martin and Li Fei-Fei and Chelsea Finn},
  year = {2021},
  eprint = {2103.04174},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2103.04174},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{saxena2021clockwork,
  title = {{Clockwork Variational Autoencoders}},
  author = {Vaibhav Saxena and Jimmy Ba and Danijar Hafner},
  year = {2021},
  eprint = {2102.09532},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2102.09532},
  code_status = {official-code},
  github = {https://github.com/vaibhavsaxena11/cwvae},
  github_stars = {50},
  github_stars_as_of = {2026-08-30},
  github_note = {第一作者发布的训练与评测实现}
}

@misc{daniel2026latent,
  title = {{Latent Particle World Models: Self-supervised Object-centric Stochastic Dynamics Modeling}},
  author = {Tal Daniel and Carl Qi and Dan Haramati and Amir Zadeh and Chuan Li and Aviv Tamar and Deepak Pathak and David Held},
  year = {2026},
  eprint = {2603.04553},
  archiveprefix = {arXiv},
  primaryclass = {cs.LG},
  url = {https://arxiv.org/abs/2603.04553},
  code_status = {official-code},
  github = {https://github.com/taldatech/lpwm},
  github_stars = {133},
  github_stars_as_of = {2026-08-30},
  github_note = {ICLR 2026 Oral 官方代码、数据与 checkpoints}
}

@misc{wang2024customvideo,
  title = {{CustomVideo: Customizing Text-to-Video Generation with Multiple Subjects}},
  author = {Zhao Wang and Aoxue Li and Lingting Zhu and Yong Guo and Qi Dou and Zhenguo Li},
  year = {2024},
  eprint = {2401.09962},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2401.09962},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{ma2024magicme,
  title = {{Magic-Me: Identity-Specific Video Customized Diffusion}},
  author = {Ze Ma and Daquan Zhou and Chun-Hsiao Yeh and Xue-She Wang and Xiuyu Li and Huanrui Yang and Zhen Dong and Kurt Keutzer and Jiashi Feng},
  year = {2024},
  eprint = {2402.09368},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2402.09368},
  code_status = {official-code},
  github = {https://github.com/Zhen-Dong/Magic-Me},
  github_stars = {458},
  github_stars_as_of = {2026-08-30},
  github_note = {作者官方 VCD 训练与推理实现；依赖第三方 SD/AnimateDiff 权重，并提供若干身份嵌入；未发布训练数据集}
}

@misc{li2024personalvideo,
  title = {{PersonalVideo: High ID-Fidelity Video Customization without Dynamic and Semantic Degradation}},
  author = {Hengjia Li and Haonan Qiu and Shiwei Zhang and Xiang Wang and Yujie Wei and Zekun Li and Yingya Zhang and Boxi Wu and Deng Cai},
  year = {2024},
  eprint = {2411.17048},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2411.17048},
  code_status = {official-project-page},
  github = {https://github.com/EchoPluto/PersonalVideo},
  github_stars = {9},
  github_stars_as_of = {2026-08-30},
  github_note = {作者占位仓；README 的 Code 仍为 Coming Soon，未发布实现、权重或数据}
}

@misc{chen2025videoalchemist,
  title = {{Multi-subject Open-set Personalization in Video Generation}},
  author = {Tsai-Shien Chen and Aliaksandr Siarohin and Willi Menapace and Yuwei Fang and Kwot Sin Lee and Ivan Skorokhodov and Kfir Aberman and Jun-Yan Zhu and Ming-Hsuan Yang and Sergey Tulyakov},
  year = {2025},
  eprint = {2501.06187},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2501.06187},
  code_status = {official-research-artifact},
  github = {https://github.com/snap-research/MSRVTT-Personalization},
  github_stars = {53},
  github_stars_as_of = {2026-08-30},
  github_note = {官方 MSRVTT-Personalization 测试集与评测协议；不含 Video Alchemist 模型实现或权重}
}

@misc{liang2025movie,
  title = {{Movie Weaver: Tuning-Free Multi-Concept Video Personalization with Anchored Prompts}},
  author = {Feng Liang and Haoyu Ma and Zecheng He and Tingbo Hou and Ji Hou and Kunpeng Li and Xiaoliang Dai and Felix Juefei-Xu and Samaneh Azadi and Animesh Sinha and Peizhao Zhang and Peter Vajda and Diana Marculescu},
  year = {2025},
  eprint = {2502.07802},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2502.07802},
  code_status = {no-official-repository},
  github = {not-available},
  github_stars = {not-applicable}
}

@misc{deng2025magref,
  title = {{MAGREF: Masked Guidance for Any-Reference Video Generation with Subject Disentanglement}},
  author = {Yufan Deng and Yuanyang Yin and Xun Guo and Yizhi Wang and Jacob Zhiyuan Fang and Shenghai Yuan and Yiding Yang and Angtian Wang and Bo Liu and Haibin Huang and Chongyang Ma},
  year = {2025},
  eprint = {2505.23742},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2505.23742},
  code_status = {official-code},
  github = {https://github.com/MAGREF-Video/MAGREF},
  github_stars = {299},
  github_stars_as_of = {2026-08-30},
  github_note = {官方推理代码与 MAGREF checkpoint；480P/14B Pro checkpoint 和训练代码仍在 TODO，训练数据未发布}
}

@misc{girish2025alchemint,
  title = {{AlcheMinT: Fine-grained Temporal Control for Multi-Reference Consistent Video Generation}},
  author = {Sharath Girish and Viacheslav Ivanov and Tsai-Shien Chen and Hao Chen and Aliaksandr Siarohin and Sergey Tulyakov},
  year = {2025},
  eprint = {2512.10943},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2512.10943},
  code_status = {official-project-page},
  github = {https://github.com/snap-research/Video-AlcheMinT},
  github_stars = {0},
  github_stars_as_of = {2026-08-30},
  github_note = {官方 README/项目占位仓；未发布方法代码、权重或 benchmark 数据}
}

@misc{huang2026rethinking,
  title = {{Rethinking Position Embedding as a Context Controller for Multi-Reference and Multi-Shot Video Generation}},
  author = {Binyuan Huang and Yuning Lu and Weinan Jia and Hualiang Wang and Mu Liu and Daiqing Yang},
  year = {2026},
  eprint = {2604.03738},
  archiveprefix = {arXiv},
  primaryclass = {cs.CV},
  url = {https://arxiv.org/abs/2604.03738},
  code_status = {official-project-page},
  github = {https://github.com/byhuang123/PoCo},
  github_stars = {20},
  github_stars_as_of = {2026-08-30},
  github_note = {官方占位与演示仓；Release Plan 中 Core model code 尚未发布，无权重或数据}
}
