DAMO-NLP-SG/VideoRefer-VideoLLaMA3-7B

Model

VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM

13

11 commits

3 linked in READMEs

updated Jun 19, 2025

See the code

README

VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM

If you like our project, please give us a star ⭐ on Github for the latest update.

📰 News

🌏 Model Zoo

📑 Citation

If you find VideoRefer Suite useful for your research and applications, please cite using this BibTeX:

@InProceedings{Yuan_2025_CVPR,
    author    = {Yuan, Yuqian and Zhang, Hang and Li, Wentong and Cheng, Zesen and Zhang, Boqiang and Li, Long and Li, Xin and Zhao, Deli and Zhang, Wenqiao and Zhuang, Yueting and Zhu, Jianke and Bing, Lidong},
    title     = {VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM},
    booktitle = {Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR)},
    month     = {June},
    year      = {2025},
    pages     = {18970-18980}
}

@article{damonlpsg2025videollama3,
  title={VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding},
  author={Boqiang Zhang, Kehan Li, Zesen Cheng, Zhiqiang Hu, Yuqian Yuan, Guanzheng Chen, Sicong Leng, Yuming Jiang, Hang Zhang, Xin Li, Peng Jin, Wenqi Zhang, Fan Wang, Lidong Bing, Deli Zhao},
  journal={arXiv preprint arXiv:2501.13106},
  year={2025},
  url = {https://arxiv.org/abs/2501.13106}
}

custom_code
large video-language model
multimodal large language model
safetensors
text-generation
transformers
videollama3_qwen2
video-text-to-text

DAMO-NLP-SG/VideoRefer-VideoLLaMA3-7B

Model

VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM

13

11 commits

3 linked in READMEs

updated Jun 19, 2025

See the code

README

VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM

If you like our project, please give us a star ⭐ on Github for the latest update.

📰 News

🌏 Model Zoo

📑 Citation

If you find VideoRefer Suite useful for your research and applications, please cite using this BibTeX:

@InProceedings{Yuan_2025_CVPR,
    author    = {Yuan, Yuqian and Zhang, Hang and Li, Wentong and Cheng, Zesen and Zhang, Boqiang and Li, Long and Li, Xin and Zhao, Deli and Zhang, Wenqiao and Zhuang, Yueting and Zhu, Jianke and Bing, Lidong},
    title     = {VideoRefer Suite: Advancing Spatial-Temporal Object Understanding with Video LLM},
    booktitle = {Proceedings of the Computer Vision and Pattern Recognition Conference (CVPR)},
    month     = {June},
    year      = {2025},
    pages     = {18970-18980}
}

@article{damonlpsg2025videollama3,
  title={VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding},
  author={Boqiang Zhang, Kehan Li, Zesen Cheng, Zhiqiang Hu, Yuqian Yuan, Guanzheng Chen, Sicong Leng, Yuming Jiang, Hang Zhang, Xin Li, Peng Jin, Wenqi Zhang, Fan Wang, Lidong Bing, Deli Zhao},
  journal={arXiv preprint arXiv:2501.13106},
  year={2025},
  url = {https://arxiv.org/abs/2501.13106}
}

custom_code
large video-language model
multimodal large language model
safetensors
text-generation
transformers
videollama3_qwen2
video-text-to-text