π Tech Report Β |Β π» GitHub Β
We introduce MOSS-Video-Preview-SFT, the offline supervised fine-tuned checkpoint in the MOSS-Video-Preview series.
[!Important] This is an offline SFT checkpoint (instruction-tuned). It is not the Real-Time SFT streaming checkpoint.
This checkpoint is intended for:
MOSS-Video-Preview is built on a Llama-3.2-Vision backbone, featuring a Pioneering Image-Video Unified Cross-Attention Architecture:
For architecture diagrams and full system details, see the top-level repository: OpenMOSS/MOSS-Video-Preview.
import torch
from transformers import AutoModelForCausalLM, AutoProcessor
# Use Hugging Face model id (or load from a local folder with the same name).
checkpoint = "OpenMOSS-Team/moss-video-preview-sft"
video_path = "data/example_video.mp4"
prompt = "Describe the video."
processor = AutoProcessor.from_pretrained(
checkpoint,
trust_remote_code=True,
frame_extract_num_threads=1,
)
model = AutoModelForCausalLM.from_pretrained(
checkpoint,
trust_remote_code=True,
device_map="auto",
torch_dtype=torch.bfloat16,
attn_implementation="flash_attention_2",
)
messages = [
{
"role": "user",
"content": [
{"type": "video"},
{"type": "text", "text": prompt},
],
}
]
input_text = processor.apply_chat_template(messages, add_generation_prompt=True)
inputs = processor(
text=input_text,
videos=[video_path],
video_fps=1.0,
video_minlen=8,
video_maxlen=16,
add_special_tokens=False,
return_tensors="pt",
).to(model.device)
with torch.no_grad():
output_ids = model.generate(**inputs, max_new_tokens=512, do_sample=False)
print(processor.decode(output_ids[0], skip_special_tokens=True))
import torch
from PIL import Image
from transformers import AutoModelForCausalLM, AutoProcessor
checkpoint = "OpenMOSS-Team/moss-video-preview-sft"
image_path = "data/example_image.jpg"
prompt = "Describe this image."
image = Image.open(image_path).convert("RGB")
processor = AutoProcessor.from_pretrained(checkpoint, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(
checkpoint,
trust_remote_code=True,
device_map="auto",
torch_dtype=torch.bfloat16,
attn_implementation="flash_attention_2",
)
messages = [
{
"role": "user",
"content": [
{"type": "image"},
{"type": "text", "text": prompt},
],
}
]
input_text = processor.apply_chat_template(messages, add_generation_prompt=True)
inputs = processor(
text=input_text,
images=[image],
add_special_tokens=False,
return_tensors="pt",
).to(model.device)
with torch.no_grad():
output_ids = model.generate(**inputs, max_new_tokens=256, do_sample=False)
print(processor.decode(output_ids[0], skip_special_tokens=True))
trust_remote_code=True for this model family (due to auto_map custom code)attn_implementation="flash_attention_2")cv2); offline demo relies on the processor's video loading backendFor full environment setup (including optional FlashAttention2 extras), see the top-level repository README.md.
[!IMPORTANT]
π Our Mission & Community Invitation
We have filled the gap in cross-attention-based foundation models for video understanding.
We warmly welcome experts in Representation Learning and Model Efficiency to explore, experiment, and innovate on top of our architecture. Let's push the boundaries of video intelligence and advance the open-source community together!
@article{wang2026mossvideo,
title = {{MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention}},
author = {Pengyu Wang, Chenkun Tan, Shaojun Zhou, Wei Huang, Qirui Zhou, Zhan Huang, Zhen Ye, Jijun Cheng, Xiaomeng Qian, Yanxin Chen, Xingyang He, Huazheng Zeng, Chenghao Wang, Pengfei Wang, Hongkai Wang, Shanqing Gao, Yixian Tian, Chenghao Liu, Xinghao Wang, Botian Jiang, Xipeng Qiu},
year = {2026},
journal = {arXiv preprint arXiv:2606.07639},
eprint = {2606.07639},
archivePrefix = {arXiv},
primaryClass = {cs.CV},
url = {https://arxiv.org/abs/2606.07639}
}
π Tech Report Β |Β π» GitHub Β
We introduce MOSS-Video-Preview-SFT, the offline supervised fine-tuned checkpoint in the MOSS-Video-Preview series.
[!Important] This is an offline SFT checkpoint (instruction-tuned). It is not the Real-Time SFT streaming checkpoint.
This checkpoint is intended for:
MOSS-Video-Preview is built on a Llama-3.2-Vision backbone, featuring a Pioneering Image-Video Unified Cross-Attention Architecture:
For architecture diagrams and full system details, see the top-level repository: OpenMOSS/MOSS-Video-Preview.
import torch
from transformers import AutoModelForCausalLM, AutoProcessor
# Use Hugging Face model id (or load from a local folder with the same name).
checkpoint = "OpenMOSS-Team/moss-video-preview-sft"
video_path = "data/example_video.mp4"
prompt = "Describe the video."
processor = AutoProcessor.from_pretrained(
checkpoint,
trust_remote_code=True,
frame_extract_num_threads=1,
)
model = AutoModelForCausalLM.from_pretrained(
checkpoint,
trust_remote_code=True,
device_map="auto",
torch_dtype=torch.bfloat16,
attn_implementation="flash_attention_2",
)
messages = [
{
"role": "user",
"content": [
{"type": "video"},
{"type": "text", "text": prompt},
],
}
]
input_text = processor.apply_chat_template(messages, add_generation_prompt=True)
inputs = processor(
text=input_text,
videos=[video_path],
video_fps=1.0,
video_minlen=8,
video_maxlen=16,
add_special_tokens=False,
return_tensors="pt",
).to(model.device)
with torch.no_grad():
output_ids = model.generate(**inputs, max_new_tokens=512, do_sample=False)
print(processor.decode(output_ids[0], skip_special_tokens=True))
import torch
from PIL import Image
from transformers import AutoModelForCausalLM, AutoProcessor
checkpoint = "OpenMOSS-Team/moss-video-preview-sft"
image_path = "data/example_image.jpg"
prompt = "Describe this image."
image = Image.open(image_path).convert("RGB")
processor = AutoProcessor.from_pretrained(checkpoint, trust_remote_code=True)
model = AutoModelForCausalLM.from_pretrained(
checkpoint,
trust_remote_code=True,
device_map="auto",
torch_dtype=torch.bfloat16,
attn_implementation="flash_attention_2",
)
messages = [
{
"role": "user",
"content": [
{"type": "image"},
{"type": "text", "text": prompt},
],
}
]
input_text = processor.apply_chat_template(messages, add_generation_prompt=True)
inputs = processor(
text=input_text,
images=[image],
add_special_tokens=False,
return_tensors="pt",
).to(model.device)
with torch.no_grad():
output_ids = model.generate(**inputs, max_new_tokens=256, do_sample=False)
print(processor.decode(output_ids[0], skip_special_tokens=True))
trust_remote_code=True for this model family (due to auto_map custom code)attn_implementation="flash_attention_2")cv2); offline demo relies on the processor's video loading backendFor full environment setup (including optional FlashAttention2 extras), see the top-level repository README.md.
[!IMPORTANT]
π Our Mission & Community Invitation
We have filled the gap in cross-attention-based foundation models for video understanding.
We warmly welcome experts in Representation Learning and Model Efficiency to explore, experiment, and innovate on top of our architecture. Let's push the boundaries of video intelligence and advance the open-source community together!
@article{wang2026mossvideo,
title = {{MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention}},
author = {Pengyu Wang, Chenkun Tan, Shaojun Zhou, Wei Huang, Qirui Zhou, Zhan Huang, Zhen Ye, Jijun Cheng, Xiaomeng Qian, Yanxin Chen, Xingyang He, Huazheng Zeng, Chenghao Wang, Pengfei Wang, Hongkai Wang, Shanqing Gao, Yixian Tian, Chenghao Liu, Xinghao Wang, Botian Jiang, Xipeng Qiu},
year = {2026},
journal = {arXiv preprint arXiv:2606.07639},
eprint = {2606.07639},
archivePrefix = {arXiv},
primaryClass = {cs.CV},
url = {https://arxiv.org/abs/2606.07639}
}