VideoChat-R1
[NIPS2025] VideoChat-R1 & R1.5: Enhancing Spatio-Temporal Perception and Reasoning via Reinforcement Fine-Tuning
File Explorer
Download Latest Version (.zip)- test.json
- train.json
- val.json
- val_2.json
- charades_test.json
- train.json
- val.json
- got_train.json
- got_val.json
- nextgqa_test.json
- nextgqa_val.json
- Quality_Access_100shot.json
- Quality_Access_16shot.json
- Quality_Access_4shot.json
- Quality_Access_test.json
- ddp.yaml
- zero2.yaml
- zero3.yaml
- commands.md
- current_tasks.md
- model_guide.md
- README.md
- run_examples.md
- task_guide.md
- action_antonym.json
- action_count.json
- action_localization.json
- action_prediction.json
- action_sequence.json
- character_order.json
- counterfactual_inference.json
- egocentric_navigation.json
- episodic_reasoning.json
- fine_grained_action.json
- fine_grained_pose.json
- moving_attribute.json
- moving_count.json
- moving_direction.json
- object_existence.json
- object_interaction.json
- object_shuffle.json
- scene_transition.json
- state_change.json
- unexpected_action.json
- README.md
- test-00000-of-00001.parquet
- README.md
- __init__.py
- filter.py
- instance.py
- metrics.py
- model.py
- registry.py
- samplers.py
- task.py
- __init__.py
- decontamination.py
- extraction.py
- selection.py
- transformation.py
- qwen_generate_utils.py
- __init__.py
- load_video.py
- my_qwen_utils.py
- __init__.py
- configuration_mplug_owl.py
- modeling_mplug_owl.py
- processing_mplug_owl.py
- tokenization_mplug_owl.py
- __init__.py
- model_utils.py
- __init__.py
- consolidate.py
- make_delta.py
- utils.py
- video_chatgpt.py
- __init__.py
- constants.py
- inference.py
- single_video_inference.py
- utils.py
- video_conversation.py
- __init__.py
- qwen2_5_vl_lxh.py
- qwen_vl.py
- file_utils.py
- gpt_eval_utils.py
- video_loader.py
- vqa_eval_metric.py
- _default_template.yaml
- dream_1k.yaml
- dream_1k_cn.yaml
- utils.py
- _default_template.yaml
- mvbench_action_antonym_nothink.yaml
- mvbench_action_count_nothink.yaml
- mvbench_action_localization_nothink.yaml
- mvbench_action_prediction_nothink.yaml
- mvbench_action_sequence_nothink.yaml
- mvbench_character_order_nothink.yaml
- mvbench_counterfactual_inference_nothink.yaml
- mvbench_egocentric_navigation_nothink.yaml
- mvbench_episodic_reasoning_nothink.yaml
- mvbench_fine_grained_action_nothink.yaml
- mvbench_fine_grained_pose_nothink.yaml
- mvbench_moving_attribute_nothink.yaml
- mvbench_moving_count_nothink.yaml
- mvbench_moving_direction_nothink.yaml
- mvbench_nothink.yaml
- mvbench_object_existence_nothink.yaml
- mvbench_object_interaction_nothink.yaml
- mvbench_object_shuffle_nothink.yaml
- mvbench_scene_transition_nothink.yaml
- mvbench_state_change_nothink.yaml
- mvbench_unexpected_action_nothink.yaml
- utils.py
- _default_template.yaml
- mvbench_action_antonym_think.yaml
- mvbench_action_count_think.yaml
- mvbench_action_localization_think.yaml
- mvbench_action_prediction_think.yaml
- mvbench_action_sequence_think.yaml
- mvbench_character_order_think.yaml
- mvbench_counterfactual_inference_think.yaml
- mvbench_egocentric_navigation_think.yaml
- mvbench_episodic_reasoning_think.yaml
- mvbench_fine_grained_action_think.yaml
- mvbench_fine_grained_pose_think.yaml
- mvbench_moving_attribute_think.yaml
- mvbench_moving_count_think.yaml
- mvbench_moving_direction_think.yaml
- mvbench_object_existence_think.yaml
- mvbench_object_interaction_think.yaml
- mvbench_object_shuffle_think.yaml
- mvbench_scene_transition_think.yaml
- mvbench_state_change_think.yaml
- mvbench_think.yaml
- mvbench_unexpected_action_think.yaml
- utils.py
- _default_template_yaml
- perceptiontest_mc_nothink.yaml
- perceptiontest_mc_think.yaml
- utils.py
- utils.py
- videomme_short_nothink.yaml
- videomme_short_nothink_glue.yaml
- videomme_short_think.yaml
- videomme_short_think_glue.yaml
- __init__.py
- __init__.py
- __main__.py
- evaluator.py
- logging_utils.py
- utils.py
- example_eval.yaml
- llava_repr_requirements.txt
- llava_result_check.md
- llava_sglang_result_check.md
- repr_scripts.sh
- repr_torch_envs.txt
- scienceqa_id.txt
- script.sh
- test_llava.py
- test_scienceqa.py
- tinyllava_repr_requirements.txt
- tinyllava_repr_scripts.sh
- eval_qwen2_5vl_all_tasks_new.sh
- eval_qwen2_5vl_mv_ptest.sh
- eval_qwen2_5vl_nothink.sh
- eval_qwen2_5vl_nothink_glue.sh
- eval_qwen2_5vl_vmme_short.sh
- __init__.py
- BaseEmbedder.py
- ClipBgeEmbedder.py
- __init__.py
- kcenter_greedy.py
- sampling_def.py
- __init__.py
- BaseShrinker.py
- EmbedShrinker.py
- embed.py
- shrink.py
- live_bench.py
- example_output.json
- example_website.png
- __init__.py
- claude.py
- extract_infomation.py
- gemini.py
- gpt4v.py
- __init__.py
- check_prompt.md
- default_criteria.md
- live_bench.py
- live_bench_data.py
- prompt.md
- qa_generator.py
- question_finalizer.py
- response.py
- score_getter.py
- score_prompt.md
- .gitignore
- __init__.py
- load_driver.py
- __init__.py
- screen.py
- screen_shoter.py
- __init__.py
- load_website.py
- website.py
- website_list.yaml
- __init__.py
- view.ipynb
- modify.ipynb
- README.md
- upload_results.py
- create_dataset.py
- data_summary.ipynb
- example.ipynb
- filter.ipynb
- pyproject.toml
- refine_all_results.py
- setup.py
- get_video_avg_time.py
- make_image_hf_dataset.ipynb
- make_vatex.py
- make_video_hf_dataset.ipynb
- makecvrr.ipynb
- .gitignore
- .pre-commit-config.yaml
- LICENSE
- pyproject.toml
- README.md
- setup.py
- __init__.py
- grpo_tasks_trainer.py
- grpo_trainer.py
- grpo_trainer_video_cls.py
- grpo_trainer_video_cls_nothink.py
- grpo_trainer_video_gqa.py
- grpo_trainer_video_gqa_nothink.py
- grpo_trainer_video_qa.py
- grpo_trainer_video_qa_nothink.py
- grpo_trainer_video_tg.py
- vllm_grpo_trainer.py
- vllm_grpo_trainer_video_tg.py
- yza_vision_process.py
- __init__.py
- evaluate.py
- generate.py
- grpo.py
- grpo_cls.py
- grpo_cls_nothink.py
- grpo_gqa.py
- grpo_gqa_nothink.py
- grpo_qa.py
- grpo_qa_nothink.py
- grpo_tasks.py
- grpo_tg.py
- grpo_video.py
- my_qwen_utils.py
- my_qwen_utils2.py
- sft_cls.py
- sft_gqa.py
- sft_grounding.py
- sft_track.py
- __init__.py
- __init__.py
- data_config.py
- eval_prompts.py
- evaluate_cls_quality_c8.py
- evaluate_gqa.py
- evaluate_grounding.py
- evaluate_qa.py
- evaluate_track.py
- my_qwen_utils.py
- run_grpo_video_cls_qa.sh
- run_grpo_video_gqa.sh
- run_grpo_video_gqa_nothink_3e.sh
- run_grpo_video_qa.sh
- run_grpo_video_qa_nothink.sh
- run_grpo_video_task.sh
- run_grpo_video_tg.sh
- run_sft_video_cls_qa.sh
- run_sft_video_gqa.sh
- run_sft_video_track.sh
- zero3_offload.json
- grpo_tasks.sh
- zero2_offload.json
- zero3.json
- zero3.yaml
- zero3_offload.json
- __init__.py
- grpo_gqa_tasks.py
- grpo_muti_gqa.py
- grpo_think.py
- grpo_trainer.py
- grpo_trainer_gqa_tg.py
- grpo_trainer_qa.py
- grpo_trainer_tasks.py
- grpo_trainer_tracking.py
- grpo_trainer_video.py
- grpo_trainer_video_gqa.py
- vllm_grpo_trainer.py
- vllm_grpo_trainer_video.py
- __init__.py
- dist_utils.py
- grpo_tasks.py
- prompts.py
- reward_funcs.py
- lvbench_clean.json
- lvbench_clean_cartoon.json
- lvbench_clean_documentary.json
- lvbench_clean_live.json
- lvbench_clean_selfmedia.json
- lvbench_clean_sport.json
- lvbench_clean_tv.json
- 1_plotQA.json
- 2_needle.json
- 3_ego.json
- 4_count.json
- 5_order.json
- 6_anomaly_reco.json
- 7_topic_reasoning.json
- action_antonym.json
- action_count.json
- action_localization.json
- action_prediction.json
- action_sequence.json
- character_order.json
- counterfactual_inference.json
- egocentric_navigation.json
- episodic_reasoning.json
- fine_grained_action.json
- fine_grained_pose.json
- moving_attribute.json
- moving_count.json
- moving_direction.json
- object_existence.json
- object_interaction.json
- object_shuffle.json
- scene_transition.json
- state_change.json
- unexpected_action.json
- Adaptation.json
- Comprehension.json
- Perception.json
- lvb_val.json
- videomme.json
- vsibench.json
- __init__.py
- eval_longvideo.py
- eval_lvbench.py
- eval_mlvu.py
- eval_mvbench.py
- eval_videomme.py
- eval_videommmu.py
- eval_vsi.py
- evaluate_anet.py
- evaluate_cgbench.py
- evaluate_chara.py
- evaluate_det.py
- evaluate_gqa.py
- evaluate_qvh.py
- evaluate_rextime.py
- evaluate_tracking.py
- my_vision_process.py
- framework.png
- perception.jpg
- perception.png
- README.md
- requirements.txt
- sotas.png
# Installation Guide
1. Get the code
git clone https://github.com/OpenGVLab/VideoChat-R1
Downloads the entire project code from GitHub to your computer.
cd VideoChat-R1
Moves into the project folder you just downloaded.
2. Python
Easy RecommendedPrerequisites
pip install -r requirements.txt
Installs the Python libraries listed in requirements.txt (or similar).
jupyter notebook
Launches Jupyter in your browser so you can open and run the notebook (.ipynb) files.
If it runs without errors and prints output in the terminal, it worked.
// repository documentation
Was this content helpful?
(0 ratings)
