T2AV-Compass
The Source Code for T2AV-Compass @ ICML 2026
File Explorer
- citation.bib
- style.css
- arxiv-logomark-small.svg
- avbench_00.jpg
- data_stats.svg
- datapipe.jpg
- favicon.svg
- main_00.jpg
- njulinkv6.jpg
- pipeline.svg
- radar_six_panels_integrated.svg
- title-icon.png
- video_if_6groups_gradient.svg
- main.js
- 1-kling.mp4
- 1-ovi.mp4
- 1-veo.mp4
- 268-kling.mp4
- 268-ovi.mp4
- 268-veo.mp4
- 268-wan26.mp4
- 377-kling.mp4
- 377-ovi.mp4
- 377-veo.mp4
- index.html
- REPO_MAINTENANCE.md
- prompts.json
- aes_model.png
- audiomos2025-track2-dev_list.csv
- audiomos2025-track2-train_list.csv
- audiomos2025_track2_human_annotations.jsonl
- README.md
- AES_natural_music.jsonl
- AES_natural_sound.jsonl
- AES_natural_speech.jsonl
- AES_PAM.jsonl
- __init__.py
- aes.py
- utils.py
- wavlm.py
- __init__.py
- cli.py
- infer.py
- utils.py
- .gitignore
- CHANGELOG.md
- CODE_OF_CONDUCT.md
- CONTRIBUTING.md
- LICENSE
- pyproject.toml
- README.md
- finetune_nisqa.yaml
- finetune_nisqa_multidimensional.yaml
- train_nisqa_cnn_lstm_avg.yaml
- train_nisqa_cnn_sa_ap.yaml
- train_nisqa_double_ended.yaml
- NISQA_lib.py
- NISQA_model.py
- LICENSE_model_weights
- nisqa.tar
- nisqa_mos_only.tar
- nisqa_tts.tar
- .gitignore
- env.yml
- LICENSE
- README.md
- run_evaluate.py
- run_predict.py
- run_train.py
- bird_audio.wav
- bird_image.jpg
- car_audio.wav
- car_image.jpg
- dog_audio.wav
- dog_image.jpg
- bpe_simple_vocab_16e6.txt.gz
- __init__.py
- helpers.py
- imagebind_model.py
- multimodal_preprocessors.py
- transformer.py
- __init__.py
- data.py
- .gitignore
- audio_embeddings.npy
- audio_names.txt
- BATCH_INFERENCE.md
- batch_inference.py
- batch_inference_audio_text.py
- batch_inference_video_text.py
- batch_pairs_test.py
- BATCH_RUN_GUIDE.md
- BATCH_TEST_ALL_GUIDE.md
- batch_test_all_models.sh
- batch_test_all_videos.py
- BATCH_TEST_GUIDE.md
- CODE_OF_CONDUCT.md
- compare_models.py
- consistency_report.txt
- CONTRIBUTING.md
- execute_tests_and_generate_readme.sh
- EXECUTION_GUIDE.md
- generate_results_readme.py
- LICENSE
- metrics.json
- model_card.md
- PROJECT_OVERVIEW_CN.md
- QUICK_REFERENCE.sh
- README.md
- requirements.txt
- run.sh
- setup.py
- similarity_matrix.npy
- TEXT_SIMILARITY_GUIDE.md
- video_embeddings.npy
- video_names.txt
- demo1_audio.wav
- demo1_video.mp4
- demo2_audio.wav
- demo2_video.mp4
- demo3_audio.wav
- demo3_video.mp4
- syncnet_16_latent.yaml
- syncnet_16_pixel.yaml
- syncnet_16_pixel_attn.yaml
- syncnet_25_pixel.yaml
- stage1.yaml
- stage1_512.yaml
- stage2.yaml
- stage2_512.yaml
- stage2_efficient.yaml
- audio.yaml
- scheduler_config.json
- changelog_v1.5.md
- changelog_v1.6.md
- framework.png
- syncnet_arch.md
- __init__.py
- box_utils.py
- nets.py
- __init__.py
- README.md
- __init__.py
- syncnet.py
- syncnet_eval.py
- draw_syncnet_lines.py
- eval_fvd.py
- eval_sync_conf.py
- eval_sync_conf.sh
- eval_syncnet_acc.py
- eval_syncnet_acc.sh
- fvd.py
- hyper_iqa.py
- inference_videos.py
- syncnet_detect.py
- syncnet_dataset.py
- unet_dataset.py
- attention.py
- motion_module.py
- resnet.py
- stable_syncnet.py
- unet.py
- unet_blocks.py
- utils.py
- wav2lip_syncnet.py
- lipsync_pipeline.py
- __init__.py
- utils.py
- videomaev2_finetune.py
- videomaev2_pretrain.py
- __init__.py
- __init__.py
- data_utils.py
- metric_utils.py
- loss.py
- affine_transform.py
- audio.py
- av_reader.py
- face_detector.py
- image_processor.py
- mask.png
- mask2.png
- mask3.png
- mask4.png
- util.py
- merges.txt
- special_tokens_map.json
- tokenizer_config.json
- vocab.json
- added_tokens.json
- merges.txt
- special_tokens_map.json
- tokenizer_config.json
- vocab.json
- mel_filters.npz
- __init__.py
- basic.py
- english.json
- english.py
- __init__.py
- __main__.py
- audio.py
- decoding.py
- model.py
- tokenizer.py
- transcribe.py
- utils.py
- audio2feature.py
- affine_transform.py
- data_processing_pipeline.py
- detect_shot.py
- filter_high_resolution.py
- filter_visual_quality.py
- remove_broken_videos.py
- remove_incorrect_affined.py
- resample_fps_hz.py
- segment_videos.py
- sync_av.py
- inference.py
- train_syncnet.py
- train_unet.py
- count_total_videos_time.py
- download_web_videos.py
- move_files_recur.py
- occupy_gpu.py
- plot_videos_time_distribution.py
- remove_outdated_files.py
- write_fileslist.py
- .gitignore
- cog.yaml
- data_processing_pipeline.sh
- gradio_app.py
- inference.sh
- LICENSE
- predict.py
- README.md
- requirements.txt
- setup_env.sh
- train_syncnet.sh
- train_unet.sh
- ft_synchability.yaml
- segment_avclip.yaml
- sync.yaml
- audioset.py
- dataset_utils.py
- lrs.py
- transforms.py
- vggsound.py
- modeling_ast.py
- ast.py
- resnet.py
- coca_base.json
- coca_roberta-ViT-B-32.json
- coca_ViT-B-32.json
- coca_ViT-L-14.json
- convnext_base.json
- convnext_base_w.json
- convnext_base_w_320.json
- convnext_large.json
- convnext_large_d.json
- convnext_large_d_320.json
- convnext_small.json
- convnext_tiny.json
- convnext_xlarge.json
- convnext_xxlarge.json
- convnext_xxlarge_320.json
- mt5-base-ViT-B-32.json
- mt5-xl-ViT-H-14.json
- RN101-quickgelu.json
- RN101.json
- RN50-quickgelu.json
- RN50.json
- RN50x16.json
- RN50x4.json
- RN50x64.json
- roberta-ViT-B-32.json
- swin_base_patch4_window7_224.json
- ViT-B-16-plus-240.json
- ViT-B-16-plus.json
- ViT-B-16.json
- ViT-B-32-plus-256.json
- ViT-B-32-quickgelu.json
- ViT-B-32.json
- ViT-bigG-14.json
- ViT-e-14.json
- ViT-g-14.json
- ViT-H-14.json
- ViT-H-16.json
- ViT-L-14-280.json
- ViT-L-14-336.json
- ViT-L-14.json
- ViT-L-16-320.json
- ViT-L-16.json
- ViT-M-16-alt.json
- ViT-M-16.json
- ViT-M-32-alt.json
- ViT-M-32.json
- ViT-S-16-alt.json
- ViT-S-16.json
- ViT-S-32-alt.json
- ViT-S-32.json
- vit_medium_patch16_gap_256.json
- vit_relpos_medium_patch16_cls_224.json
- xlm-roberta-base-ViT-B-32.json
- xlm-roberta-large-ViT-H-14.json
- __init__.py
- bpe_simple_vocab_16e6.txt.gz
- coca_model.py
- constants.py
- factory.py
- generation_utils.py
- hf_configs.py
- hf_model.py
- loss.py
- model.py
- modified_resnet.py
- openai.py
- pretrained.py
- push_to_hf_hub.py
- timm_model.py
- tokenizer.py
- transform.py
- transformer.py
- utils.py
- version.py
- .gitignore
- __init__.py
- data.py
- distributed.py
- file_utils.py
- imagenet_zeroshot_data.py
- logger.py
- params.py
- precision.py
- profile.py
- scheduler.py
- train.py
- train_clip.py
- zero_shot.py
- __init__.py
- divided_224_16x4.yaml
- joint_224_16x4.yaml
- motionformer_224_16x4.yaml
- nystrom_helper.py
- orthoformer_helper.py
- performer_helper.py
- video_model_builder.py
- vit_helper.py
- __init__.py
- motionformer.py
- s3d.py
- bridges.py
- transformer.py
- sync_model.py
- sbatch_resume_train_segment_avclip.sh
- sbatch_resume_train_sync.sh
- sbatch_test_probe.sh
- sbatch_test_syncability.sh
- sbatch_train_segment_avclip.sh
- sbatch_train_sync.sh
- sbatch_train_syncability.sh
- test_syncability.py
- train_sync.py
- train_utils.py
- logger.py
- utils.py
- .gitignore
- batch_inference.py
- batch_test_folder.py
- batch_test_multiple_models.py
- conda_env.yml
- conda_env_for_AMD_ROCm.yml
- EVALUATION_GUIDE.md
- example.ipynb
- example.py
- LICENSE
- main.py
- README.md
- run.sh
- run_clap_scoring.py
- run_clip_scoring.py
- settings.json
- example.png
- Dockerfile
- aesthetic_predictor_v2_5.pth
- __init__.py
- siglip_v2_5.py
- .gitignore
- app.py
- LICENSE
- Makefile
- pyproject.toml
- README.md
- jekyll-gh-pages.yml
- 1724.mp4
- 17734.mp4
- e4dv6ZsFzHE_000243_000253.mp4
- error_test.mp4
- eWgy8eG3stA_000164_000174.mp4
- FEQ-gdkTN4Q_000008_000018.mp4
- HyZMiWeJyMw_000072_000082.mp4
- __init__.cpython-37.pyc
- __init__.cpython-38.pyc
- __init__.cpython-39.pyc
- __init__.cpython-37.pyc
- __init__.cpython-38.pyc
- __init__.cpython-39.pyc
- basic_datasets.cpython-37.pyc
- basic_datasets.cpython-38.pyc
- basic_datasets.cpython-39.pyc
- dover_datasets.cpython-38.pyc
- fusion_datasets.cpython-37.pyc
- fusion_datasets.cpython-38.pyc
- fusion_datasets.cpython-39.pyc
- inference_dataset.cpython-38.pyc
- __init__.py
- basic_datasets.py
- dover_datasets.py
- __init__.cpython-37.pyc
- __init__.cpython-38.pyc
- __init__.cpython-39.pyc
- backbone.cpython-38.pyc
- backbone_v0_1.cpython-38.pyc
- conv_backbone.cpython-37.pyc
- conv_backbone.cpython-38.pyc
- conv_backbone.cpython-39.pyc
- evaluator.cpython-37.pyc
- evaluator.cpython-38.pyc
- evaluator.cpython-39.pyc
- head.cpython-37.pyc
- head.cpython-38.pyc
- head.cpython-39.pyc
- swin_backbone.cpython-37.pyc
- swin_backbone.cpython-38.pyc
- swin_backbone.cpython-39.pyc
- xclip_backbone.cpython-37.pyc
- xclip_backbone.cpython-38.pyc
- __init__.py
- backbone_get_attention.py
- backbone_v0_1.py
- conv_backbone.py
- evaluator.py
- head.py
- swin_backbone.py
- xclip_backbone.py
- __init__.py
- version.py
- livevqc-checkpoint.png
- ltest-checkpoint.png
- ytugc-checkpoint.png
- kinetics_400_1.csv
- livevqc.png
- val-cvd2014.pkl
- val-kv1k.pkl
- val-l1080p.pkl
- val-livevqc.pkl
- val-ltest.pkl
- val-ytugc.pkl
- yfcc_100m_1.csv
- labels.txt
- train_labels.txt
- val_labels.txt
- test_labels.txt
- training_labels.txt
- validation_labels.txt
- labels-checkpoint.txt
- labels.txt
- labels.txt
- mp4labels.txt
- labels-checkpoint.txt
- names-checkpoint.txt
- scores-checkpoint.txt
- labels.txt
- names.txt
- scores.txt
- labels-checkpoint.txt
- labels.txt
- labels-checkpoint.txt
- labels_1080p-checkpoint.txt
- labels_test-checkpoint.txt
- labels.txt
- labels_1080p.txt
- labels_test.txt
- labels.txt
- labels.txt
- labels-checkpoint.txt
- labels.txt
- train_labels.txt
- problem_definition-checkpoint.png
- approach.png
- Fine-graind Subjective Studies Result Demo.mp4
- In-the-wild Demos.mp4
- in_the_wild_on_kinetics.png
- problem_definition.png
- README.md
- _config.yaml
- convert_to_onnx.py
- default_infer.py
- divide.yml
- dover-mobile.yml
- dover.yml
- evaluate_a_set_of_videos.py
- evaluate_one_video.py
- Generate_Divergence_Maps_and_gMAD.ipynb
- LICENSE
- onnx_inference.py
- Prepare_Video_Pairs_for_Subjective_Studies.ipynb
- README.md
- requirements.txt
- S-Lab-LICENSE
- setup.py
- training_with_divide.py
- transfer_learning.py
- batch_dover.py
- batch_eval_all.sh
- batch_lipsync.py
- batch_video_aesthetic.py
- common.sh
- eval_all_metrics.sh
- eval_audio_aesthetic.sh
- eval_audio_video_alignment.sh
- eval_av_sync.sh
- eval_lipsync.sh
- eval_speech_quality.sh
- eval_text_audio_alignment.sh
- eval_text_video_alignment.sh
- eval_video_aesthetic.sh
- eval_video_technical.sh
- extract_audio.sh
- README.md
- run_audiobox_batch.py
- AAS.md
- MSS.md
- MTC.md
- OIS.md
- TCS.md
- eval_checklist.py
- eval_realism.py
- README.md
- .gitattributes
- .gitignore
- README.md
- README_cn.md
- run_objective_batch.sh
- setup_objective.sh
# Use via CDN
jsDelivrjsDelivr serves any public GitHub repository as a CDN with zero setup. Pick a version and a file to get a ready-to-paste link and snippet.
Command Glossary
Commands referenced in this DOCs, explained below.
ffmpeg
View Details ▼
ffmpeg
Video conversion tool.
See also: `gst-launch-1.0`.
ffmpeg -i {{path/to/video.mp4}} -vn {{path/to/sound.mp3}}
Extract the sound from a video and save it as MP3:
ffmpeg -i {{path/to/input_audio.flac}} -ar 44100 -sample_fmt s16 {{path/to/output_audio.wav}}
Transcode a FLAC file to Red Book CD format (44100kHz, 16bit):
ffmpeg -i {{path/to/video.mp4}} {{[-vf|-filter:v]}} 'scale=-1:1000' -r 15 {{path/to/output.gif}}
Save a video as GIF, scaling the height to 1000px and setting framerate to 15:
git clone
View Details ▼
git clone
Clone an existing repository.
git clone {{remote_repository_location}} {{path/to/directory}}
Clone an existing repository into a new directory (the default directory is the repository name):
git clone --recursive {{remote_repository_location}}
Clone an existing repository and its submodules:
git clone {{[-n|--no-checkout]}} {{remote_repository_location}}
Clone only the `.git` directory of an existing repository:
