VNext
Next-generation Video instance recognition framework on top of Detectron2 which supports InstMove (CVPR 2023), SeqFormer(ECCV Oral), and IDOL(ECCV Oral))
File Explorer
Download Latest Version (.zip)- arch.png
- ovis_results.png
- vid_116.gif
- vid_2.gif
- vid_61.gif
- vid_96.gif
- ytvis2019_results.png
- ytvis2021_results.png
- SeqFormer_arch.png
- SeqFormer_sota.png
- vid_133.gif
- vid_15.gif
- vid_210.gif
- vid_78.gif
- ytvis2019_results.png
- ytvis2021_results.png
- VNext.png
- mask_rcnn_R_50_FPN.yaml
- fast_rcnn_R_50_FPN_1x.yaml
- faster_rcnn_R_101_C4_3x.yaml
- faster_rcnn_R_101_DC5_3x.yaml
- faster_rcnn_R_101_FPN_3x.yaml
- faster_rcnn_R_50_C4_1x.yaml
- faster_rcnn_R_50_C4_3x.yaml
- faster_rcnn_R_50_DC5_1x.yaml
- faster_rcnn_R_50_DC5_3x.yaml
- faster_rcnn_R_50_FPN_1x.yaml
- faster_rcnn_R_50_FPN_3x.yaml
- faster_rcnn_X_101_32x8d_FPN_3x.yaml
- fcos_R_50_FPN_1x.py
- retinanet_R_101_FPN_3x.yaml
- retinanet_R_50_FPN_1x.py
- retinanet_R_50_FPN_1x.yaml
- retinanet_R_50_FPN_3x.yaml
- rpn_R_50_C4_1x.yaml
- rpn_R_50_FPN_1x.yaml
- mask_rcnn_R_101_C4_3x.yaml
- mask_rcnn_R_101_DC5_3x.yaml
- mask_rcnn_R_101_FPN_3x.yaml
- mask_rcnn_R_50_C4_1x.py
- mask_rcnn_R_50_C4_1x.yaml
- mask_rcnn_R_50_C4_3x.yaml
- mask_rcnn_R_50_DC5_1x.yaml
- mask_rcnn_R_50_DC5_3x.yaml
- mask_rcnn_R_50_FPN_1x.py
- mask_rcnn_R_50_FPN_1x.yaml
- mask_rcnn_R_50_FPN_1x_giou.yaml
- mask_rcnn_R_50_FPN_3x.yaml
- mask_rcnn_regnetx_4gf_dds_fpn_1x.py
- mask_rcnn_regnety_4gf_dds_fpn_1x.py
- mask_rcnn_X_101_32x8d_FPN_3x.yaml
- Base-Keypoint-RCNN-FPN.yaml
- keypoint_rcnn_R_101_FPN_3x.yaml
- keypoint_rcnn_R_50_FPN_1x.py
- keypoint_rcnn_R_50_FPN_1x.yaml
- keypoint_rcnn_R_50_FPN_3x.yaml
- keypoint_rcnn_X_101_32x8d_FPN_3x.yaml
- Base-Panoptic-FPN.yaml
- panoptic_fpn_R_101_3x.yaml
- panoptic_fpn_R_50_1x.py
- panoptic_fpn_R_50_1x.yaml
- panoptic_fpn_R_50_3x.yaml
- coco.py
- coco_keypoint.py
- coco_panoptic_separated.py
- cascade_rcnn.py
- fcos.py
- keypoint_rcnn_fpn.py
- mask_rcnn_c4.py
- mask_rcnn_fpn.py
- panoptic_fpn.py
- retinanet.py
- coco_schedule.py
- optim.py
- README.md
- train.py
- faster_rcnn_R_50_FPN_noaug_1x.yaml
- keypoint_rcnn_R_50_FPN_1x.yaml
- mask_rcnn_R_50_FPN_noaug_1x.yaml
- README.md
- mask_rcnn_R_101_FPN_1x.yaml
- mask_rcnn_R_50_FPN_1x.yaml
- mask_rcnn_X_101_32x8d_FPN_1x.yaml
- mask_rcnn_R_101_FPN_1x.yaml
- mask_rcnn_R_50_FPN_1x.yaml
- mask_rcnn_X_101_32x8d_FPN_1x.yaml
- cascade_mask_rcnn_R_50_FPN_1x.yaml
- cascade_mask_rcnn_R_50_FPN_3x.yaml
- cascade_mask_rcnn_X_152_32x8d_FPN_IN5k_gn_dconv.yaml
- mask_rcnn_R_50_FPN_1x_cls_agnostic.yaml
- mask_rcnn_R_50_FPN_1x_dconv_c3-c5.yaml
- mask_rcnn_R_50_FPN_3x_dconv_c3-c5.yaml
- mask_rcnn_R_50_FPN_3x_gn.yaml
- mask_rcnn_R_50_FPN_3x_syncbn.yaml
- mmdet_mask_rcnn_R_50_FPN_1x.py
- panoptic_fpn_R_101_dconv_cascade_gn_3x.yaml
- scratch_mask_rcnn_R_50_FPN_3x_gn.yaml
- scratch_mask_rcnn_R_50_FPN_9x_gn.yaml
- scratch_mask_rcnn_R_50_FPN_9x_syncbn.yaml
- semantic_R_50_FPN_1x.yaml
- torchvision_imagenet_R_50.py
- mask_rcnn_R_101_FPN_100ep_LSJ.py
- mask_rcnn_R_101_FPN_200ep_LSJ.py
- mask_rcnn_R_101_FPN_400ep_LSJ.py
- mask_rcnn_R_50_FPN_100ep_LSJ.py
- mask_rcnn_R_50_FPN_200ep_LSJ.py
- mask_rcnn_R_50_FPN_400ep_LSJ.py
- mask_rcnn_R_50_FPN_50ep_LSJ.py
- mask_rcnn_regnetx_4gf_dds_FPN_100ep_LSJ.py
- mask_rcnn_regnetx_4gf_dds_FPN_200ep_LSJ.py
- mask_rcnn_regnetx_4gf_dds_FPN_400ep_LSJ.py
- mask_rcnn_regnety_4gf_dds_FPN_100ep_LSJ.py
- mask_rcnn_regnety_4gf_dds_FPN_200ep_LSJ.py
- mask_rcnn_regnety_4gf_dds_FPN_400ep_LSJ.py
- faster_rcnn_R_50_C4.yaml
- faster_rcnn_R_50_FPN.yaml
- cascade_mask_rcnn_R_50_FPN_inference_acc_test.yaml
- cascade_mask_rcnn_R_50_FPN_instant_test.yaml
- fast_rcnn_R_50_FPN_inference_acc_test.yaml
- fast_rcnn_R_50_FPN_instant_test.yaml
- keypoint_rcnn_R_50_FPN_inference_acc_test.yaml
- keypoint_rcnn_R_50_FPN_instant_test.yaml
- keypoint_rcnn_R_50_FPN_normalized_training_acc_test.yaml
- keypoint_rcnn_R_50_FPN_training_acc_test.yaml
- mask_rcnn_R_50_C4_GCV_instant_test.yaml
- mask_rcnn_R_50_C4_inference_acc_test.yaml
- mask_rcnn_R_50_C4_instant_test.yaml
- mask_rcnn_R_50_C4_training_acc_test.yaml
- mask_rcnn_R_50_DC5_inference_acc_test.yaml
- mask_rcnn_R_50_FPN_inference_acc_test.yaml
- mask_rcnn_R_50_FPN_instant_test.yaml
- mask_rcnn_R_50_FPN_pred_boxes_training_acc_test.yaml
- mask_rcnn_R_50_FPN_training_acc_test.yaml
- panoptic_fpn_R_50_inference_acc_test.yaml
- panoptic_fpn_R_50_instant_test.yaml
- panoptic_fpn_R_50_training_acc_test.yaml
- README.md
- retinanet_R_50_FPN_inference_acc_test.yaml
- retinanet_R_50_FPN_instant_test.yaml
- rpn_R_50_FPN_inference_acc_test.yaml
- rpn_R_50_FPN_instant_test.yaml
- semantic_R_50_FPN_inference_acc_test.yaml
- semantic_R_50_FPN_instant_test.yaml
- semantic_R_50_FPN_training_acc_test.yaml
- Base-RCNN-C4.yaml
- Base-RCNN-DilatedC5.yaml
- Base-RCNN-FPN.yaml
- Base-RetinaNet.yaml
- prepare_ade20k_sem_seg.py
- prepare_cocofied_lvis.py
- prepare_for_tests.sh
- prepare_panoptic_fpn.py
- README.md
- demo.py
- predictor.py
- README.md
- __init__.py
- c2_model_loading.py
- catalog.py
- detection_checkpoint.py
- __init__.py
- compat.py
- config.py
- defaults.py
- instantiate.py
- lazy.py
- __init__.py
- builtin.py
- builtin_meta.py
- cityscapes.py
- cityscapes_panoptic.py
- coco.py
- coco_panoptic.py
- lvis.py
- lvis_v0_5_categories.py
- lvis_v1_categories.py
- pascal_voc.py
- README.md
- register_coco.py
- __init__.py
- distributed_sampler.py
- grouped_batch_sampler.py
- __init__.py
- augmentation.py
- augmentation_impl.py
- transform.py
- __init__.py
- benchmark.py
- build.py
- catalog.py
- common.py
- dataset_mapper.py
- detection_utils.py
- __init__.py
- defaults.py
- hooks.py
- launch.py
- train_loop.py
- __init__.py
- cityscapes_evaluation.py
- coco_evaluation.py
- evaluator.py
- fast_eval_api.py
- lvis_evaluation.py
- panoptic_evaluation.py
- pascal_voc_evaluation.py
- rotated_coco_evaluation.py
- sem_seg_evaluation.py
- testing.py
- __init__.py
- api.py
- c10.py
- caffe2_export.py
- caffe2_inference.py
- caffe2_modeling.py
- caffe2_patch.py
- flatten.py
- README.md
- shared.py
- torchscript.py
- torchscript_patch.py
- box_iou_rotated.h
- box_iou_rotated_cpu.cpp
- box_iou_rotated_cuda.cu
- box_iou_rotated_utils.h
- cocoeval.cpp
- cocoeval.h
- deform_conv.h
- deform_conv_cuda.cu
- deform_conv_cuda_kernel.cu
- nms_rotated.h
- nms_rotated_cpu.cpp
- nms_rotated_cuda.cu
- ROIAlignRotated.h
- ROIAlignRotated_cpu.cpp
- ROIAlignRotated_cuda.cu
- cuda_version.cu
- README.md
- vision.cpp
- __init__.py
- aspp.py
- batch_norm.py
- blocks.py
- deform_conv.py
- losses.py
- mask_ops.py
- nms.py
- roi_align.py
- roi_align_rotated.py
- rotated_boxes.py
- shape_spec.py
- wrappers.py
- __init__.py
- configs
- model_zoo.py
- __init__.py
- backbone.py
- build.py
- fpn.py
- regnet.py
- resnet.py
- __init__.py
- build.py
- dense_detector.py
- fcos.py
- panoptic_fpn.py
- rcnn.py
- retinanet.py
- semantic_seg.py
- __init__.py
- build.py
- proposal_utils.py
- rpn.py
- rrpn.py
- __init__.py
- box_head.py
- cascade_rcnn.py
- fast_rcnn.py
- keypoint_head.py
- mask_head.py
- roi_heads.py
- rotated_fast_rcnn.py
- __init__.py
- anchor_generator.py
- box_regression.py
- matcher.py
- mmdet_wrapper.py
- poolers.py
- postprocessing.py
- sampling.py
- test_time_augmentation.py
- __init__.py
- README.md
- __init__.py
- build.py
- lr_scheduler.py
- __init__.py
- boxes.py
- image_list.py
- instances.py
- keypoints.py
- masks.py
- rotated_boxes.py
- __init__.py
- base_tracker.py
- bbox_iou_tracker.py
- hungarian_tracker.py
- iou_weighted_hungarian_bbox_iou_tracker.py
- utils.py
- vanilla_hungarian_bbox_iou_tracker.py
- __init__.py
- analysis.py
- collect_env.py
- colormap.py
- comm.py
- develop.py
- env.py
- events.py
- file_io.py
- logger.py
- memory.py
- README.md
- registry.py
- serialize.py
- testing.py
- video_visualizer.py
- visualizer.py
- __init__.py
- build_all_wheels.sh
- build_wheel.sh
- gen_install_table.py
- gen_wheel_index.sh
- pkg_helpers.bash
- README.md
- linter.sh
- parse_results.sh
- README.md
- run_inference_tests.sh
- run_instant_tests.sh
- deploy.Dockerfile
- docker-compose.yml
- Dockerfile
- README.md
- custom.css
- checkpoint.rst
- config.rst
- data.rst
- data_transforms.rst
- engine.rst
- evaluation.rst
- export.rst
- fvcore.rst
- index.rst
- layers.rst
- model_zoo.rst
- modeling.rst
- solver.rst
- structures.rst
- utils.rst
- benchmarks.md
- changelog.md
- compatibility.md
- contributing.md
- index.rst
- augmentation.md
- builtin_datasets.md
- configs.md
- data_loading.md
- datasets.md
- deployment.md
- evaluation.md
- extend.md
- getting_started.md
- index.rst
- install.md
- lazyconfigs.md
- models.md
- README.md
- training.md
- write-models.md
- .gitignore
- conf.py
- index.rst
- Makefile
- README.md
- requirements.txt
- r50_coco_sequence.yaml
- swin_coco_sequence.yaml
- ovis_r50.yaml
- ovis_swin.yaml
- ytvis19_r101.yaml
- ytvis19_r50.yaml
- ytvis19_swinL.yaml
- ytvis21_r101.yaml
- ytvis21_r50.yaml
- ytvis21_swinL.yaml
- __init__.py
- swin.py
- __init__.py
- builtin.py
- ytvis.py
- __init__.py
- augmentation.py
- build.py
- coco.py
- coco_clip.py
- coco_dataset_mapper.py
- dataset_mapper.py
- ytvis_eval.py
- __init__.py
- ms_deform_attn_func.py
- __init__.py
- ms_deform_attn.py
- ms_deform_attn_cpu.cpp
- ms_deform_attn_cpu.h
- ms_deform_attn_cuda.cu
- ms_deform_attn_cuda.h
- ms_deform_im2col_cuda.cuh
- ms_deform_attn.h
- vision.cpp
- make.sh
- setup.py
- test.py
- __init__.py
- backbone.py
- deformable_detr.py
- deformable_transformer.py
- matcher.py
- pos_neg_select.py
- position_encoding.py
- segmentation_condInst.py
- tracker.py
- __init__.py
- box_ops.py
- misc.py
- plot_utils.py
- __init__.py
- config.py
- idol.py
- IDOL.md
- train_net.py
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r1.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r10.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r5.yaml
- Base-OVIS-VideoInstanceSegmentation.yaml
- video_maskformer2_R50_bs32_8ep_frame.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r1.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r10.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r5.yaml
- Base-YouTubeVIS-VideoInstanceSegmentation.yaml
- video_maskformer2_R50_bs32_8ep_frame.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r1.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r10.yaml
- video_maskformer2_swin_large_IN21k_384_bs32_8ep_frame_r5.yaml
- Base-YouTubeVIS-VideoInstanceSegmentation.yaml
- video_maskformer2_R50_bs32_8ep_frame.yaml
- video_maskformer2_R50_bs32_8ep_frame_ytvis22.yaml
- README.md
- demo.py
- predictor.py
- visualizer.py
- __init__.py
- coco_instance_new_baseline_dataset_mapper.py
- coco_panoptic_new_baseline_dataset_mapper.py
- mask_former_instance_dataset_mapper.py
- mask_former_panoptic_dataset_mapper.py
- mask_former_semantic_dataset_mapper.py
- __init__.py
- register_ade20k_full.py
- register_ade20k_instance.py
- register_ade20k_panoptic.py
- register_coco_panoptic_annos_semseg.py
- register_coco_stuff_10k.py
- register_mapillary_vistas.py
- register_mapillary_vistas_panoptic.py
- __init__.py
- __init__.py
- instance_evaluation.py
- __init__.py
- swin.py
- __init__.py
- mask_former_head.py
- per_pixel_baseline.py
- __init__.py
- ms_deform_attn_func.py
- __init__.py
- ms_deform_attn.py
- ms_deform_attn_cpu.cpp
- ms_deform_attn_cpu.h
- ms_deform_attn_cuda.cu
- ms_deform_attn_cuda.h
- ms_deform_im2col_cuda.cuh
- ms_deform_attn.h
- vision.cpp
- make.sh
- setup.py
- test.py
- __init__.py
- fpn.py
- msdeformattn.py
- __init__.py
- mask2former_transformer_decoder.py
- maskformer_transformer_decoder.py
- position_encoding.py
- transformer.py
- __init__.py
- criterion.py
- matcher.py
- __init__.py
- misc.py
- __init__.py
- config.py
- maskformer_model.py
- test_time_augmentation.py
- __init__.py
- ytvos.py
- ytvoseval.py
- __init__.py
- builtin.py
- ytvis.py
- __init__.py
- augmentation.py
- build.py
- dataset_mapper.py
- ytvis_eval.py
- __init__.py
- position_encoding.py
- video_mask2former_transformer_decoder.py
- __init__.py
- criterion.py
- matcher.py
- __init__.py
- memory.py
- __init__.py
- config.py
- video_maskformer_model.py
- __init__.py
- ytvos.py
- ytvoseval.py
- __init__.py
- builtin.py
- ytvis.py
- __init__.py
- augmentation.py
- build.py
- dataset_mapper.py
- ytvis_eval.py
- __init__.py
- config.py
- video_mask2former_transformer_decoder.py
- video_maskformer_model.py
- __init__.py
- box_ops.py
- misc.py
- plot_utils.py
- backbone.py
- convlstm.py
- get_backbone.py
- model_withImgR6.py
- INSTALL.md
- LICENSE
- requirements.txt
- train_net_video.py
- InstMove.md
- swin_ytvis.yaml
- base_ytvis.yaml
- __init__.py
- swin.py
- __init__.py
- builtin.py
- ytvis.py
- __init__.py
- augmentation.py
- build.py
- coco.py
- coco_dataset_mapper.py
- dataset_mapper.py
- ytvis_eval.py
- __init__.py
- ms_deform_attn_func.py
- __init__.py
- ms_deform_attn.py
- ms_deform_attn_cpu.cpp
- ms_deform_attn_cpu.h
- ms_deform_attn_cuda.cu
- ms_deform_attn_cuda.h
- ms_deform_im2col_cuda.cuh
- ms_deform_attn.h
- vision.cpp
- make.sh
- setup.py
- test.py
- __init__.py
- backbone.py
- clip_output.py
- deformable_detr.py
- deformable_transformer.py
- matcher.py
- position_encoding.py
- segmentation_condInst.py
- __init__.py
- box_ops.py
- misc.py
- plot_utils.py
- __init__.py
- config.py
- seqformer.py
- SeqFormer.md
- train_net.py
- dir1_a.py
- dir1_b.py
- root_cfg.py
- test_instantiate_config.py
- test_lazy_config.py
- test_yacs_config.py
- __init__.py
- test_coco.py
- test_coco_evaluation.py
- test_dataset.py
- test_detection_utils.py
- test_rotation_transform.py
- test_sampler.py
- test_transforms.py
- __init__.py
- test_blocks.py
- test_deformable.py
- test_losses.py
- test_mask_ops.py
- test_nms.py
- test_nms_rotated.py
- test_roi_align.py
- test_roi_align_rotated.py
- __init__.py
- test_anchor_generator.py
- test_backbone.py
- test_box2box_transform.py
- test_fast_rcnn.py
- test_matcher.py
- test_mmdet.py
- test_model_e2e.py
- test_roi_heads.py
- test_roi_pooler.py
- test_rpn.py
- __init__.py
- test_boxes.py
- test_imagelist.py
- test_instances.py
- test_keypoints.py
- test_masks.py
- test_rotated_boxes.py
- __init__.py
- test_bbox_iou_tracker.py
- test_hungarian_tracker.py
- test_iou_weighted_hungarian_bbox_iou_tracker.py
- test_vanilla_hungarian_bbox_iou_tracker.py
- __init__.py
- README.md
- test_checkpoint.py
- test_engine.py
- test_events.py
- test_export_caffe2.py
- test_export_torchscript.py
- test_model_analysis.py
- test_model_zoo.py
- test_packaging.py
- test_registry.py
- test_scheduler.py
- test_solver.py
- test_visualizer.py
- CMakeLists.txt
- export_model.py
- README.md
- torchscript_mask_rcnn.cpp
- __init__.py
- analyze_model.py
- benchmark.py
- convert-pretrained-swin-model-to-d2.py
- convert-torchvision-to-d2.py
- lazyconfig_train_net.py
- lightning_train_net.py
- plain_train_net.py
- README.md
- train_net.py
- visualize_data.py
- visualize_json_results.py
- .gitignore
- INSTALL.md
- LICENSE
- README.md
- requirements.txt
- setup.cfg
- setup.py
# Installation Guide
git clone https://github.com/FoundationVision/VNext
Downloads the entire project code from GitHub to your computer.
cd VNext
Moves into the project folder you just downloaded.
2. Docker
Easy Recommended- Git Needed to download the project code from GitHub.
- Docker Desktop Needed to build and run containers. Install it and keep it running in the background.
docker compose -f docker/docker-compose.yml up -d --build
Runs the command against the services defined in the compose file.
3. CMake
Mediumcd tools/deploy
This project's files live in a subfolder, so move into it first.
mkdir build && cd build
Creates a folder to hold the build output and moves into it.
cmake ..
Analyzes the source code and generates build configuration files (must be run inside the build folder).
make
Compiles the code based on the generated build configuration to produce an executable.
4. Python
Easypip install -r projects/InstMove/MinVIS_motion/requirements.txt
Installs the Python libraries listed in requirements.txt (or similar).
python <μ€νν νμΌλͺ
>.py # READMEμμ μ νν μ€ν νμΌλͺ
μ νμΈνμΈμ
Runs the Python script (or module).
