# Auto-generated: CVPR2025 paper notes ignore
# 由 src/sync_notes.py 根据 TODO.md 自动生成

# 默认忽略所有笔记
*.md
**/*.md

# 保留元数据文件
!TODO.md
!INDEX.md
!index.md

# 排除已完成笔记所在的子目录
!3d_vision/
!ai_safety/
!aigc_detection/
!audio_speech/
!autonomous_driving/
!causal_inference/
!computational_biology/
!earth_science/
!graph_learning/
!hallucination/
!human_understanding/
!image_generation/
!image_restoration/
!information_retrieval/
!interpretability/
!knowledge_editing/
!llm_agent/
!llm_alignment/
!llm_efficiency/
!llm_evaluation/
!llm_nlp/
!llm_pretraining/
!llm_reasoning/
!llm_safety/
!medical_imaging/
!model_compression/
!multi_agent/
!multilingual_mt/
!multimodal_vlm/
!nlp_generation/
!object_detection/
!optimization/
!others/
!physics/
!recommender/
!reinforcement_learning/
!remote_sensing/
!robotics/
!segmentation/
!self_supervised/
!signal_comm/
!social_computing/
!time_series/
!video_generation/
!video_understanding/
!vlm_efficiency/
!vlm_reasoning/

# 已完成笔记 (1820 篇)
!3d_vision/3d-gsw_3d_gaussian_splatting_for_robust_watermarking.md
!3d_vision/3d-hgs_3d_half-gaussian_splatting.md
!3d_vision/3d-llava_towards_generalist_3d_lmms_with_omni_superpoint_transformer.md
!3d_vision/3d-mem_3d_scene_memory_for_embodied_exploration_and_reasoning.md
!3d_vision/3d-slnr_a_super_lightweight_neural_representation_for_large-scale_3d_mapping.md
!3d_vision/3d_convex_splatting_radiance_field_rendering_with_3d_smooth_convexes.md
!3d_vision/3d_dental_model_segmentation_with_geometrical_boundary_preserving.md
!3d_vision/3d_gaussian_head_avatars_with_expressive_dynamic_appearances_by_compact_tensoria.md
!3d_vision/3d_gaussian_inpainting_with_depth-guided_cross-view_consistency.md
!3d_vision/3d_student_splatting_and_scooping.md
!3d_vision/3denhancer_consistent_multi-view_diffusion_for_3d_enhancement.md
!3d_vision/3dgut_enabling_distorted_cameras_and_secondary_rays_in_gaussian_splatting.md
!3d_vision/4deform_neural_surface_deformation_for_robust_shape_interpolation.md
!3d_vision/4dequine_disentangling_motion_and_appearance_for_4d_equine_reconstruction_from_m.md
!3d_vision/4dgc_rate-aware_4d_gaussian_compression_for_efficient_streamable_free-viewpoint_.md
!3d_vision/4dtam_non-rigid_tracking_and_mapping_via_dynamic_surface_gaussians.md
!3d_vision/a2z-10m_geometric_deep_learning_with_a-to-z_brep_annotations_for_ai-assisted_cad.md
!3d_vision/a_lightweight_udf_learning_framework_for_3d_reconstruction_based_on_local_shape_.md
!3d_vision/a_unified_image-dense_annotation_generation_model_for_underwater_scenes.md
!3d_vision/activegamer_active_gaussian_mapping_through_efficient_rendering.md
!3d_vision/aerialmegadepth_learning_aerial-ground_reconstruction_and_view_synthesis.md
!3d_vision/anigs_animatable_gaussian_avatar_from_a_single_image_with_inconsistent_gaussian_.md
!3d_vision/any3dis_class-agnostic_3d_instance_segmentation_by_2d_mask_tracking.md
!3d_vision/arm_appearance_reconstruction_model_for_relightable_3d_generation.md
!3d_vision/ashita_automatic_scene-grounded_hierarchical_task_analysis.md
!3d_vision/bfanet_revisiting_3d_semantic_segmentation_with_boundary_feature_analysis.md
!3d_vision/blade_single-view_body_mesh_estimation_through_accurate_depth_estimation.md
!3d_vision/blurry-edges_photon-limited_depth_estimation_from_defocused_boundaries.md
!3d_vision/cadcrafter_generating_computer-aided_design_models_from_unconstrained_images.md
!3d_vision/caddreamer_cad_object_generation_from_single-view_images.md
!3d_vision/category-agnostic_neural_object_rigging.md
!3d_vision/cmmloc_advancing_text-to-pointcloud_localization_with_cauchy-mixture-model_based.md
!3d_vision/cob-gs_clear_object_boundaries_in_3dgs_segmentation_based_on_boundary-adaptive_g.md
!3d_vision/cocogaussian_leveraging_circle_of_confusion_for_gaussian_splatting_from_defocuse.md
!3d_vision/coherent_3d_portrait_video_reconstruction_via_triplane_fusion.md
!3d_vision/colabsfm_collaborative_structure-from-motion_by_point_cloud_registration.md
!3d_vision/comapgs_covisibility_map-based_gaussian_splatting_for_sparse_novel_view_synthesi.md
!3d_vision/comatcher_multi-view_collaborative_feature_matching.md
!3d_vision/consistency-aware_self-training_for_iterative-based_stereo_matching.md
!3d_vision/continuous_3d_perception_model_with_persistent_state.md
!3d_vision/cross-view_completion_models_are_zero-shot_correspondence_estimators.md
!3d_vision/crossover_3d_scene_cross-modal_alignment.md
!3d_vision/ctrl-d_controllable_dynamic_3d_scene_editing_with_personalized_2d_diffusion.md
!3d_vision/dagsm_disentangled_avatar_generation_with_gs-enhanced_mesh.md
!3d_vision/dashgaussian_optimizing_3d_gaussian_splatting_in_200_seconds.md
!3d_vision/decompositional_neural_scene_reconstruction_with_generative_diffusion_prior.md
!3d_vision/defom-stereo_depth_foundation_model_based_stereo_matching.md
!3d_vision/deformable_radial_kernel_splatting.md
!3d_vision/denoising_functional_maps_diffusion_models_for_shape_correspondence.md
!3d_vision/dense-sfm_structure_from_motion_with_dense_consistent_matching.md
!3d_vision/depth-guided_bundle_sampling_for_efficient_generalizable_neural_radiance_field_r.md
!3d_vision/depth_any_camera_zero-shot_metric_depth_estimation_from_any_camera.md
!3d_vision/depthcrafter_generating_consistent_long_depth_sequences_for_open-world_videos.md
!3d_vision/depthcues_evaluating_monocular_depth_perception_in_large_vision_models.md
!3d_vision/depthsplat_connecting_gaussian_splatting_and_depth.md
!3d_vision/desplat_decomposed_gaussian_splatting_for_distractor-free_rendering.md
!3d_vision/diet-gs_diffusion_prior_and_event_stream-assisted_motion_deblurring_3d_gaussian_.md
!3d_vision/diffportrait360_consistent_portrait_diffusion_for_360_view_synthesis.md
!3d_vision/difix3d_improving_3d_reconstructions_with_single-step_diffusion_models.md
!3d_vision/digital_twin_catalog_a_large-scale_photorealistic_3d_object_digital_twin_dataset.md
!3d_vision/disco4d_disentangled_4d_human_generation_and_animation_from_a_single_image.md
!3d_vision/dof-gaussian_controllable_depth-of-field_for_3d_gaussian_splatting.md
!3d_vision/doppelgangers_improved_visual_disambiguation_with_geometric_3d_features.md
!3d_vision/dr_splat_directly_referring_3d_gaussian_splatting_via_direct_language_embedding_.md
!3d_vision/dronesplat_3d_gaussian_splatting_for_robust_3d_reconstruction_from_in-the-wild_d.md
!3d_vision/dropgaussian_structural_regularization_for_sparse-view_gaussian_splatting.md
!3d_vision/dropoutgs_dropping_out_gaussians_for_better_sparse-view_rendering.md
!3d_vision/dspnet_dual-vision_scene_perception_for_robust_3d_question_answering.md
!3d_vision/dual_exposure_stereo_extended_dr_3d.md
!3d_vision/dual_exposure_stereo_for_extended_dynamic_range_3d_imaging.md
!3d_vision/dualpm_dual_point_maps_shape_pose.md
!3d_vision/dualpm_dual_posed-canonical_point_maps_for_3d_shape_and_pose_reconstruction.md
!3d_vision/dune_distilling_a_universal_encoder_from_heterogeneous_2d_and_3d_teachers.md
!3d_vision/dune_universal_encoder_distillation.md
!3d_vision/dyn-hamr_recovering_4d_interacting_hand_motion_from_a_dynamic_camera.md
!3d_vision/dynamic_neural_surfaces_for_elastic_4d_shape_representation_and_analysis.md
!3d_vision/efficient_depth_estimation_for_unstable_stereo_camera_systems_on_ar_glasses.md
!3d_vision/efficient_dynamic_scene_editing_via_4d_gaussian-based_static-dynamic_separation.md
!3d_vision/egopressure_a_dataset_for_hand_pressure_and_pose_estimation_in_egocentric_vision.md
!3d_vision/eigengs_representation_from_eigenspace_to_gaussian_image_space.md
!3d_vision/empowering_large_language_models_with_3d_situation_awareness.md
!3d_vision/end-to-end_hoi_reconstruction_transformer_with_graph-based_encoding.md
!3d_vision/end-to-end_implicit_neural_representations_for_classification.md
!3d_vision/envgs_modeling_view-dependent_appearance_with_environment_gaussian.md
!3d_vision/erupt_efficient_rendering_with_unposed_patch_transformer.md
!3d_vision/escape_equivariant_shape_completion_via_anchor_point_encoding.md
!3d_vision/estimating_body_and_hand_motion_in_an_ego-sensed_world.md
!3d_vision/eval3d_interpretable_and_fine-grained_evaluation_for_3d_generation.md
!3d_vision/event_fields_capturing_light_fields_at_high_speed_resolution_and_dynamic_range.md
!3d_vision/eventfly_event_camera_perception_from_ground_to_the_sky.md
!3d_vision/evolving_high-quality_rendering_and_reconstruction_in_a_unified_framework_with_c.md
!3d_vision/exploiting_deblurring_networks_for_radiance_fields.md
!3d_vision/extreme_rotation_estimation_in_the_wild.md
!3d_vision/fast3r_towards_3d_reconstruction_of_1000_images_in_one_forward_pass.md
!3d_vision/faster_focal_token_acquiring-and-scaling_transformer_for_long-term_3d_objection_.md
!3d_vision/feat2gs_probing_visual_foundation_models_with_gaussian_splatting.md
!3d_vision/feature-preserving_mesh_decimation_for_normal_integration.md
!3d_vision/ffacenerf_few-shot_face_editing_in_neural_radiance_fields.md
!3d_vision/flare_feed-forward_geometry_appearance_and_camera_estimation_from_uncalibrated_s.md
!3d_vision/flare_sparse_view_reconstruction.md
!3d_vision/floating_no_more_object-ground_reconstruction_from_a_single_image.md
!3d_vision/flow-nerf_joint_learning_of_geometry_poses_and_dense_flow_within_unified_neural_.md
!3d_vision/flowing_from_words_to_pixels_a_noise-free_framework_for_cross-modality_evolution.md
!3d_vision/floxels_fast_unsupervised_voxel_based_scene_flow_estimation.md
!3d_vision/fluidnexus_3d_fluid_reconstruction_and_prediction_from_a_single_video.md
!3d_vision/foundationstereo_zero-shot_stereo_matching.md
!3d_vision/framevggt_frame_evidence_rolling_memory_for_streaming_vggt.md
!3d_vision/freegave_3d_physics_learning_from_dynamic_videos_by_gaussian_velocity.md
!3d_vision/freescene_mixed_graph_diffusion_for_3d_scene_synthesis_from_free_prompts.md
!3d_vision/fruitninja_3d_object_interior_texture_generation_with_gaussian_splatting.md
!3d_vision/fshnet_fully_sparse_hybrid_network_for_3d_object_detection.md
!3d_vision/functionality_understanding_and_segmentation_in_3d_scenes.md
!3d_vision/gasp_gaussian_avatars_with_synthetic_priors.md
!3d_vision/gausshdr_high_dynamic_range_gaussian_splatting_via_learning_unified_3d_and_2d_lo.md
!3d_vision/gaussian_eigen_models_for_human_heads.md
!3d_vision/gaussian_splatting_feature_fields_for_privacy-preserving_visual_localization.md
!3d_vision/gaussian_splatting_for_efficient_satellite_image_photogrammetry.md
!3d_vision/gaussianudf_inferring_unsigned_distance_functions_through_3d_gaussian_splatting.md
!3d_vision/gaustar_gaussian_surface_tracking_and_reconstruction.md
!3d_vision/geal_generalizable_3d_affordance_learning_with_cross-modal_consistency.md
!3d_vision/gen3deval_using_vllms_for_automatic_evaluation_of_generated_3d_objects.md
!3d_vision/generating_3d-consistent_videos_from_unposed_internet_photos.md
!3d_vision/generative_multiview_relighting_for_3d_reconstruction_under_extreme_illumination.md
!3d_vision/generative_omnimatte_learning_to_decompose_video_into_layers.md
!3d_vision/genfusion_closing_the_loop_between_reconstruction_and_generation_via_videos.md
!3d_vision/genpc_zero-shot_point_cloud_completion_via_3d_generative_priors.md
!3d_vision/genvdm_generating_vector_displacement_maps_from_a_single_image.md
!3d_vision/geometry_field_splatting_with_gaussian_surfels.md
!3d_vision/geometry_in_style_3d_stylization_via_surface_normal_deformation.md
!3d_vision/gifstream_4d_gaussian-based_immersive_video_with_feature_stream.md
!3d_vision/glossy_object_reconstruction_with_cost-effective_polarized_acquisition.md
!3d_vision/go-n3rdet_geometry_optimized_nerf-enhanced_3d_object_detector.md
!3d_vision/great_geometry-intention_collaborative_inference_for_open-vocabulary_3d_object_a.md
!3d_vision/grounding_3d_object_affordance_with_language_instructions_visual_observations_an.md
!3d_vision/gs-2dgs_geometrically_supervised_2dgs_for_reflective_object_reconstruction.md
!3d_vision/guardsplat_efficient_and_robust_watermarking_for_3d_gaussian_splatting.md
!3d_vision/guiding_human-object_interactions_with_rich_geometry_and_relations.md
!3d_vision/handos_3d_hand_reconstruction_in_one_stage.md
!3d_vision/hardware-rasterized_ray-based_gaussian_splatting.md
!3d_vision/hash3d_training-free_acceleration_for_3d_generation.md
!3d_vision/hawor_world-space_hand_motion_reconstruction_from_egocentric_videos.md
!3d_vision/hd-epic_a_highly-detailed_egocentric_video_dataset.md
!3d_vision/hearing_hands_generating_sounds_from_physical_interactions_in_3d_scenes.md
!3d_vision/heatformer_a_neural_optimizer_for_multiview_human_mesh_recovery.md
!3d_vision/helvipad_a_real-world_dataset_for_omnidirectional_stereo_depth_estimation.md
!3d_vision/high-fidelity_3d_object_generation_from_single_image_with_rgbn-volume_gaussian_r.md
!3d_vision/hoi3dgen_generating_high-quality_human-object-interactions_in_3d.md
!3d_vision/horizon-gs_unified_3d_gaussian_splatting_for_large-scale_aerial-to-ground_scenes.md
!3d_vision/hot3d_hand_and_object_tracking_in_3d_from_egocentric_multi-view_videos.md
!3d_vision/hravatar_high-quality_and_relightable_gaussian_head_avatar.md
!3d_vision/hybrid_etfce-grf_exact_cluster-size_retrieval_with_analytical_p-values_for_voxel.md
!3d_vision/hybridgs_decoupling_transients_and_statics_with_2d_and_3d_gaussian_splatting.md
!3d_vision/hypergs_hyperspectral_3d_gaussian_splatting.md
!3d_vision/iaao_interactive_affordance_learning_for_articulated_objects_in_3d_environments.md
!3d_vision/identity-preserving_distillation_sampling_by_fixed-point_iterator.md
!3d_vision/imfine_3d_inpainting_via_geometry-guided_multi-view_refinement.md
!3d_vision/improving_gaussian_splatting_with_localized_points_management.md
!3d_vision/inceventgs_pose-free_gaussian_splatting_from_a_single_event_camera.md
!3d_vision/instant3dit_multiview_inpainting_for_fast_editing_of_3d_objects.md
!3d_vision/instanthdr_single-forward_gaussian_splatting_for_high_dynamic_range_3d_reconstru.md
!3d_vision/interactvlm_3d_interaction_reasoning_from_2d_foundational_models.md
!3d_vision/irgs_inter-reflective_gaussian_splatting_with_2d_gaussian_ray_tracing.md
!3d_vision/iris_inverse_rendering_of_indoor_scenes_from_low_dynamic_range_images.md
!3d_vision/isegman_interactive_segment-and-manipulate_3d_gaussians.md
!3d_vision/joint_optimization_of_neural_radiance_fields_and_continuous_camera_motion_from_a.md
!3d_vision/jopp-3d_joint_open_vocabulary_semantic_segmentation_on_point_clouds_and_panorama.md
!3d_vision/kiss3dgen_repurposing_image_diffusion_models_for_3d_asset_generation.md
!3d_vision/layered_motion_fusion_lifting_motion_segmentation_to_3d_in_egocentric_videos.md
!3d_vision/learnable_infinite_taylor_gaussian_for_dynamic_view_rendering.md
!3d_vision/learning_class_prototypes_for_unified_sparse-supervised_3d_object_detection.md
!3d_vision/leveraging_3d_geometric_priors_in_2d_rotation_symmetry_detection.md
!3d_vision/light3r-sfm_towards_feed-forward_structure-from-motion.md
!3d_vision/lim_large_interpolator_model_for_dynamic_reconstruction.md
!3d_vision/lookcloser_frequency-aware_radiance_field_for_tiny-detail_scene.md
!3d_vision/lt3sd_latent_trees_for_3d_scene_diffusion.md
!3d_vision/lucas_layered_universal_codec_avatars.md
!3d_vision/mac-ego3d_multi-agent_gaussian_consensus_for_real-time_collaborative_ego-motion_.md
!3d_vision/magic-slam_multi-agent_gaussian_globally_consistent_slam.md
!3d_vision/mani-gs_gaussian_splatting_manipulation_with_triangular_mesh.md
!3d_vision/mar-3d_progressive_masked_auto-regressor_for_high-resolution_3d_generation.md
!3d_vision/marvel-40m_multi-level_visual_elaboration_for_high-fidelity_text-to-3d_content_c.md
!3d_vision/masked_point-entity_contrast_for_open-vocabulary_3d_scene_understanding.md
!3d_vision/maskgaussian_adaptive_3d_gaussian_representation_from_probabilistic_masks.md
!3d_vision/mast3r-slam_real-time_dense_slam_with_3d_reconstruction_priors.md
!3d_vision/matcha_gaussians_atlas_of_charts_for_high-quality_geometry_and_photorealism_from.md
!3d_vision/material_anything_generating_materials_for_any_3d_object_via_diffusion.md
!3d_vision/matrix3d_large_photogrammetry_model_all-in-one.md
!3d_vision/mega_masked_generative_autoencoder_for_human_mesh_recovery.md
!3d_vision/megasam_accurate_fast_and_robust_structure_and_motion_from_casual_dynamic_videos.md
!3d_vision/megasynth_scaling_up_3d_scene_reconstruction_with_synthesized_data.md
!3d_vision/mesh_mamba_a_unified_state_space_model_for_saliency_prediction_in_non-textured_a.md
!3d_vision/meshart_generating_articulated_meshes_with_structure-guided_transformers.md
!3d_vision/met3r_measuring_multi-view_consistency_in_generated_images.md
!3d_vision/metascenes_towards_automated_replica_creation_for_real-world_3d_scans.md
!3d_vision/micas_multi-grained_in-context_adaptive_sampling_for_3d_point_cloud_processing.md
!3d_vision/midi_multi-instance_diffusion_for_single_image_to_3d_scene_generation.md
!3d_vision/mitigating_ambiguities_in_3d_classification_with_gaussian_splatting.md
!3d_vision/mne-slam_multi-agent_neural_slam_for_mobile_robots.md
!3d_vision/mobile-gs_real-time_gaussian_splatting_for_mobile_devices.md
!3d_vision/mono2stereo_a_benchmark_and_empirical_study_for_stereo_conversion.md
!3d_vision/monocular_and_generalizable_gaussian_talking_head_animation.md
!3d_vision/monoplace3d_learning_3d-aware_object_placement_for_3d_monocular_detection.md
!3d_vision/morpheus_text-driven_3d_gaussian_splat_shape_and_color_stylization.md
!3d_vision/mosaic3d_foundation_dataset_and_model_for_open-vocabulary_3d_segmentation.md
!3d_vision/mosca_dynamic_gaussian_fusion_from_casual_videos_via_4d_motion_scaffolds.md
!3d_vision/most_efficient_monarch_sparse_tuning_for_3d_representation_learning.md
!3d_vision/motionanymesh_physics-grounded_articulation_for_simulation-ready_digital_twins.md
!3d_vision/motionpro_exploring_the_role_of_pressure_in_human_mocap_and_beyond.md
!3d_vision/movis_enhancing_multi-object_novel_view_synthesis_for_indoor_scenes.md
!3d_vision/mp-sfm_monocular_surface_priors_for_robust_structure-from-motion.md
!3d_vision/multi-view_pose-agnostic_change_localization_with_zero_labels.md
!3d_vision/multi-view_reconstruction_via_sfm-guided_monocular_depth_estimation.md
!3d_vision/murre_sfm_guided_depth_reconstruction.md
!3d_vision/must3r_multi-view_network_for_stereo_3d_reconstruction.md
!3d_vision/mv-dust3r_single-stage_scene_reconstruction_from_sparse_views_in_2_seconds.md
!3d_vision/mv_3dcd_multiview_change_detection.md
!3d_vision/mvboost_boost_3d_reconstruction_with_multi-view_refinement.md
!3d_vision/mvgenmaster_scaling_multi-view_generation_from_any_image_via_3d_priors_enhanced_.md
!3d_vision/mvpaint_synchronized_multi-view_diffusion_for_painting_anything_3d.md
!3d_vision/mvsanywhere_zero-shot_multi-view_stereo.md
!3d_vision/nerfprior_learning_neural_radiance_field_as_a_prior_for_indoor_scene_reconstruct.md
!3d_vision/neuro-3d_towards_3d_visual_decoding_from_eeg_signals.md
!3d_vision/node-rf_learning_generalized_continuous_space-time_scene_dynamics_with_neural_od.md
!3d_vision/nopain_no-box_point_cloud_attack_via_optimal_transport_singular_boundary.md
!3d_vision/novel_view_synthesis_with_pixel-space_diffusion_models.md
!3d_vision/odhsr_online_dense_3d_reconstruction_of_humans_and_scenes_from_monocular_videos.md
!3d_vision/offsetopt_explicit_surface_reconstruction_without_normals.md
!3d_vision/olympus_a_universal_task_router_for_computer_vision_tasks.md
!3d_vision/on_denoising_walking_videos_for_gait_recognition.md
!3d_vision/one_diffusion_to_generate_them_all.md
!3d_vision/open-vocabulary_functional_3d_scene_graphs_for_real-world_indoor_spaces.md
!3d_vision/open-world_amodal_appearance_completion.md
!3d_vision/ouroboros3d_image-to-3d_generation_via_3d-aware_recursive_diffusion.md
!3d_vision/p-slcr_unsupervised_point_cloud_semantic_segmentation_via_prototypes_structure_l.md
!3d_vision/pano360_perspective_to_panoramic_vision_with_geometric_consistency.md
!3d_vision/parametric_point_cloud_completion_for_polygonal_surface_reconstruction.md
!3d_vision/partrm_modeling_part-level_dynamics_with_large_cross-state_reconstruction_model.md
!3d_vision/pbr-nerf_inverse_rendering_with_physics-based_neural_fields.md
!3d_vision/pcdreamer_point_cloud_completion_through_multi-view_diffusion_priors.md
!3d_vision/perception_tokens_enhance_visual_reasoning_in_multimodal_language_models.md
!3d_vision/perceptual_inductive_bias_is_what_you_need_before_contrastive_learning.md
!3d_vision/perla_perceptive_3d_language_assistant.md
!3d_vision/perse_personalized_3d_generative_avatars_from_a_single_portrait.md
!3d_vision/perturb-and-revise_flexible_3d_editing_with_generative_trajectories.md
!3d_vision/pgc_physics-based_gaussian_cloth_from_a_single_pose.md
!3d_vision/physanimator_physics-guided_generative_cartoon_animation.md
!3d_vision/physgen3d_crafting_a_miniature_interactive_world_from_a_single_image.md
!3d_vision/pico_reconstructing_3d_people_in_contact_with_objects.md
!3d_vision/pixel-aligned_rgb-nir_stereo_imaging_and_dataset_for_robot_vision.md
!3d_vision/pma_towards_parameter-efficient_point_cloud_understanding_via_point_mamba_adapte.md
!3d_vision/pointlora_low-rank_adaptation_with_token_selection_for_point_cloud_learning.md
!3d_vision/pop-gs_next_best_view_in_3d-gaussian_splatting_with_p-optimality.md
!3d_vision/pow3r_empowering_unconstrained_3d_reconstruction_with_camera_and_scene_priors.md
!3d_vision/prada_projective_radial_distortion_averaging.md
!3d_vision/preconditioners_for_the_stochastic_training_of_neural_fields.md
!3d_vision/preditor3d_fast_and_precise_3d_shape_editing.md
!3d_vision/probesdf_light_field_probes_for_neural_surface_reconstruction.md
!3d_vision/prompthmr_promptable_human_mesh_recovery.md
!3d_vision/protodepth_unsupervised_continual_depth_completion_with_prototypes.md
!3d_vision/proxytransformation_preshaping_point_cloud_manifold_with_proxy_attention_for_3d_.md
!3d_vision/ps-eip_robust_photometric_stereo_based_on_event_interval_profile.md
!3d_vision/pup_3d-gs_principled_uncertainty_pruning_for_3d_gaussian_splatting.md
!3d_vision/rainygs_efficient_rain_synthesis_with_physically-based_gaussian_splatting.md
!3d_vision/rasp_revisiting_3d_anamorphic_art_for_shadow-guided_packing_of_irregular_objects.md
!3d_vision/rdd_robust_feature_detector_and_descriptor_using_deformable_transformer.md
!3d_vision/recap_better_gaussian_relighting_with_cross-environment_captures.md
!3d_vision/reconstructing_animals_and_the_wild.md
!3d_vision/reconstructing_close_human_interaction_with_appearance_and_proxemics_reasoning.md
!3d_vision/reconstructing_humans_with_a_biomechanically_accurate_skeleton.md
!3d_vision/reconstructing_in-the-wild_open-vocabulary_human-object_interactions.md
!3d_vision/reconstructing_people_places_and_cameras.md
!3d_vision/recovering_dynamic_3d_sketches_from_videos.md
!3d_vision/ref-gs_directional_factorization_for_2d_gaussian_splatting.md
!3d_vision/reference-based_3d-aware_image_editing_with_triplanes.md
!3d_vision/regularizing_inr_with_diffusion_prior_self-supervised_3d_reconstruction_of_neutr.md
!3d_vision/relation3d_enhancing_relation_modeling_for_point_cloud_instance_segmentation.md
!3d_vision/relationfield_relate_anything_in_radiance_fields.md
!3d_vision/relative_pose_estimation_through_affine_corrections_of_monocular_depth_priors.md
!3d_vision/rethinking_end-to-end_2d_to_3d_scene_segmentation_in_gaussian_splatting.md
!3d_vision/rewis3d_reconstruction_improves_weakly-supervised_semantic_segmentation.md
!3d_vision/riggs_rigging_of_3d_gaussians_for_modeling_articulated_objects_in_videos.md
!3d_vision/rng_relightable_neural_gaussians.md
!3d_vision/roomtour3d_geometry-aware_video-instruction_tuning_for_embodied_navigation.md
!3d_vision/s2gaussian_sparse-view_super-resolution_3d_gaussian_splatting.md
!3d_vision/sar3d_autoregressive_3d_object_generation_and_understanding_via_multi-scale_3d_v.md
!3d_vision/sat-hmr_real-time_multi-person_3d_mesh_estimation_via_scale-adaptive_tokens.md
!3d_vision/scalable_autoregressive_monocular_depth_estimation.md
!3d_vision/scaling_mesh_generation_via_compressive_tokenization.md
!3d_vision/scaling_properties_of_diffusion_models_for_perceptual_tasks.md
!3d_vision/scenefactor_factored_latent_3d_diffusion_for_controllable_3d_scene_generation.md
!3d_vision/scflow2_plug-and-play_object_pose_refiner_with_shape-constraint_scene_flow.md
!3d_vision/scope_scene-contextualized_incremental_few-shot_3d_segmentation.md
!3d_vision/seeground_see_and_ground_for_zero-shot_open-vocabulary_3d_visual_grounding.md
!3d_vision/seeing_a_3d_world_in_a_grain_of_sand.md
!3d_vision/selfsplat_pose-free_and_3d_prior-free_generalizable_3d_gaussian_splatting.md
!3d_vision/semalign3d_semantic_correspondence_between_rgb-images_through_aligning_3d_object.md
!3d_vision/seurat_from_moving_points_to_depth.md
!3d_vision/sfm-free_3d_gaussian_splatting_via_hierarchical_training.md
!3d_vision/sgcr_spherical_gaussians_for_efficient_3d_curve_reconstruction.md
!3d_vision/shapeshifter_3d_variations_using_multiscale_and_sparse_point-voxel_diffusion.md
!3d_vision/sharp-it_a_multi-view_to_multi-view_diffusion_model_for_3d_synthesis_and_manipul.md
!3d_vision/sharpdepth_sharpening_metric_depth_predictions_using_diffusion_distillation.md
!3d_vision/simavatar_simulation-ready_avatars_with_layered_hair_and_clothing.md
!3d_vision/simvs_simulating_world_inconsistencies_for_robust_view_synthesis.md
!3d_vision/sinr_sparsity_driven_compressed_implicit_neural_representations.md
!3d_vision/sketchy_bounding-box_supervision_for_3d_instance_segmentation.md
!3d_vision/slam3r_real-time_dense_scene_reconstruction_from_monocular_rgb_videos.md
!3d_vision/sogs_second-order_anchor_for_advanced_3d_gaussian_splatting.md
!3d_vision/sonata_self-supervised_learning_of_reliable_point_representations.md
!3d_vision/soundvista_novel-view_ambient_sound_synthesis_via_visual-acoustic_binding.md
!3d_vision/sp3d_boosting_sparsely-supervised_3d_object_detection_via_accurate_cross-modal_s.md
!3d_vision/spar3d_stable_point-aware_reconstruction_of_3d_objects_from_single_images.md
!3d_vision/spars3r_semantic_prior_alignment_and_regularization_for_sparse_3d_reconstruction.md
!3d_vision/sparse_point_cloud_patches_rendering_via_splitting_2d_gaussians.md
!3d_vision/sparse_voxels_rasterization_real-time_high-fidelity_radiance_field_rendering.md
!3d_vision/spatialdreamer_self-supervised_stereo_video_synthesis_from_monocular_input.md
!3d_vision/spectral_defense_against_resource-targeting_attack_in_3d_gaussian_splatting.md
!3d_vision/spectral_informed_mamba_for_robust_point_cloud_processing.md
!3d_vision/spectromotion_dynamic_3d_reconstruction_of_specular_scenes.md
!3d_vision/speedy-splat_fast_3d_gaussian_splatting_with_sparse_pixels_and_sparse_primitives.md
!3d_vision/sphereuformer_a_u-shaped_transformer_for_spherical_360_perception.md
!3d_vision/splatflow_multi-view_rectified_flow_model_for_3d_gaussian_splatting_synthesis.md
!3d_vision/splinegs_robust_motion-adaptive_spline_for_real-time_dynamic_3d_gaussians_from_m.md
!3d_vision/stable-score_a_stable_registration-based_framework_for_3d_shape_correspondence.md
!3d_vision/stagedesigner_artistic_stage_generation_for_scenography_via_theater_scripts.md
!3d_vision/stdgen_semantic-decomposed_3d_character_generation_from_single_images.md
!3d_vision/steepest_descent_density_control_for_compact_3d_gaussian_splatting.md
!3d_vision/stereo4d_learning_how_things_move_in_3d_from_internet_stereo_videos.md
!3d_vision/structure_from_collision.md
!3d_vision/structured_3d_latents_for_scalable_and_versatile_3d_generation.md
!3d_vision/sum_parts_benchmarking_part-level_semantic_segmentation_of_urban_meshes.md
!3d_vision/svg-ir_spatially-varying_gaussian_splatting_for_inverse_rendering.md
!3d_vision/symmetry_strikes_back_from_single-image_symmetry_detection_to_3d_generation.md
!3d_vision/synthetic_prior_for_few-shot_drivable_head_avatar_inversion.md
!3d_vision/targeted_forgetting_of_image_subgroups_in_clip_models.md
!3d_vision/text-guided_sparse_voxel_pruning_for_efficient_3d_visual_grounding.md
!3d_vision/textured_gaussians_for_enhanced_3d_scene_appearance_modeling.md
!3d_vision/touch2shape_touch-conditioned_3d_diffusion_for_shape_exploration_and_reconstruct.md
!3d_vision/toward_robust_neural_reconstruction_from_sparse_point_sets.md
!3d_vision/towards_high-fidelity_3d_talking_avatar_with_personalized_dynamic_texture.md
!3d_vision/towards_realistic_example-based_modeling_via_3d_gaussian_stitching.md
!3d_vision/treemeshgpt_artistic_mesh_generation_with_autoregressive_tree_sequencing.md
!3d_vision/tritex_learning_texture_from_a_single_mesh_via_triplane_semantic_features.md
!3d_vision/turbo3d_ultra-fast_text-to-3d_generation.md
!3d_vision/twinner_shining_light_on_digital_twins_in_a_few_snaps.md
!3d_vision/uncommon_objects_in_3d.md
!3d_vision/unik3d_universal_camera_monocular_3d_estimation.md
!3d_vision/unipre3d_unified_pre-training_of_3d_point_cloud_models_with_cross-modal_gaussian.md
!3d_vision/uvgs_reimagining_unstructured_3d_gaussian_splatting_using_uv_mapping.md
!3d_vision/varsplat_uncertainty-aware_3d_gaussian_splatting_for_robust_rgb-d_slam.md
!3d_vision/vggt_visual_geometry_grounded_transformer.md
!3d_vision/vid2avatar-pro_authentic_avatar_from_videos_in_the_wild_via_universal_prior.md
!3d_vision/vid2sim_realistic_and_interactive_simulation_from_video_for_urban_navigation.md
!3d_vision/video_depth_anything_consistent_depth_estimation_for_super-long_videos.md
!3d_vision/video_depth_without_video_models.md
!3d_vision/vision-language_embodiment_for_monocular_depth_estimation.md
!3d_vision/volumetric_surfaces_representing_fuzzy_geometries_with_layered_meshes.md
!3d_vision/volumetrically_consistent_3d_gaussian_rasterization.md
!3d_vision/wildgs-slam_monocular_gaussian_splatting_slam_in_dynamic_environments.md
!3d_vision/wonderland_navigating_3d_scenes_from_a_single_image.md
!3d_vision/wonderworld_interactive_3d_scene_generation_from_a_single_image.md
!3d_vision/you_see_it_you_got_it_learning_3d_creation_on_pose-free_videos_at_scale.md
!3d_vision/zero-shot_monocular_scene_flow_estimation_in_the_wild.md
!3d_vision/zero-shot_novel_view_and_depth_synthesis_with_multi-view_geometric_diffusion.md
!ai_safety/a_simple_data_augmentation_for_feature_distribution_skewed_federated_learning.md
!ai_safety/data-free_universal_adversarial_perturbation_with_pseudo-semantic_prior.md
!ai_safety/deal_data-efficient_adversarial_learning_for_high-quality_infrared_imaging.md
!ai_safety/dede_detecting_backdoor_samples_for_ssl_encoders_via_decoders.md
!ai_safety/detecting_backdoor_attacks_in_federated_learning_via_direction_alignment_inspect.md
!ai_safety/detecting_out-of-distribution_through_the_lens_of_neural_collapse.md
!ai_safety/dynamic_integration_of_task-specific_adapters_for_class_incremental_learning.md
!ai_safety/fedawa_adaptive_optimization_of_aggregation_weights_in_federated_learning_using_.md
!ai_safety/forensics_adapter_adapting_clip_for_generalizable_face_forgery_detection.md
!ai_safety/geometric_knowledge-guided_localized_global_distribution_alignment_for_federated.md
!ai_safety/gradient_inversion_attacks_on_parameter-efficient_fine-tuning.md
!ai_safety/h2st_hierarchical_two-sample_tests_for_continual_out-of-distribution_detection.md
!ai_safety/infighting_in_the_dark_multi-label_backdoor_attack_in_federated_learning.md
!ai_safety/invisible_backdoor_attack_against_self-supervised_learning.md
!ai_safety/joint_out-of-distribution_filtering_and_data_discovery_active_learning.md
!ai_safety/leveraging_perturbation_robustness_to_enhance_out-of-distribution_detection.md
!ai_safety/lyapunov_stable_graph_neural_flow.md
!ai_safety/mind_the_gap_detecting_black-box_adversarial_attacks_in_the_making_through_query.md
!ai_safety/mos-attack_a_scalable_multi-objective_adversarial_attack_framework.md
!ai_safety/not_federated_unlearning_via_weight_negation.md
!ai_safety/oodd_test-time_out-of-distribution_detection_with_dynamic_dictionary.md
!ai_safety/psbd_prediction_shift_uncertainty_unlocks_backdoor_detection.md
!ai_safety/split_adaptation_for_pre-trained_vision_transformers.md
!ai_safety/stacking_brick_by_brick_aligned_feature_isolation_for_incremental_face_forgery_d.md
!ai_safety/towards_general_visual-linguistic_face_forgery_detection.md
!ai_safety/towards_source-free_machine_unlearning.md
!ai_safety/where_the_devil_hides_deepfake_detectors_can_no_longer_be_trusted.md
!aigc_detection/enhancing_few-shot_class-incremental_learning_via_training-free_bi-level_modalit.md
!aigc_detection/proapo_progressively_automatic_prompt_optimization_for_visual_classification.md
!aigc_detection/sgc-net_stratified_granular_comparison_network_for_open-vocabulary_hoi_detection.md
!audio_speech/contextual_ad_narration_with_interleaved_multimodal_sequence.md
!audio_speech/crab_a_unified_audio-visual_scene_understanding_model_with_explicit_cooperation.md
!audio_speech/distinctad_distinctive_audio_description_generation_in_contexts.md
!audio_speech/dualtalk_dual-speaker_interaction_for_3d_talking_head_conversations.md
!audio_speech/emova_empowering_language_models_to_see_hear_and_speak_with_vivid_emotions.md
!audio_speech/enhancing_dance-to-music_generation_via_negative_conditioning_latent_diffusion_m.md
!audio_speech/hop_heterogeneous_topology-based_multimodal_entanglement_for_co-speech_gesture_g.md
!audio_speech/improving_sound_source_localization_with_joint_slot_attention_on_image_and_audio.md
!audio_speech/imvid_immersive_volumetric_videos_for_enhanced_vr_engagement.md
!audio_speech/learning_to_highlight_audio_by_watching_movies.md
!audio_speech/livecc_learning_video_llm_with_streaming_speech_transcription_at_scale.md
!audio_speech/object-aware_sound_source_localization_via_audio-visual_scene_understanding.md
!audio_speech/synchronized_video-to-audio_generation_via_mel_quantization-continuum_decomposit.md
!audio_speech/team_ras_in_10th_abaw_competition_multimodal_valence_and_arousal_estimation_appr.md
!audio_speech/towards_lossless_implicit_neural_representation_via_bit_plane_decomposition.md
!audio_speech/towards_open-vocabulary_audio-visual_event_localization.md
!audio_speech/uwav_uncertainty-weighted_weakly-supervised_audio-visual_video_parsing.md
!audio_speech/video-guided_foley_sound_generation_with_multimodal_controls.md
!audio_speech/vintage_joint_video_and_text_conditioning_for_holistic_audio_generation.md
!autonomous_driving/3d-avs_lidar-based_3d_auto-vocabulary_segmentation.md
!autonomous_driving/3d_occupancy_prediction_with_low-resolution_queries_via_prototype-aware_view_tra.md
!autonomous_driving/a_dataset_for_semantic_segmentation_in_the_presence_of_unknowns.md
!autonomous_driving/a_neuro-symbolic_framework_combining_inductive_and_deductive_reasoning_for_auton.md
!autonomous_driving/a_prediction-as-perception_framework_for_3d_object_detection.md
!autonomous_driving/cawm-mamba_a_unified_model_for_infrared-visible_image_fusion_and_compound_advers.md
!autonomous_driving/certified_human_trajectory_prediction.md
!autonomous_driving/climbingcap_multi-modal_dataset_and_method_for_rock_climbing_in_world_.md
!autonomous_driving/closed-loop_supervised_fine-tuning_of_tokenized_traffic_models.md
!autonomous_driving/composing_driving_worlds_through_disentangled_control_for_adversarial_scenario_g.md
!autonomous_driving/cubify_anything_scaling_indoor_3d_object_detection.md
!autonomous_driving/decoupledgaussian_object-scene_decoupling_for_physics-based_interaction.md
!autonomous_driving/diffusiondrive_truncated_diffusion_model_for_end-to-end_autonomous_driving.md
!autonomous_driving/distilling_monocular_foundation_model_for_fine-grained_depth_completion.md
!autonomous_driving/distilling_multi-modal_large_language_models_for_autonomous_driving.md
!autonomous_driving/driving_by_the_rules_a_benchmark_for_integrating_traffic_sign_regulations_into_v.md
!autonomous_driving/drivingsphere_building_a_high-fidelity_4d_world_for_closed-loop_simulation.md
!autonomous_driving/ev-3dod_pushing_the_temporal_boundaries_of_3d_object_detection_with_event_camera.md
!autonomous_driving/evolsplat_efficient_volume-based_gaussian_splatting_for_urban_view_synthesis.md
!autonomous_driving/exploring_scene_affinity_for_semi-supervised_lidar_semantic_segmentation.md
!autonomous_driving/forestlpr_lidar_place_recognition_in_forests_attentioning_multiple_bev_density_i.md
!autonomous_driving/freesim_toward_free-viewpoint_camera_simulation_in_driving_scenes.md
!autonomous_driving/gaussianformer-2_probabilistic_gaussian_superposition_for_efficient_3d_occupancy.md
!autonomous_driving/gaussianworld_gaussian_world_model_for_streaming_3d_occupancy_prediction.md
!autonomous_driving/gdfusion_temporal_fusion_occupancy.md
!autonomous_driving/generating_multimodal_driving_scenes_via_next-scene_prediction.md
!autonomous_driving/generative_gaussian_splatting_for_unbounded_3d_city_generation.md
!autonomous_driving/glane3d_detecting_lanes_with_graph_of_3d_keypoints.md
!autonomous_driving/interactionmap_improving_online_vectorized_hdmap_construction_with_interaction.md
!autonomous_driving/learning_to_detect_objects_from_multi-agent_lidar_scans_without_manual_labels.md
!autonomous_driving/lidar-rt_gaussian-based_ray_tracing_for_dynamic_lidar_re-simulation.md
!autonomous_driving/lightloc_learning_outdoor_lidar_localization_at_light_speed.md
!autonomous_driving/limoe_mixture_of_lidar_representation_learners_from_automotive_scenes.md
!autonomous_driving/lisu_a_dataset_and_method_for_lidar_surface_normal_estimation.md
!autonomous_driving/lr-sgs_robust_lidar-reflectance-guided_salient_gaussian_splatting_for_self-drivi.md
!autonomous_driving/m2-occ_resilient_3d_semantic_occupancy_prediction_for_autonomous_driving_with_in.md
!autonomous_driving/mapgclr_geospatial_contrastive_learning_of_representations_for_online_vectorized.md
!autonomous_driving/maskgwm_a_generalizable_driving_world_model_with_video_mask_reconstruction.md
!autonomous_driving/mitracker_multi-view_integration_for_visual_object_tracking.md
!autonomous_driving/modeseq_taming_sparse_multimodal_motion_prediction_with_sequential_mode_modeling.md
!autonomous_driving/neural_inverse_rendering_from_propagating_light.md
!autonomous_driving/o3n_omnidirectional_open-vocabulary_occupancy_prediction.md
!autonomous_driving/occmamba_semantic_occupancy_prediction_with_state_space_models.md
!autonomous_driving/online_video_understanding_ovbench_and_videochat-online.md
!autonomous_driving/open-canopy_towards_very_high_resolution_forest_monitoring.md
!autonomous_driving/panoramic_multimodal_semantic_occupancy_prediction_for_quadruped_robots.md
!autonomous_driving/pansplat_4k_panorama_synthesis_with_feed-forward_gaussian_splatting.md
!autonomous_driving/physical_plausibility-aware_trajectory_prediction_via_locomotion_embodiment.md
!autonomous_driving/pidloc_cross-view_pose_optimization_network_inspired_by_pid_controllers.md
!autonomous_driving/point-to-region_loss_for_semi-supervised_point-based_crowd_counting.md
!autonomous_driving/poly-autoregressive_prediction_for_modeling_interactions.md
!autonomous_driving/prompting_depth_anything_for_4k_resolution_accurate_metric_depth_estimation.md
!autonomous_driving/psa-ssl_pose_and_size-aware_self-supervised_learning_on_lidar_point_clouds.md
!autonomous_driving/racformer_towards_high-quality_3d_object_detection_via_query-based_radar-camera_.md
!autonomous_driving/rc-autocalib_an_end-to-end_radar-camera_automatic_calibration_network.md
!autonomous_driving/recondreamer_crafting_world_models_for_driving_scene_reconstruction_via_online_r.md
!autonomous_driving/reno_real-time_neural_compression_for_3d_lidar_point_clouds.md
!autonomous_driving/rethinking_lanes_and_points_in_complex_scenarios_for_monocular_3d_lane_detection.md
!autonomous_driving/rethinking_temporal_fusion_with_a_unified_gradient_descent_view_for_3d_semantic_.md
!autonomous_driving/scenario_dreamer_vectorized_latent_diffusion_for_generating_driving_simulation_e.md
!autonomous_driving/scenecrafter_controllable_multi-view_driving_scene_editing.md
!autonomous_driving/scenediffuser_city-scale_traffic_simulation_via_a_generative_world_model.md
!autonomous_driving/sdgocc_semantic_and_depth-guided_birds-eye_view_transformation_for_3d_multimodal.md
!autonomous_driving/segment_anything_even_occluded.md
!autonomous_driving/single_pixel_image_classification_using_an_ultrafast_digital_light_projector.md
!autonomous_driving/socialmoif_multi-order_intention_fusion_for_pedestrian_trajectory_prediction.md
!autonomous_driving/solve_synergy_of_language-vision_and_end-to-end_networks_for_autonomous_driving.md
!autonomous_driving/sparsealign_a_fully_sparse_framework_for_cooperative_object_detection.md
!autonomous_driving/spatiotemporal_decoupling_for_efficient_vision-based_occupancy_forecasting.md
!autonomous_driving/spectral-geometric_neural_fields_for_pose-free_lidar_view_synthesis.md
!autonomous_driving/superpc_a_single_diffusion_model_for_point_cloud_completion_upsampling_denoising.md
!autonomous_driving/t2sg_traffic_topology_scene_graph_for_topology_reasoning_in_autonomous_driving.md
!autonomous_driving/tacodepth_towards_efficient_radar-camera_depth_estimation_with_one-stage_fusion.md
!autonomous_driving/temporal_action_detection_model_compression_by_progressive_block_drop.md
!autonomous_driving/toward_real-world_bev_perception_depth_uncertainty_estimation_via_gaussian_splat.md
!autonomous_driving/towards_autonomous_micromobility_through_scalable_urban_simulation.md
!autonomous_driving/towards_satellite_image_road_graph_extraction_a_global-scale_dataset_and_a_novel.md
!autonomous_driving/tra-moe_learning_trajectory_prediction_model_from_multiple_domains_for_adaptive_.md
!autonomous_driving/trajectory_mamba_efficient_attention-mamba_forecasting_model_based_on_selective_.md
!autonomous_driving/uncertainty-instructed_structure_injection_for_generalizable_hd_map_construction.md
!autonomous_driving/uniscene_unified_occupancy-centric_driving_scene_generation.md
!autonomous_driving/unlocking_generalization_power_in_lidar_point_cloud_registration.md
!autonomous_driving/v2x-r_cooperative_lidar-4d_radar_fusion_with_denoising_diffusion_for_3d_object_d.md
!autonomous_driving/vird_view-invariant_representation_through_dual-axis_transformation_for_cross-vi.md
!autonomous_driving/visionpad_a_vision-centric_pre-training_paradigm_for_autonomous_driving.md
!autonomous_driving/voteflow_enforcing_local_rigidity_in_self-supervised_scene_flow.md
!autonomous_driving/weathergen_a_unified_diverse_weather_generator_for_lidar_point_clouds_via_spider.md
!autonomous_driving/zero-shot_4d_lidar_panoptic_segmentation.md
!autonomous_driving/zerovo_visual_odometry_with_minimal_assumptions.md
!causal_inference/adventurer_optimizing_vision_mamba_architecture_designs_for_efficiency.md
!causal_inference/image_quality_assessment_investigating_causal_perceptual_effects_with_abductive_.md
!causal_inference/joint_scheduling_of_causal_prompts_and_tasks_for_multi-task_learning.md
!causal_inference/towards_fine-grained_interpretability_counterfactual_explanations_for_misclassif.md
!computational_biology/diffvsgg_diffusion-driven_online_video_scene_graph_generation.md
!computational_biology/multimodal_protein_language_models_for_enzyme_kinetic_parameters_from_substrate_.md
!computational_biology/semantic_and_expressive_variations_in_image_captions_across_languages.md
!computational_biology/shrec_a_spectral_embedding-based_approach_for_ab-initio_reconstruction_of_helica.md
!computational_biology/synthetic_visual_genome.md
!computational_biology/towards_spatio-temporal_world_scene_graph_generation_from_monocular_videos.md
!computational_biology/unsupervised_foundation_model-agnostic_slide-level_representation_learning.md
!earth_science/geochemad_benchmarking_unsupervised_geochemical_anomaly_detection_for_mineral_ex.md
!graph_learning/coeff-tuning_a_graph_filter_subspace_view_for_tuning_attention-based_large_model.md
!graph_learning/dvhgnn_multi-scale_dilated_vision_hgnn_for_efficient_vision_recognition.md
!graph_learning/hypergraph_vision_transformers_images_are_more_than_nodes_more_than_edges.md
!graph_learning/knowledge_bridger_towards_training-free_missing_modality_completion.md
!graph_learning/nn-former_rethinking_graph_structure_in_neural_architecture_representation.md
!graph_learning/unbiased_video_scene_graph_generation_via_visual_and_semantic_dual_debiasing.md
!graph_learning/universal_scene_graph_generation.md
!hallucination/3d-grand_a_million-scale_dataset_for_3d-llms_with_better_grounding_and_less_hall.md
!hallucination/antidote_a_unified_framework_for_mitigating_lvlm_hallucinations_in_counterfactua.md
!hallucination/halloc_token-level_localization_of_hallucinations_for_vision_language_models.md
!hallucination/octopus_alleviating_hallucination_via_dynamic_contrastive_decoding.md
!hallucination/ode_open-set_evaluation_of_hallucinations_in_multimodal_large_language_models.md
!hallucination/one_token_two_fates_a_unified_framework_via_vision_token_manipulation_against_ml.md
!hallucination/phd_a_chatgpt-prompted_visual_hallucination_evaluation_dataset.md
!hallucination/seeing_far_and_clearly_mitigating_hallucinations_in_mllms_with_attention_causal_.md
!hallucination/stop_learning_it_all_to_mitigate_visual_hallucination_focus_on_the_hallucination.md
!human_understanding/3d_face_reconstruction_from_radar_images.md
!human_understanding/3d_prior_is_all_you_need_cross-task_few-shot_2d_gaze_estimation.md
!human_understanding/analyzing_the_synthetic-to-real_domain_gap_in_3d_hand_pose_estimation.md
!human_understanding/any6d_model-free_6d_pose_estimation_of_novel_objects.md
!human_understanding/chatgarment_garment_estimation_generation_and_editing_via_large_language_models.md
!human_understanding/co-op_correspondence-based_novel_object_pose_estimation.md
!human_understanding/controlface_harnessing_facial_parametric_control_for_face_rigging.md
!human_understanding/crisp_object_pose_and_shape_estimation_with_test-time_adaptation.md
!human_understanding/cryptoface_end-to-end_encrypted_face_recognition.md
!human_understanding/d3-human_dynamic_disentangled_digital_human_from_monocular_video.md
!human_understanding/design2garmentcode_turning_design_concepts_to_tangible_garments_through_program_.md
!human_understanding/efficient_video_face_enhancement_with_enhanced_spatial-temporal_consistency.md
!human_understanding/ego4o_egocentric_human_motion_capture_and_understanding_from_multi-modal_input.md
!human_understanding/enhancing_3d_gaze_estimation_in_the_wild_using_weak_supervision_with_gaze_follow.md
!human_understanding/esc_erasing_space_concept_for_knowledge_deletion.md
!human_understanding/exploring_timeline_control_for_facial_motion_generation.md
!human_understanding/fate_full-head_gaussian_avatar_with_textural_editing_from_monocular_video.md
!human_understanding/few-shot_personalized_scanpath_prediction.md
!human_understanding/freecloth_free-form_generation_enhances_challenging_clothed_human_modeling.md
!human_understanding/freeuv_ground-truth-free_realistic_facial_uv_texture_recovery_via_cross-assembly.md
!human_understanding/fresa_feedforward_reconstruction_of_personalized_skinned_avatars_from_few_images.md
!human_understanding/fsboard_over_3_million_characters_of_asl_fingerspelling_collected_via_smartphone.md
!human_understanding/fsfm_a_generalizable_face_security_foundation_model_via_self-supervised_facial_r.md
!human_understanding/ga3ce_unconstrained_3d_gaze_estimation_with_gaze-aware_3d_context_encoding.md
!human_understanding/gaussianip_identity-preserving_realistic_3d_human_generation_via_human-centric_d.md
!human_understanding/gce-pose_global_context_enhancement_for_category-level_object_pose_estimation.md
!human_understanding/hipart_hierarchical_pose_autoregressive_transformer_for_occluded_3d_human_pose_e.md
!human_understanding/homogeneous_dynamics_space_for_heterogeneous_humans.md
!human_understanding/hsemotion_team_at_abaw-10_competition_facial_expression_recognition_valence-arou.md
!human_understanding/human_motion_instruction_tuning.md
!human_understanding/humanmm_global_human_motion_recovery_from_multi-shot_videos.md
!human_understanding/keyface_expressive_audio-driven_facial_animation_for_long_sequences_via_keyframe.md
!human_understanding/learning_affine_correspondences_by_integrating_geometric_constraints.md
!human_understanding/lost_in_translation_found_in_context_sign_language_translation_with_contextual_c.md
!human_understanding/moee_mixture_of_emotion_experts_for_audio-driven_portrait_animation.md
!human_understanding/motionmap_representing_multimodality_in_human_pose_forecasting.md
!human_understanding/motionrefit_motion_editing.md
!human_understanding/nbavatar_neural_billboards_avatars_with_realistic_hand-face_interaction.md
!human_understanding/omni-id_holistic_identity_representation_designed_for_generative_tasks.md
!human_understanding/one2any_one-reference_6d_pose_estimation_for_any_object.md
!human_understanding/optimal_transport-guided_source-free_adaptation_for_face_anti-spoofing.md
!human_understanding/personabooth_personalized_text-to-motion_generation.md
!human_understanding/physmodpo_physically-plausible_humanoid_motion_with_preference_optimization.md
!human_understanding/pose_priors_from_language_models.md
!human_understanding/posebh_prototypical_multi-dataset_training_beyond_human_pose_estimation.md
!human_understanding/probabilistic_prompt_distribution_learning_for_animal_pose_estimation.md
!human_understanding/quaffure_real-time_quasi-static_neural_hair_simulation.md
!human_understanding/recurrent_feature_mining_and_keypoint_mixup_padding_for_category-agnostic_pose_e.md
!human_understanding/reference-free_image_quality_assessment_for_virtual_try-on_via_human_feedback.md
!human_understanding/remote_photoplethysmography_in_real-world_and_extreme_lighting_scenarios.md
!human_understanding/reperformer_immersive_human-centric_volumetric_videos_from_playback_to_photoreal.md
!human_understanding/rgbavatar_reduced_gaussian_blendshapes_for_online_modeling_of_head_avatars.md
!human_understanding/rubik_a_structured_benchmark_for_image_matching_across_geometric_challenges.md
!human_understanding/semgeomo_dynamic_contextual_human_motion_generation_with_semantic_and_geometric_.md
!human_understanding/shape_my_moves_text-driven_shape-aware_synthesis_of_human_motions.md
!human_understanding/showmak3r_compositional_tv_show_reconstruction.md
!human_understanding/simmotionedit_text-based_human_motion_editing_with_motion_similarity_prediction.md
!human_understanding/socialgesture_delving_into_multi-person_gesture_understanding.md
!human_understanding/sonic_shifting_focus_to_global_audio_perception_in_portrait_animation.md
!human_understanding/stickmotion_generating_3d_human_motions_by_drawing_a_stickman.md
!human_understanding/stochastic_human_motion_prediction_with_memory_of_action_transition_and_action_c.md
!human_understanding/structure-aware_correspondence_learning_for_relative_pose_estimation.md
!human_understanding/team_leya_in_10th_abaw_competition_multimodal_ambivalencehesitancy_recognition_a.md
!human_understanding/two_by_two_learning_multi-task_pairwise_objects_assembly_for_generalizable_robot.md
!human_understanding/two_is_better_than_one_efficient_ensemble_defense_for_robust_and_compact_models.md
!human_understanding/unihope_a_unified_approach_for_hand-only_and_hand-object_pose_estimation.md
!human_understanding/unipose_a_unified_multimodal_framework_for_human_pose_comprehension_generation_a.md
!human_understanding/unopose_unseen_object_pose_estimation_with_an_unposed_rgb-d_reference_image.md
!human_understanding/vi3nr_variance_informed_initialization_for_implicit_neural_representations.md
!human_understanding/vton_360_high-fidelity_virtual_try-on_from_any_viewing_direction.md
!human_understanding/wildavatar_learning_in-the-wild_3d_avatars_from_the_web.md
!human_understanding/wilor_end-to-end_3d_hand_localization_and_reconstruction_in-the-wild.md
!human_understanding/x-dyna_expressive_dynamic_human_image_animation.md
!image_generation/3dtopia-xl_scaling_high-quality_3d_asset_generation_via_primitive_diffusion.md
!image_generation/a_bias-free_training_paradigm_for_more_general_ai-generated_image_detection.md
!image_generation/a_comprehensive_study_of_decoder-only_llms_for_text-to-image_generation.md
!image_generation/aesthetic_post-training_diffusion_models_from_generic_preferences_with_step-by-s.md
!image_generation/an_image-like_diffusion_method_for_human-object_interaction_detection.md
!image_generation/anidoc_animation_creation_made_easier.md
!image_generation/animer_animal_pose_and_shape_estimation_using_family_aware_transformer.md
!image_generation/any-resolution_ai-generated_image_detection_by_spectral_learning.md
!image_generation/arbitrary-steps_image_super-resolution_via_diffusion_inversion.md
!image_generation/artifade_learning_to_generate_high-quality_subject_from_blemished_images.md
!image_generation/as-bridge_a_bidirectional_generative_framework_bridging_next-generation_astronom.md
!image_generation/autopresent_designing_structured_visuals_from_scratch.md
!image_generation/autoregressive_distillation_of_diffusion_transformers.md
!image_generation/avatarartist_open-domain_4d_avatarization.md
!image_generation/beyond_convolution_a_taxonomy_of_structured_operators_for_learning-based_image_p.md
!image_generation/bias_for_action_video_implicit_neural_representations_with_bias_modulation.md
!image_generation/bigain_unified_token_compression_for_joint_generation_and_classification.md
!image_generation/boost_your_human_image_generation_model_via_direct_preference_optimization.md
!image_generation/bootplace_bootstrapped_object_placement_with_detection_transformers.md
!image_generation/boow-vton_boosting_in-the-wild_virtual_try-on_via_mask-free_pseudo_data_training.md
!image_generation/cachequant_comprehensively_accelerated_diffusion_models.md
!image_generation/calibrated_multi-preference_optimization_for_aligning_diffusion_models.md
!image_generation/camfreediff_camera-free_image_to_panorama_generation_with_diffusion_model.md
!image_generation/can_generative_video_models_help_pose_estimation.md
!image_generation/channel-wise_noise_scheduled_diffusion_for_inverse_rendering_in_indoor_scenes.md
!image_generation/chatgen_automatic_text-to-image_generation_from_freestyle_chatting.md
!image_generation/classifier-free_guidance_inside_the_attraction_basin_may_cause_memorization.md
!image_generation/cleandift_diffusion_features_without_noise.md
!image_generation/clip_under_the_microscope_a_fine-grained_analysis_of_multi-object_representation.md
!image_generation/co-spy_combining_semantic_and_pixel_features_to_detect_synthetic_images_by_ai.md
!image_generation/codrawagents_a_multi-agent_dialogue_framework_for_compositional_image_generation.md
!image_generation/collaborative_decoding_makes_visual_auto-regressive_modeling_efficient.md
!image_generation/color_alignment_in_diffusion.md
!image_generation/community_forensics_using_thousands_of_generators_to_train_fake_image_detectors.md
!image_generation/compass_control_multi_object_orientation_control_for_text-to-image_generation.md
!image_generation/composing_parts_for_expressive_object_generation.md
!image_generation/comprehensive_relighting_generalizable_and_consistent_monocular_human_relighting.md
!image_generation/concept_lancet_image_editing_with_compositional_representation_transplant.md
!image_generation/concept_replacer_replacing_sensitive_concepts_in_diffusion_models_via_precision_.md
!image_generation/conceptguard_continual_personalized_text-to-image_generation_with_forgetting_and.md
!image_generation/conditional_balance_improving_multi-conditioning_trade-offs_in_image_generation.md
!image_generation/consistent_and_controllable_image_animation_with_motion_diffusion_models.md
!image_generation/controllable_human_image_generation_with_personalized_multi-garments.md
!image_generation/ctrl-o_language-controllable_object-centric_visual_representation_learning.md
!image_generation/curriculum_direct_preference_optimization_for_diffusion_and_consistency_models.md
!image_generation/custany_customizing_anything_from_a_single_example.md
!image_generation/data-free_group-wise_fully_quantized_winograd_convolution_via_learnable_scales.md
!image_generation/decentralized_diffusion_models.md
!image_generation/decloth_decomposable_3d_cloth_and_human_body_reconstruction_from_a_single_image.md
!image_generation/decouple-then-merge_finetune_diffusion_models_as_multi-task_learning.md
!image_generation/decoupling_training-free_guided_diffusion_by_admm.md
!image_generation/derivative-free_diffusion_manifold-constrained_gradient_for_unified_xai.md
!image_generation/detecting_adversarial_data_using_perturbation_forgery.md
!image_generation/dic_rethinking_conv3x3_designs_in_diffusion_models.md
!image_generation/diff2flow_training_flow_matching_models_via_diffusion_model_alignment.md
!image_generation/difflocks_generating_3d_hair_from_a_single_image_using_diffusion_models.md
!image_generation/diffsensei_bridging_multi-modal_llms_and_diffusion_models_for_customized_manga_g.md
!image_generation/diffusion-4k_ultra-high-resolution_image_synthesis_with_latent_diffusion_models.md
!image_generation/diffusion_self-distillation_for_zero-shot_customized_image_generation.md
!image_generation/dig_scalable_and_efficient_diffusion_models_with_gated_linear_attention.md
!image_generation/dissecting_and_mitigating_diffusion_bias_via_mechanistic_interpretability.md
!image_generation/dit-ic_aligned_diffusion_transformer_for_efficient_image_compression.md
!image_generation/diverseflow_sample-efficient_diverse_mode_coverage_in_flows.md
!image_generation/divide_and_conquer_heterogeneous_noise_integration_for_diffusion-based_adversari.md
!image_generation/divot_diffusion_powers_video_tokenizer_for_comprehension_and_generation.md
!image_generation/dnf_unconditional_4d_generation_with_dictionary-based_neural_fields.md
!image_generation/do_visual_imaginations_improve_vision-and-language_navigation_agents.md
!image_generation/doracycle_domain-oriented_adaptation_of_unified_generative_model_in_multimodal_c.md
!image_generation/dreamcache_finetuning-free_lightweight_personalized_image_generation_via_feature.md
!image_generation/dreamomni_unified_image_generation_and_editing.md
!image_generation/dreamrelation_bridging_customization_and_relation_generation.md
!image_generation/dreamvideo-omni_omni-motion_controlled_multi-subject_video_customization_with_la.md
!image_generation/dual-interrelated_diffusion_model_for_few-shot_anomaly_image_generation.md
!image_generation/dual_diffusion_for_unified_image_generation_and_understanding.md
!image_generation/dual_diffusion_unified_generation_understanding.md
!image_generation/dual_prompting_image_restoration_with_diffusion_transformers.md
!image_generation/dualanodiff_few_shot_anomaly_image_generation.md
!image_generation/dynamic_motion_blending_for_versatile_motion_editing.md
!image_generation/easycraft_a_robust_and_efficient_framework_for_automatic_avatar_crafting.md
!image_generation/easycraft_avatar_crafting.md
!image_generation/eden_enhanced_diffusion_for_high-quality_large-motion_video_frame_interpolation.md
!image_generation/editing_away_the_evidence_diffusion-based_image_manipulation_and_the_failure_mod.md
!image_generation/efficient_fine-tuning_and_concept_suppression_for_pruned_diffusion_models.md
!image_generation/efficient_long_video_tokenization_via_coordinate-based_patch_reconstruction.md
!image_generation/efficient_personalization_of_quantized_diffusion_model_without_backpropagation.md
!image_generation/emodubber_towards_high_quality_and_emotion_controllable_movie_dubbing.md
!image_generation/emoedit_evoking_emotions_through_image_manipulation.md
!image_generation/enhancing_creative_generation_on_stable_diffusion-based_models.md
!image_generation/enhancing_facial_privacy_protection_via_weakening_diffusion_purification.md
!image_generation/enhancing_image_aesthetics_with_dual-conditioned_diffusion_models_guided_by_mult.md
!image_generation/enhancing_privacy-utility_trade-offs_to_mitigate_memorization_in_diffusion_model.md
!image_generation/enhancing_vision-language_compositional_understanding_with_multimodal_synthetic_.md
!image_generation/erasing_undesirable_influence_in_diffusion_models.md
!image_generation/everything_to_the_synthetic_diffusion-driven_test-time_adaptation_via_synthetic-.md
!image_generation/evotok_a_unified_image_tokenizer_via_residual_latent_evolution_for_visual_unders.md
!image_generation/exploring_sparse_moe_in_gans_for_text-conditioned_image_synthesis.md
!image_generation/fade_fine_grained_erasure_diffusion.md
!image_generation/faithdiff_unleashing_diffusion_priors_for_faithful_image_super-resolution.md
!image_generation/fdeid-toolbox_face_de-identification_toolbox.md
!image_generation/filmcomposer_llm-driven_music_production_for_silent_film_clips.md
!image_generation/filmcomposer_llm_music_production.md
!image_generation/fine-grained_erasure_in_text-to-image_diffusion-based_foundation_models.md
!image_generation/finelip_clip_long_text_fine_grained.md
!image_generation/finelip_extending_clips_reach_via_fine-grained_alignment_with_longer_text_inputs.md
!image_generation/finite_difference_flow_optimization_for_rl_post-training_of_text-to-image_models.md
!image_generation/flipsketch_flipping_static_drawings_to_text-guided_sketch_animations.md
!image_generation/flipsketch_sketch_animation.md
!image_generation/focus-n-fix_region-aware_fine-tuning_for_text-to-image_generation.md
!image_generation/font-agent_enhancing_font_understanding_with_large_language_models.md
!image_generation/foundhand_large-scale_domain-specific_learning_for_controllable_hand_image_gener.md
!image_generation/fractals_made_practical_denoising_diffusion_as_partitioned_iterated_function_sys.md
!image_generation/free-viewpoint_human_animation_with_pose-correlated_reference_selection.md
!image_generation/from_elements_to_design_a_layered_approach_for_automatic_graphic_design_composit.md
!image_generation/from_words_to_structured_visuals_a_benchmark_and_framework_for_text-to-diagram_g.md
!image_generation/gcc_generative_color_constancy_via_diffusing_a_color_checker.md
!image_generation/gendeg_diffusion-based_degradation_synthesis_for_generalizable_all-in-one_image_.md
!image_generation/generation_of_maximal_snake_polyominoes_using_a_deep_neural_network.md
!image_generation/generative_image_layer_decomposition_with_visual_effects.md
!image_generation/generative_modeling_of_class_probability_for_multi_modal_representation_learning.md
!image_generation/generative_multimodal_pretraining_with_discrete_diffusion_timestep_tokens.md
!image_generation/generative_photomontage.md
!image_generation/gif_generative_inspiration_for_face_recognition_at_scale.md
!image_generation/glass_guided_latent_slot_diffusion_for_object-centric_learning.md
!image_generation/glyphmastero_a_glyph_encoder_for_high-fidelity_scene_text_editing.md
!image_generation/goku_flow_based_video_generative_foundation_models.md
!image_generation/gps_as_a_control_signal_for_image_generation.md
!image_generation/grade_benchmarking_discipline-informed_reasoning_in_image_editing.md
!image_generation/graphgpt-o_synergistic_multimodal_comprehension_and_generation_on_graphs.md
!image_generation/h-edit_effective_and_flexible_diffusion-based_editing_via_doobs_h-transform.md
!image_generation/hiding_images_in_diffusion_models_by_editing_learned_score_functions.md
!image_generation/hierarchical_flow_diffusion_for_efficient_frame_interpolation.md
!image_generation/hmar_efficient_hierarchical_masked_auto-regressive_image_generation.md
!image_generation/hsi_a_holistic_style_injector_for_arbitrary_style_transfer.md
!image_generation/ice_intrinsic_concept_extraction_from_a_single_image_via_diffusion_models.md
!image_generation/idea-bench_how_far_are_generative_models_from_professional_designing.md
!image_generation/idprotector_an_adversarial_noise_encoder_to_protect_against_id-preserving_image_.md
!image_generation/ilias_instance-level_image_retrieval_at_scale.md
!image_generation/image_generation_diversity_issues_and_how_to_tame_them.md
!image_generation/image_referenced_sketch_colorization_based_on_animation_creation_workflow.md
!image_generation/implicit_bias_injection_attacks_against_text-to-image_diffusion_models.md
!image_generation/improving_diffusion_inverse_problem_solving_with_decoupled_noise_annealing.md
!image_generation/improving_editability_in_image_generation_with_layer-wise_memory.md
!image_generation/inpo_inversion_preference_optimization_diffusion_alignment.md
!image_generation/inpo_inversion_preference_optimization_with_reparametrized_ddim_for_efficient_di.md
!image_generation/insightedit_towards_better_instruction_following_for_image_editing.md
!image_generation/instant_adversarial_purification_with_adversarial_consistency_distillation.md
!image_generation/interact_advancing_large-scale_versatile_3d_human-object_interaction_generation.md
!image_generation/interedit_navigating_text-guided_multi-human_3d_motion_editing.md
!image_generation/intermimic_towards_universal_whole-body_control_for_physics-based_human-object_i.md
!image_generation/interpretable_generative_models_through_post-hoc_concept_bottlenecks.md
!image_generation/janusflow_harmonizing_autoregression_and_rectified_flow_for_unified_multimodal_u.md
!image_generation/k-lora_unlocking_training-free_fusion_of_any_subject_and_style_loras.md
!image_generation/language-guided_image_tokenization_for_generation.md
!image_generation/latent_space_imaging.md
!image_generation/latexblend_scaling_multi-concept_customized_generation_with_latent_textual_blend.md
!image_generation/lavin-dit_large_vision_diffusion_transformer.md
!image_generation/learning_flow_fields_in_attention_for_controllable_person_image_generation.md
!image_generation/learning_to_sample_effective_and_diverse_prompts_for_text-to-image_generation.md
!image_generation/learning_visual_generative_priors_without_text.md
!image_generation/lediff_latent_exposure_diffusion_for_hdr_generation.md
!image_generation/lifting_motion_to_the_3d_world_via_2d_diffusion.md
!image_generation/lookingglass_generative_anamorphoses_via_laplacian_pyramid_warping.md
!image_generation/loraclr_contrastive_adaptation_for_customization_of_diffusion_models.md
!image_generation/low-biased_general_annotated_dataset_generation.md
!image_generation/luminet_latent_intrinsics_meets_diffusion_models_for_indoor_scene_relighting.md
!image_generation/magicquill_an_intelligent_interactive_image_editing_system.md
!image_generation/make_it_count_text-to-image_generation_with_an_accurate_number_of_objects.md
!image_generation/manganinja_line_art_colorization_with_precise_reference_following.md
!image_generation/marble_material_recomposition_and_blending_in_clip-space.md
!image_generation/mca_ctrl_attention_control_customization.md
!image_generation/mccd_multi-agent_collaboration-based_compositional_diffusion_for_complex_text-to.md
!image_generation/memories_of_forgotten_concepts.md
!image_generation/metashadow_object-centered_shadow_detection_removal_and_synthesis.md
!image_generation/mexd_an_expert-infused_diffusion_model_for_whole-slide_image_classification.md
!image_generation/minima_modality_invariant_image_matching.md
!image_generation/minority-focused_text-to-image_generation_via_prompt_optimization.md
!image_generation/mirrorverse_pushing_diffusion_models_to_realistically_reflect_the_world.md
!image_generation/mitigating_memorization_in_text-to-image_diffusion_via_region-aware_prompt_augme.md
!image_generation/mixermdm_learnable_composition_of_human_motion_diffusion_models.md
!image_generation/mmar_towards_lossless_multi-modal_auto-regressive_probabilistic_modeling.md
!image_generation/mobileportrait_real-time_one-shot_neural_head_avatars_on_mobile_devices.md
!image_generation/modeling_thousands_of_human_annotators_for_generalizable_text-to-image_person_re.md
!image_generation/move-in-2d_2d-conditioned_human_motion_generation.md
!image_generation/mtadiffusion_mask_text_alignment_diffusion_model_for_object_inpainting.md
!image_generation/multi-focal_conditioned_latent_diffusion_for_person_image_synthesis.md
!image_generation/multi-group_proportional_representations_for_text-to-image_models.md
!image_generation/multi-party_collaborative_attention_control_for_image_customization.md
!image_generation/multitwine_multi-object_compositing_with_text_and_layout_control.md
!image_generation/mvportrait_text-guided_motion_and_emotion_control_for_multi-view_vivid_portrait_.md
!image_generation/navigating_image_restoration_with_vars_distribution_alignment_prior.md
!image_generation/nearly_zero-cost_protection_against_mimicry_by_personalized_diffusion_models.md
!image_generation/nested_diffusion_models_using_hierarchical_latent_priors.md
!image_generation/noise_diffusion_for_enhancing_semantic_faithfulness_in_text-to-image_synthesis.md
!image_generation/nonisotropic_gaussian_diffusion_for_realistic_3d_human_motion_prediction.md
!image_generation/not_all_parameters_matter_masking_diffusion_models_for_enhancing_generation_abil.md
!image_generation/not_just_text_uncovering_vision_modality_typographic_threats_in_image_generation.md
!image_generation/objectmover_generative_object_movement_with_video_prior.md
!image_generation/ofer_occluded_face_expression_reconstruction.md
!image_generation/omniflow_any-to-any_generation_with_multi-modal_rectified_flows.md
!image_generation/omnigen_unified_image_generation.md
!image_generation/omnistyle_filtering_high_quality_style_transfer_data_at_scale.md
!image_generation/one_model_many_budgets_elastic_latent_interfaces_for_diffusion_transformers.md
!image_generation/opensdi_spotting_diffusion-generated_images_in_the_open_world.md
!image_generation/optimizing_for_the_shortest_path_in_denoising_diffusion_model.md
!image_generation/orida_object-centric_real-world_image_composition_dataset.md
!image_generation/osdface_one-step_diffusion_model_for_face_restoration.md
!image_generation/panorama_generation_from_nfov_image_done_right.md
!image_generation/parallel_sequence_modeling_via_generalized_spatial_propagation_network.md
!image_generation/patchdpo_patch-level_dpo_for_finetuning-free_personalized_image_generation.md
!image_generation/pattern_analogies_learning_to_perform_programmatic_image_edits_by_analogy.md
!image_generation/pcm_picard_consistency_model_for_fast_parallel_sampling_of_diffusion_models.md
!image_generation/personalized_preference_fine-tuning_of_diffusion_models.md
!image_generation/physicsgen_can_generative_models_learn_from_images_to_predict_complex_physical_r.md
!image_generation/picd_versatile_perceptual_image_compression_with_diffusion_rendering.md
!image_generation/pippo_high-resolution_multi-view_humans_from_a_single_image.md
!image_generation/pqpp_a_joint_benchmark_for_text-to-image_prompt_and_query_performance_prediction.md
!image_generation/precise_fast_and_low-cost_concept_erasure_in_value_space_orthogonal_complement_m.md
!image_generation/precisecam_precise_camera_control_for_text-to-image_generation.md
!image_generation/probability_density_geodesics_in_image_diffusion_latent_space.md
!image_generation/proreflow_progressive_reflow_with_decomposed_velocity.md
!image_generation/pursuing_temporal-consistent_video_virtual_try-on_via_dynamic_pose_interaction.md
!image_generation/rad_region-aware_diffusion_models_for_image_inpainting.md
!image_generation/randar_decoder-only_autoregressive_visual_generation_in_random_orders.md
!image_generation/random_conditioning_for_diffusion_model_compression_with_distillation.md
!image_generation/rayflow_instance-aware_diffusion_acceleration_via_adaptive_flow_trajectories.md
!image_generation/re-hold_video_hand_object_interaction_reenactment_via_adaptive_layout-instructed.md
!image_generation/rectified_diffusion_guidance_for_conditional_generation.md
!image_generation/redefining_creative_in_dictionary_towards_an_enhanced_semantic_understanding_of_.md
!image_generation/reneg_learning_negative_embedding_with_reward_guidance.md
!image_generation/reversing_flow_for_image_restoration.md
!image_generation/reward_fine-tuning_two-step_diffusion_models_via_learning_differentiable_latent-.md
!image_generation/roompainter_view-integrated_diffusion_for_consistent_indoor_scene_texturing.md
!image_generation/rorem_training_a_robust_object_remover_with_human-in-the-loop.md
!image_generation/salad_skeleton-aware_latent_diffusion_for_text-driven_motion_generation_and_edit.md
!image_generation/samam_style-aware_state_space_model_for_arbitrary_image_style_transfer.md
!image_generation/scaling_down_text_encoders_of_text-to-image_diffusion_models.md
!image_generation/science-t2i_addressing_scientific_illusions_in_image_synthesis.md
!image_generation/scribblelight_single_image_indoor_relighting_with_scribbles.md
!image_generation/scsa_a_plug-and-play_semantic_continuous-sparse_attention_for_arbitrary_semantic.md
!image_generation/see_further_when_clear_curriculum_consistency_model.md
!image_generation/self-cross_diffusion_guidance_for_text-to-image_synthesis_of_similar_subjects.md
!image_generation/self-supervised_controlnet_with_spatio-temporal_mamba_for_real-world_video_super.md
!image_generation/semanticdraw_towards_real-time_interactive_content_creation_from_image_diffusion.md
!image_generation/sgmatch_semantic-guided_non-rigid_shape_matching_with_flow_regularization.md
!image_generation/shapewords_guiding_text-to-image_synthesis_with_3d_shape-aware_prompts.md
!image_generation/shining_yourself_high-fidelity_ornaments_virtual_try-on_with_diffusion_model.md
!image_generation/showhowto_generating_scene-conditioned_step-by-step_visual_instructions.md
!image_generation/sir-diff_sparse_image_sets_restoration_with_multi-view_diffusion_model.md
!image_generation/six-cd_benchmarking_concept_removals_for_text-to-image_diffusion_models.md
!image_generation/sleepermark_towards_robust_watermark_against_fine-tuning_text-to-image_diffusion.md
!image_generation/snapgen-v_generating_a_five-second_video_within_five_seconds_on_a_mobile_device.md
!image_generation/softvq-vae_efficient_1-dimensional_continuous_tokenizer.md
!image_generation/spatial_transport_optimization_by_repositioning_attention_map_for_training-free_.md
!image_generation/stable_flow_vital_layers_for_training-free_image_editing.md
!image_generation/stableanimator_high-quality_identity-preserving_human_image_animation.md
!image_generation/stretching_each_dollar_diffusion_training_from_scratch_on_a_micro-budget.md
!image_generation/stylemaster_stylize_your_video_with_artistic_generation_and_translation.md
!image_generation/stylestudio_text-driven_style_transfer_with_selective_control_of_style_elements.md
!image_generation/svfr_a_unified_framework_for_generalized_video_face_restoration.md
!image_generation/swiftedit_lightning_fast_text-guided_image_editing_via_one-step_diffusion.md
!image_generation/symbolic_representation_for_any-to-any_generative_tasks.md
!image_generation/syncsde_a_probabilistic_framework_for_diffusion_synchronization.md
!image_generation/syncvp_joint_diffusion_for_synchronous_multi-modal_video_prediction.md
!image_generation/taming_score-based_denoisers_in_admm_a_convergent_plug-and-play_framework.md
!image_generation/taste_more_taste_better_diverse_data_and_strong_model_boost_semi-supervised_crow.md
!image_generation/tcfg_tangential_damping_classifier-free_guidance.md
!image_generation/temporal_score_analysis_for_understanding_and_correcting_diffusion_artifacts.md
!image_generation/the_art_of_deception_color_visual_illusions_and_diffusion_models.md
!image_generation/tiled_diffusion.md
!image_generation/tinyfusion_diffusion_transformers_learned_shallow.md
!image_generation/tkg-dm_training-free_chroma_key_content_generation_diffusion_model.md
!image_generation/tokenflow_unified_image_tokenizer_for_multimodal_understanding_and_generation.md
!image_generation/towards_scalable_human-aligned_benchmark_for_text-guided_image_editing.md
!image_generation/towards_transformer-based_aligned_generation_with_self-coherence_guidance.md
!image_generation/towards_understanding_and_quantifying_uncertainty_for_text-to-image_generation.md
!image_generation/training_data_provenance_verification_did_your_model_use_synthetic_data_from_my_.md
!image_generation/traversing_distortion-perception_tradeoff_using_a_single_score-based_generative_.md
!image_generation/trust_your_critic_robust_reward_modeling_and_reinforcement_learning_for_faithful.md
!image_generation/turbofill_adapting_few-step_text-to-image_model_for_fast_image_inpainting.md
!image_generation/uibdiffusion_universal_imperceptible_backdoor_attack_for_diffusion_models.md
!image_generation/ultrafusion_ultra_high_dynamic_imaging_using_exposure_fusion.md
!image_generation/uncertainty-guided_perturbation_for_image_super-resolution_diffusion_model.md
!image_generation/uni-renderer_unifying_rendering_and_inverse_rendering_via_dual_stream_diffusion.md
!image_generation/unic-adapter_unified_image-instruction_adapter_with_multi-modal_transformer_for_.md
!image_generation/unicom_unified_multimodal_modeling_via_compressed_continuous_semantic_representa.md
!image_generation/unified_uncertainty-aware_diffusion_for_multi-agent_trajectory_modeling.md
!image_generation/unireal_universal_image_generation_and_editing_via_learning_real-world_dynamics.md
!image_generation/unveil_inversion_and_invariance_in_flow_transformer_for_versatile_image_editing.md
!image_generation/using_powerful_prior_knowledge_of_diffusion_model_in_deep_unfolding_networks_for.md
!image_generation/v-bridge_bridging_video_generative_priors_to_versatile_few-shot_image_restoratio.md
!image_generation/verbdiff_text-only_diffusion_models_with_enhanced_interaction_awareness.md
!image_generation/videoworld_exploring_knowledge_learning_from_unlabeled_videos.md
!image_generation/visual-erm_reward_modeling_for_visual_equivalence.md
!image_generation/visual_lexicon_rich_image_features_in_language_space.md
!image_generation/visual_persona_foundation_model_for_full-body_human_customization.md
!image_generation/viunit_visual_unit_tests_for_more_robust_visual_programming.md
!image_generation/vlog_video-language_models_by_generative_retrieval_of_narration_vocabulary.md
!image_generation/vlogger_multimodal_diffusion_for_embodied_avatar_synthesis.md
!image_generation/wegen_a_unified_model_for_interactive_multimodal_generation_as_we_chat.md
!image_generation/wheres_the_liability_in_the_generative_era_recovery-based_black-box_detection_of.md
!image_generation/yochameleon_personalized_vision_and_language_generation.md
!image_generation/z-magic_zero-shot_multiple_attributes_guided_image_creator.md
!image_generation/zero-shot_image_restoration_using_few-step_guidance_of_consistency_models_and_be.md
!image_generation/zero-shot_styled_text_image_generation_but_make_it_autoregressive.md
!image_generation/zoomldm_latent_diffusion_model_for_multi-scale_image_generation.md
!image_restoration/a_flag_decomposition_for_hierarchical_datasets.md
!image_restoration/a_physics-informed_blur_learning_framework_for_imaging_systems.md
!image_restoration/a_regularization-guided_equivariant_approach_for_image_restoration.md
!image_restoration/adversarial_diffusion_compression_for_real-world_image_super-resolution.md
!image_restoration/augmenting_perceptual_super-resolution_via_image_quality_predictors.md
!image_restoration/classic_video_denoising_in_a_machine_learning_world_robust_fast_and_controllable.md
!image_restoration/complexity_experts_are_task-discriminative_learners_for_any_image_restoration.md
!image_restoration/darkir_robust_low-light_image_restoration.md
!image_restoration/degradation-aware_feature_perturbation_for_all-in-one_image_restoration.md
!image_restoration/detail-preserving_latent_diffusion_for_stable_shadow_removal.md
!image_restoration/dnlut_ultra-efficient_color_image_denoising_via_channel-aware_lookup_tables.md
!image_restoration/dpir_dual_prompting_restoration_dit.md
!image_restoration/echomimicv2_towards_striking_simplified_and_semi-body_human_animation.md
!image_restoration/efficient_diffusion_as_low_light_enhancer.md
!image_restoration/efficient_visual_state_space_model_for_image_deblurring.md
!image_restoration/fire_fixed-points_of_restoration_priors_for_solving_inverse_problems.md
!image_restoration/generalized_recorrupted-to-recorrupted_self-supervised_learning_beyond_gaussian_.md
!image_restoration/gyro-based_neural_single_image_deblurring.md
!image_restoration/hvi_a_new_color_space_for_low-light_image_enhancement.md
!image_restoration/infp_audio-driven_interactive_head_generation_in_dyadic_conversations.md
!image_restoration/iterative_predictor-critic_code_decoding_for_real-world_image_dehazing.md
!image_restoration/mair_a_locality-_and_continuity-preserving_mamba_for_image_restoration.md
!image_restoration/mambairv2_attentive_state_space_restoration.md
!image_restoration/one-step_event-driven_high-speed_autofocus.md
!image_restoration/pidsr_complementary_polarized_image_demosaicing_and_super-resolution.md
!image_restoration/pixel-level_and_semantic-level_adjustable_super-resolution_a_dual-lora_approach.md
!image_restoration/polarfree_polarization-based_reflection-free_imaging.md
!image_restoration/polishing_the_sky_wide-field_and_high-dynamic_range_interferometric_image_recons.md
!image_restoration/prior_does_matter_visual_navigation_via_denoising_diffusion_bridge_models.md
!image_restoration/progressive_focused_transformer_for_single_image_super-resolution.md
!image_restoration/proximal_algorithm_unrolling_flexible_and_efficient_reconstruction_networks_for_.md
!image_restoration/qmambabsr_burst_image_super-resolution_with_query_state_space_model.md
!image_restoration/reversible_decoupling_network_for_single_image_reflection_removal.md
!image_restoration/rotation-equivariant_self-supervised_method_in_image_denoising.md
!image_restoration/softshadow_leveraging_soft_masks_for_penumbra-aware_shadow_removal.md
!image_restoration/tokenize_image_patches_global_context_fusion_for_effective_haze_removal_in_large.md
!image_restoration/towards_universal_computational_aberration_correction_in_photographic_cameras_a_.md
!image_restoration/urwkv_unified_rwkv_model_with_multi-state_perspective_for_low-light_image_restor.md
!image_restoration/variational_garrote_for_sparse_inverse_problems.md
!image_restoration/vision-language_gradient_descent-driven_all-in-one_deep_unfolding_networks.md
!image_restoration/visual-instructed_degradation_diffusion_for_all-in-one_image_restoration.md
!information_retrieval/advancing_myopia_to_holism_fully_contrastive_language-image_pre-training.md
!information_retrieval/chathuman_chatting_about_3d_humans_with_tools.md
!information_retrieval/cobra_combinatorial_retrieval_augmentation_for_few-shot_adaptation.md
!information_retrieval/ezsr_event-based_zero-shot_recognition.md
!information_retrieval/few-shot_recognition_via_stage-wise_retrieval-augmented_finetuning.md
!information_retrieval/goal_global-local_object_alignment_learning.md
!information_retrieval/lotusfilter_fast_diverse_nearest_neighbor_search_via_a_learned_cutoff_table.md
!information_retrieval/preserving_clusters_in_prompt_learning_for_unsupervised_domain_adaptation.md
!information_retrieval/range_retrieval_augmented_neural_fields_for_multi-resolution_geo-embeddings.md
!information_retrieval/retrieving_semantics_from_the_deep_an_rag_solution_for_gesture_synthesis.md
!information_retrieval/towards_smart_point-and-shoot_photography.md
!information_retrieval/vdocrag_retrieval-augmented_generation_over_visually-rich_documents.md
!interpretability/albm_attribute_concept_space.md
!interpretability/attribute-formed_class-specific_concept_space_endowing_language_bottleneck_model.md
!interpretability/differentiable_inverse_rendering_with_interpretable_basis_brdfs.md
!interpretability/geometry-guided_camera_motion_understanding_in_videollms.md
!interpretability/interpretable_image_classification_via_non-parametric_part_prototype_learning.md
!interpretability/kvq_boosting_video_quality_assessment_via_saliency-guided_local_perception.md
!interpretability/l-swag_layer-sample_wise_activation_with_gradients_information_for_zero-shot_nas.md
!interpretability/language_guided_concept_bottleneck_models_for_interpretable_continual_learning.md
!interpretability/learning_on_model_weights_using_tree_experts.md
!interpretability/learning_visual_composition_through_improved_semantic_guidance.md
!interpretability/lswag_zero_shot_nas.md
!interpretability/on_the_possible_detectability_of_image-in-image_steganography.md
!interpretability/open_ad-hoc_categorization_with_contextualized_feature_learning.md
!interpretability/probing_the_mid-level_vision_capabilities_of_self-supervised_learning.md
!interpretability/prompt-cam_making_vision_transformers_interpretable_for_fine-grained_analysis.md
!interpretability/sample-_and_parameter-efficient_auto-regressive_image_models.md
!interpretability/scaling_vision_pre-training_to_4k_resolution.md
!interpretability/tide_domain_generalization.md
!interpretability/tide_training_locally_interpretable_domain_generalization_models_enables_test-ti.md
!interpretability/towards_human-understandable_multi-dimensional_concept_discovery.md
!interpretability/why_does_it_look_there_structured_explanations_for_image_classification.md
!knowledge_editing/mokus_leveraging_cross-modal_knowledge_transfer_for_knowledge-aware_concept_cust.md
!llm_agent/ata_adaptive_transformation_agent_for_text-guided_subject-position_variable_back.md
!llm_agent/feature4x_bridging_any_monocular_video_to_4d_agentic_ai_with_versatile_gaussian_.md
!llm_agent/gui-xplore_empowering_generalizable_gui_agents_with_one_exploration.md
!llm_agent/rl-rc-dot_a_block-level_rl_agent_for_task-aware_video_compression.md
!llm_agent/sceneassistant_a_visual_feedback_agent_for_open-vocabulary_3d_scene_generation.md
!llm_agent/sketchtopia_a_dataset_and_foundational_agents_for_benchmarking_asynchronous_mult.md
!llm_agent/spiritsight_agent_advanced_gui_agent_with_one_look.md
!llm_agent/tango_training-free_embodied_ai_agents_for_open-world_tasks.md
!llm_agent/visual_agentic_ai_for_spatial_reasoning_with_a_dynamic_api.md
!llm_alignment/bases_of_steerable_kernels_for_equivariant_cnns_from_2d_rotations_to_the_lorentz.md
!llm_alignment/cad-llama_leveraging_large_language_models_for_computer-aided_design_parametric_.md
!llm_alignment/continual_sft_matches_multimodal_rlhf_with_negative_supervision.md
!llm_alignment/do_we_really_need_curated_malicious_data_for_safety_alignment_in_multi-modal_lar.md
!llm_alignment/jailbreaking_the_non-transferable_barrier_via_test-time_data_disguising.md
!llm_efficiency/associative_transformer.md
!llm_efficiency/efficient_data_driven_mixture-of-expert_extraction_from_trained_networks.md
!llm_efficiency/locore_image_re-ranking_with_long-context_sequence_modeling.md
!llm_efficiency/moee_mixture_expert_extraction.md
!llm_efficiency/spatial-ttt_streaming_visual-based_spatial_intelligence_with_test-time_training.md
!llm_evaluation/erase_diffusion_empowering_object_removal_through_calibrating_diffusion_pathways.md
!llm_evaluation/postero_structuring_layout_trees_to_enable_language_models_in_generalized_conten.md
!llm_evaluation/roadsocial_a_diverse_videoqa_dataset_and_benchmark_for_road_event_understanding_.md
!llm_evaluation/unigoal_towards_universal_zero-shot_goal-oriented_navigation.md
!llm_nlp/building_vision_models_upon_heat_conduction.md
!llm_nlp/chat-based_person_retrieval_via_dialogue-refined_cross-modal_alignment.md
!llm_nlp/comrope_rotary_position.md
!llm_nlp/dora_sampling_and_benchmarking_for_3d_shape_variational_auto-encoders.md
!llm_nlp/exposure-slot_exposure-centric_representations_learning_with_slot-in-slot_attent.md
!llm_nlp/imagine_and_seek_improving_composed_image_retrieval_with_an_imagined_proxy.md
!llm_nlp/learning_textual_prompts_for_open-world_semi-supervised_learning.md
!llm_nlp/making_old_film_great_again_degradation-aware_state_space_model_for_old_film_res.md
!llm_nlp/mg-motionllm_a_unified_framework_for_motion_comprehension_and_generation_across_.md
!llm_nlp/rethinking_spiking_self-attention_mechanism_implementing_a-xnor_similarity_calcu.md
!llm_nlp/spiking_transformer_introducing_accurate_addition-only_spiking_self-attention_fo.md
!llm_nlp/spiking_transformer_with_spatial-temporal_attention.md
!llm_nlp/staa-snn_spatial-temporal_attention_aggregator_for_spiking_neural_networks.md
!llm_nlp/test-time_visual_in-context_tuning.md
!llm_nlp/the_change_you_want_to_detect_semantic_change_detection_in_earth_observation_wit.md
!llm_pretraining/a_unified_framework_for_heterogeneous_semi-supervised_learning.md
!llm_pretraining/amo_sampler_enhancing_text_rendering_with_overshooting.md
!llm_pretraining/bridging_the_vision-brain_gap_with_an_uncertainty-aware_blur_prior.md
!llm_pretraining/context-cir_learning_from_concepts_in_text_for_composed_image_retrieval.md
!llm_pretraining/dreamtext_high_fidelity_scene_text_synthesis.md
!llm_pretraining/exploration-driven_generative_interactive_environments.md
!llm_pretraining/improving_autoregressive_visual_generation_with_cluster-oriented_token_predictio.md
!llm_pretraining/influence_malleability_in_linearized_attention_dual_implications_of_non-converge.md
!llm_pretraining/mr_plip_multi_resolution_pathology.md
!llm_pretraining/planarsplatting_accurate_planar_surface_reconstruction_in_3_minutes.md
!llm_pretraining/precise_event_spotting_in_sports_videos_solving_long-range_dependency_and_class_.md
!llm_pretraining/robust_message_embedding_via_attention_flow-based_steganography.md
!llm_pretraining/scamo_exploring_the_scaling_law_in_autoregressive_motion_generation_model.md
!llm_pretraining/seeing_what_matters_empowering_clip_with_patch_generation-to-selection.md
!llm_pretraining/the_scene_language_representing_scenes_with_programs_words_and_embeddings.md
!llm_reasoning/argus_vision-centric_reasoning_with_grounded_chain-of-thought.md
!llm_reasoning/enhancing_video-llm_reasoning_via_agent-of-thoughts_distillation.md
!llm_reasoning/interleaved-modal_chain-of-thought.md
!llm_reasoning/learning-enabled_polynomial_lyapunov_function_synthesis_via_high-accuracy_counte.md
!llm_reasoning/osrcir_reflective_cot.md
!llm_reasoning/style_evolving_along_chain-of-thought_for_unknown-domain_object_detection.md
!llm_reasoning/videoespresso_a_large-scale_chain-of-thought_dataset_for_fine-grained_video_reas.md
!llm_safety/a_closed-form_solution_for_debiasing_vision-language_models_with_utility_guarant.md
!llm_safety/dual_consolidation_for_pre-trained_model-based_domain-incremental_learning.md
!llm_safety/empowering_llms_to_understand_and_generate_complex_vector_graphics.md
!llm_safety/forensiczip_more_tokens_are_better_but_not_necessary_in_forensic_vision-language.md
!llm_safety/hyperbolic_safety-aware_vision-language_models.md
!llm_safety/lotus_large-scale_machine_unlearning_with_a_taste_of_uncertainty.md
!llm_safety/low-rank_adaptation_in_multilinear_operator_networks_for_security-preserving_inc.md
!llm_safety/mp-gui_modality_perception_with_mllms_for_gui_understanding.md
!llm_safety/neural_gate_mitigating_privacy_risks_in_lvlms_via_neuron-level_gradient_gating.md
!llm_safety/protecting_your_video_content_disrupting_automated_video-based_llm_annotations.md
!llm_safety/steering_away_from_harm_an_adaptive_approach_to_defending_vision_language_model_.md
!llm_safety/tapt_test-time_adversarial_prompt_tuning_for_robust_inference_in_vision-language.md
!llm_safety/test-time_attention_purification_for_backdoored_large_vision_language_models.md
!llm_safety/towards_all-in-one_medical_image_re-identification.md
!medical_imaging/a_semi-supervised_framework_for_breast_ultrasound_segmentation_with_training-fre.md
!medical_imaging/accelerating_stroke_mri_with_diffusion_probabilistic_models_through_large-scale_.md
!medical_imaging/adaptation_of_weakly_supervised_localization_in_histopathology_by_debiasing_pred.md
!medical_imaging/addressing_data_scarcity_in_3d_trauma_detection_through_self-supervised_and_semi.md
!medical_imaging/are_general-purpose_vision_models_all_we_need_for_2d_medical_image_segmentation_.md
!medical_imaging/association_of_radiologic_ppfe_change_with_mortality_in_lung_cancer_screening_co.md
!medical_imaging/automated_detection_of_malignant_lesions_in_the_ovary_using_deep_learning_models.md
!medical_imaging/biclip_bidirectional_and_consistent_language-image_processing_for_robust_medical.md
!medical_imaging/boltzmann_attention_sampling_for_image_analysis_with_small_objects.md
!medical_imaging/bridging_the_skill_gap_in_clinical_cbct_interpretation_with_cbctrepd.md
!medical_imaging/carl_a_framework_for_equivariant_image_registration.md
!medical_imaging/cholectrack20_a_multi-perspective_tracking_dataset_for_surgical_tools.md
!medical_imaging/cloe_expert_consistency_learning_for_missing_modality_segmentation.md
!medical_imaging/crosssdf_3d_reconstruction_of_thin_structures_from_cross-sections.md
!medical_imaging/cycleulm_a_unified_label-free_deep_learning_framework_for_ultrasound_localisatio.md
!medical_imaging/decoding_matters_efficient_mamba-based_decoder_with_distribution-aware_deep_supe.md
!medical_imaging/deep_learning-based_assessment_of_the_relation_between_the_third_molar_and_mandi.md
!medical_imaging/deep_learning_based_estimation_of_blood_glucose_levels_from_multidirectional_scl.md
!medical_imaging/developing_foundation_models_for_universal_segmentation_from_3d_whole-body_posit.md
!medical_imaging/dflmoe_decentralized_federated_learning_via_mixture_of_experts_for_medical_data_.md
!medical_imaging/diffusion-based_feature_denoising_and_using_nnmf_for_robust_brain_tumor_classifi.md
!medical_imaging/din_diffusion_model_for_robust_medical_vqa_with_semantic_noisy_labels.md
!medical_imaging/distilled_prompt_learning_for_incomplete_multimodal_survival_prediction.md
!medical_imaging/domain_adaptive_diabetic_retinopathy_grading_with_model_absence_and_flowing_data.md
!medical_imaging/echoone_segmenting_multiple_echocardiography_planes_in_one_model.md
!medical_imaging/echoworld_learning_motion-aware_world_models_for_echocardiography_probe_guidance.md
!medical_imaging/enhanced_contrastive_learning_with_multi-view_longitudinal_data_for_chest_x-ray_.md
!medical_imaging/enhancing_sam_with_efficient_prompting_and_preference_optimization_for_semi-supe.md
!medical_imaging/enhancing_virtual_try-on_with_synthetic_pairs_and_error-aware_noise_scheduling.md
!medical_imaging/equivania_a_spectral_method_for_rotation-equivariant_anisotropic_image_analysis.md
!medical_imaging/evidential_learning_driven_breast_tumor_segmentation_with_stage-divided_vision-l.md
!medical_imaging/federated_modality-specific_encoders_and_partially_personalized_fusion_decoder_f.md
!medical_imaging/giim_graph-based_learning_of_inter-_and_intra-view_dependencies_for_multi-view_m.md
!medical_imaging/human_knowledge_integrated_multi-modal_learning_for_single_source_domain_general.md
!medical_imaging/interactive_medical_image_analysis_with_concept-based_similarity_reasoning.md
!medical_imaging/interactive_medical_image_segmentation_a_benchmark_dataset_and_baseline.md
!medical_imaging/latent_drifting_in_diffusion_models_for_counterfactual_medical_image_synthesis.md
!medical_imaging/mil-pf_multiple_instance_learning_on_precomputed_features_for_mammography_classi.md
!medical_imaging/moedit_on_learning_quantity_perception_for_multi-object_image_editing.md
!medical_imaging/multi-modal_vision_pre-training_for_medical_image_analysis.md
!medical_imaging/multi-resolution_pathology-language_pre-training_model_with_text-guided_visual_r.md
!medical_imaging/multimodal_classification_of_radiation-induced_contrast_enhancements_and_tumor_r.md
!medical_imaging/multimorph_on-demand_atlas_construction.md
!medical_imaging/multiscale_structure-guided_latent_diffusion_for_multimodal_mri_translation.md
!medical_imaging/noir_neural_operator_mapping_for_implicit_representations.md
!medical_imaging/noise-consistent_siamese-diffusion_for_medical_image_synthesis_and_segmentation.md
!medical_imaging/novel_architecture_of_rpa_in_oral_cancer_lesion_detection.md
!medical_imaging/nyxus_a_next_generation_image_feature_extraction_library_for_the_big_data_and_ai.md
!medical_imaging/openmibood_open_medical_imaging_benchmarks_for_out-of-distribution_detection.md
!medical_imaging/paper_title_lov3d_grounding_cognitive_prognosis_reasoning_in_longitudinal_3d_bra.md
!medical_imaging/prototype-based_knowledge_guidance_for_fine-grained_structured_radiology_reporti.md
!medical_imaging/reanimating_images_using_neural_representations_of_dynamic_stimuli.md
!medical_imaging/reinforcing_the_weakest_links_modernizing_siena_with_targeted_deep_learning_inte.md
!medical_imaging/residual_sodap_residual_self-organizing_domain-adaptive_prompting_with_structura.md
!medical_imaging/revisiting_mae_pre-training_for_3d_medical_image_segmentation.md
!medical_imaging/sacb-net_spatial-awareness_convolutions_for_medical_image_registration.md
!medical_imaging/salient_frequency-aware_paired_diffusion_for_controllable_long-tail_ct_detection.md
!medical_imaging/sam_dpo_semi_supervised.md
!medical_imaging/sapiensid_foundation_for_human_recognition.md
!medical_imaging/sealion_semantic_part-aware_latent_point_diffusion_models_for_3d_generation.md
!medical_imaging/semantic_class_distribution_learning_for_debiasing_semi-supervised_medical_image.md
!medical_imaging/semitooth_a_generalizable_semi-supervised_framework_for_multi-source_tooth_segme.md
!medical_imaging/show_and_segment_universal_medical_image_segmentation_via_in-context_learning.md
!medical_imaging/surg-r1_a_hierarchical_reasoning_foundation_model_for_scalable_and_interpretable.md
!medical_imaging/t-fake_synthesizing_thermal_images_for_facial_landmarking.md
!medical_imaging/thin-shell-sft_fine-grained_monocular_non-rigid_3d_surface_tracking_with_neural_.md
!medical_imaging/topocellgen_generating_histopathology_cell_topology_with_a_diffusion_model.md
!medical_imaging/transformer-based_multi-region_segmentation_and_radiomic_analysis_of_hr-pqct_ima.md
!medical_imaging/ultrasoundagents_hierarchical_multi-agent_evidence-chain_reasoning_for_breast_ul.md
!medical_imaging/uncertainty-aware_concept_and_motion_segmentation_for_semi-supervised_angiograph.md
!medical_imaging/unistainnet_foundation-model-guided_virtual_staining_of_he_to_ihc.md
!medical_imaging/unleashing_video_language_models_for_fine-grained_hrct_report_generation.md
!medical_imaging/unmasking_biases_and_reliability_concerns_in_convolutional_neural_networks_analy.md
!medical_imaging/unraveling_normal_anatomy_via_fluid-driven_anomaly_randomization.md
!medical_imaging/vesselfm_a_foundation_model_for_universal_3d_blood_vessel_segmentation.md
!medical_imaging/vista3d_a_unified_segmentation_foundation_model_for_3d_medical_imaging.md
!medical_imaging/weakly_supervised_teacher-student_framework_with_progressive_pseudo-mask_refinem.md
!medical_imaging/wise_a_framework_for_gigapixel_whole-slide-image_lossless_compression.md
!model_compression/adapter_merging_with_centroid_prototype_mapping_for_scalable_class-incremental_l.md
!model_compression/alternating_gradient_flow_utility_a_unified_metric_for_structural_pruning_and_dy.md
!model_compression/an_fpga_implementation_of_displacement_vector_search_for_intra_pattern_copy_in_j.md
!model_compression/arche_autoregressive_residual_compression_with_hyperprior_and_excitation.md
!model_compression/autossvh_exploring_automated_frame_sampling_for_efficient_self-supervised_video_h.md
!model_compression/bhvit_binarized_hybrid_vision_transformer.md
!model_compression/binarized_mamba-transformer_for_lightweight_quad_bayer_hybridevs_demosaicing.md
!model_compression/chapter-llama_efficient_chaptering_in_hour-long_videos_with_llms.md
!model_compression/charm_the_missing_piece_in_vit_fine-tuning_for_image_aesthetic_assessment.md
!model_compression/cl-lora_continual_low-rank_adaptation_for_rehearsal-free_class-incremental_learn.md
!model_compression/coa_towards_real_image_dehazing_via_compression-and-adaptation.md
!model_compression/curriculum_coarse-to-fine_selection_for_high-ipc_dataset_distillation.md
!model_compression/dataset_distillation_with_neural_characteristic_function_a_minmax_perspective.md
!model_compression/delt_a_simple_diversity-driven_earlylate_training_for_dataset_distillation.md
!model_compression/ders_towards_extremely_efficient_upcycled_mixture-of-experts_models.md
!model_compression/distilling_long-tailed_datasets.md
!model_compression/dkdm_data-free_knowledge_distillation_for_diffusion_models_with_any_architecture.md
!model_compression/dycoke_dynamic_compression_of_tokens_for_fast_video_large_language_models.md
!model_compression/ecvc_exploiting_non-local_correlations_in_multiple_frames_for_contextual_video_c.md
!model_compression/efficientvim_efficient_vision_mamba_with_hidden_state_mixer_based_state_space_du.md
!model_compression/embracing_collaboration_over_competition_condensing_multiple_prompts_for_visual_.md
!model_compression/emphasizing_discriminative_features_for_dataset_distillation_in_complex_scenario.md
!model_compression/enhancing_dataset_distillation_via_non-critical_region_refinement.md
!model_compression/expert_pyramid_tuning_efficient_parameter_fine-tuning_for_expertise-driven_task_.md
!model_compression/faster_parameter-efficient_tuning_with_token_redundancy_reduction.md
!model_compression/fima-q_post-training_quantization_for_vision_transformers_by_fisher_information_.md
!model_compression/gaze-lle_gaze_target_estimation_via_large-scale_learned_encoders.md
!model_compression/good_cheap_and_fast_overfitted_image_compression_with_wasserstein_distortion.md
!model_compression/hiap_a_multi-granular_stochastic_auto-pruning_framework_for_vision_transformers.md
!model_compression/hot_hadamard-based_optimized_training.md
!model_compression/hyperlora_parameter-efficient_adaptive_generation_for_portrait_synthesis.md
!model_compression/incremental_object_keypoint_learning.md
!model_compression/instag_learning_personalized_3d_talking_head_from_few-second_video.md
!model_compression/iteris_iterative_inference-solving_alignment_for_lora_merging.md
!model_compression/jamma_ultra-lightweight_local_feature_matching_with_joint_mamba.md
!model_compression/l_swag_zero_shot_nas_vision_transformers.md
!model_compression/layered_image_vectorization_via_semantic_simplification.md
!model_compression/learned_image_compression_with_dictionary-based_entropy_model.md
!model_compression/learning_compatible_multi-prize_subnetworks_for_asymmetric_retrieval.md
!model_compression/less_is_more_efficient_model_merging_with_binary_task_switch.md
!model_compression/linear_attention_modeling_for_learned_image_compression.md
!model_compression/logits_deconfusion_with_clip_for_few-shot_learning.md
!model_compression/lora_subtraction_for_drift-resistant_space_in_exemplar-free_continual_learning.md
!model_compression/lsnet_see_large_focus_small.md
!model_compression/mamba-adaptor_state_space_model_adaptor_for_visual_recognition.md
!model_compression/mambaic_state_space_models_for_high-performance_learned_image_compression.md
!model_compression/masking_meets_supervision_a_strong_learning_alliance.md
!model_compression/mdp_multidimensional_vision_model_pruning_with_latency_constraint.md
!model_compression/mobilemamba_lightweight_multi-receptive_visual_mamba_network.md
!model_compression/multi-modal_knowledge_distillation-based_human_trajectory_forecasting.md
!model_compression/mutri_multi-view_tri-alignment_for_oct_to_octa_3d_image_translation.md
!model_compression/mxnorm_reusing_mxfp_block_scales_for_efficient_tensor_normalisation.md
!model_compression/parameter_efficient_mamba_tuning_via_projector-targeted_diagonal-centric_linear_.md
!model_compression/plug-and-play_versatile_compressed_video_enhancement.md
!model_compression/q-dit_accurate_post-training_quantization_for_diffusion_transformers.md
!model_compression/quartdepth_post-training_quantization_for_real-time_depth_estimation_on_the_edge.md
!model_compression/sampling_innovation-based_adaptive_compressive_sensing.md
!model_compression/sketch_down_the_flops_towards_efficient_networks_for_human_sketch.md
!model_compression/style_quantization_for_data-efficient_gan_training.md
!model_compression/tadformer_task-adaptive_dynamic_transformer_for_efficient_multi-task_learning.md
!model_compression/task_singular_vectors_reducing_task_interference_in_model_merging.md
!model_compression/towards_practical_real-time_neural_video_compression.md
!model_compression/tripartite_weight-space_ensemble_for_few-shot_class-incremental_learning.md
!model_compression/understanding_multi-layered_transmission_matrices.md
!model_compression/wave_weight_templates_for_adaptive_initialization_of_variable-sized_models.md
!model_compression/what_makes_a_good_dataset_for_knowledge_distillation.md
!multi_agent/collaborative_tree_search_for_enhancing_embodied_multi-agent_collaboration.md
!multi_agent/comfybench_benchmarking_llm-based_agents_in_comfyui_for_autonomously_designing_c.md
!multi_agent/nader_neural_architecture_design_via_multi-agent_collaboration.md
!multilingual_mt/smtpd_a_new_benchmark_for_temporal_prediction_of_social_media_popularity.md
!multimodal_vlm/4d_langsplat_4d_language_gaussian_splatting_via_multimodal_large_language_models.md
!multimodal_vlm/active_data_curation_effectively_distills_large-scale_multimodal_models.md
!multimodal_vlm/asap_advancing_semantic_alignment_promotes_multi-modal_manipulation_de.md
!multimodal_vlm/asap_advancing_semantic_alignment_promotes_multi-modal_manipulation_detecting_an.md
!multimodal_vlm/beyond_words_augmenting_discriminative_richness_via_diffusions_in_unsupervised_p.md
!multimodal_vlm/calico_part-focused_semantic_co-segmentation_with_large_vision-language_models.md
!multimodal_vlm/can_large_vision-language_models_correct_semantic_grounding_errors_by_themselves.md
!multimodal_vlm/codepercept_code-grounded_visual_stem_perception_for_mllms.md
!multimodal_vlm/collm_a_large_language_model_for_composed_image_retrieval.md
!multimodal_vlm/comm_a_coherent_interleaved_image-text_dataset_for_multimodal_understanding_and_.md
!multimodal_vlm/completion_as_enhancement_a_degradation-aware_selective_image_guided_network_for.md
!multimodal_vlm/compositional_caching_for_training-free_open-vocabulary_attribute_detection.md
!multimodal_vlm/context-aware_multimodal_pretraining.md
!multimodal_vlm/continual_learning_with_vision-language_models_via_semantic-geometry_preservatio.md
!multimodal_vlm/counts_benchmarking_object_detectors_and_multimodal_large_language_models_under_.md
!multimodal_vlm/cropper_vision-language_model_for_image_cropping_through_in-context_learning.md
!multimodal_vlm/cross-modal_information_flow_in_multimodal_large_language_models.md
!multimodal_vlm/data_distributional_properties_as_inductive_bias_for_systematic_generalization.md
!multimodal_vlm/debiasing_multimodal_large_language_models_via_noise-aware_preference_optimizati.md
!multimodal_vlm/distraction_is_all_you_need_for_multimodal_large_language_model_jailbreaking.md
!multimodal_vlm/docopilot_improving_multimodal_models_for_document-level_understanding.md
!multimodal_vlm/docvlm_make_your_vlm_an_efficient_reader.md
!multimodal_vlm/dpc_dual-prompt_collaboration_for_tuning_vision-language_models.md
!multimodal_vlm/dynamic_updates_for_language_adaptation_in_visual-language_tracking.md
!multimodal_vlm/dynrefer_delving_into_region-level_multimodal_tasks_via_dynamic_resolution.md
!multimodal_vlm/efficient_motion-aware_video_mllm.md
!multimodal_vlm/egolm_multi-modal_language_model_of_egocentric_motions.md
!multimodal_vlm/embodied_scene_understanding_for_vision_language_models_via_metavqa.md
!multimodal_vlm/evaluating_model_perception_of_color_illusions_in_photorealistic_scenes.md
!multimodal_vlm/evaluating_vision-language_models_as_evaluators_in_path_planning.md
!multimodal_vlm/eventgpt_event_stream_understanding_with_multimodal_large_language_models.md
!multimodal_vlm/every_sam_drop_counts_embracing_semantic_priors_for_multi-modality_image_fusion_.md
!multimodal_vlm/fastvlm_efficient_vision_encoding_for_vision_language_models.md
!multimodal_vlm/finer-cam_spotting_the_difference_reveals_finer_details_for_visual_explanation.md
!multimodal_vlm/flair_vlm_with_fine-grained_language-informed_image_representations.md
!multimodal_vlm/florence-vl_enhancing_vision-language_models_with_generative_vision_encoder_and_.md
!multimodal_vlm/free_on_the_fly_enhancing_flexibility_in_test-time_adaptation_with_online_em.md
!multimodal_vlm/from_multimodal_llms_to_generalist_embodied_agents_methods_and_lessons.md
!multimodal_vlm/galaxy_walker_geometry-aware_vlms_for_galaxy-scale_understanding.md
!multimodal_vlm/generalized_few-shot_3d_point_cloud_segmentation_with_vision-language_model.md
!multimodal_vlm/genius_a_generative_framework_for_universal_multimodal_search.md
!multimodal_vlm/geomm_on_geodesic_perspective_for_multi-modal_learning.md
!multimodal_vlm/global-local_tree_search_in_vlms_for_3d_indoor_scene_generation.md
!multimodal_vlm/ground-v_teaching_vlms_to_ground_complex_instructions_in_pixels.md
!multimodal_vlm/harnessing_frozen_unimodal_encoders_for_flexible_multimodal_alignment.md
!multimodal_vlm/heie_mllm-based_hierarchical_explainable_aigc_image_implausibility_evaluator.md
!multimodal_vlm/hificl_high-fidelity_in-context_learning_for_multimodal_tasks.md
!multimodal_vlm/homesafe-bench_evaluating_vision-language_models_on_unsafe_action_detection_for_.md
!multimodal_vlm/identifying_and_mitigating_position_bias_of_multi-image_vision-language_models.md
!multimodal_vlm/img-diff_contrastive_data_synthesis_for_multimodal_large_language_models.md
!multimodal_vlm/improving_personalized_search_with_regularized_low-rank_parameter_updates.md
!multimodal_vlm/instruction-based_image_manipulation_by_watching_how_things_move.md
!multimodal_vlm/its_a_blind_match_towards_vision-language_correspondence_without_parallel_data.md
!multimodal_vlm/joint_vision-language_social_bias_removal_for_clip.md
!multimodal_vlm/lamra_large_multimodal_model_as_your_advanced_retrieval_assistant.md
!multimodal_vlm/layoutvlm_differentiable_optimization_of_3d_layout_via_vision-language_models.md
!multimodal_vlm/llava-critic_learning_to_evaluate_multimodal_models.md
!multimodal_vlm/locality-aware_zero-shot_human-object_interaction_detection.md
!multimodal_vlm/markushgrapher_joint_visual_and_textual_recognition_of_markush_structures.md
!multimodal_vlm/marten_visual_question_answering_with_mask_generation_for_multi-modal_document_u.md
!multimodal_vlm/mastering_negation_boosting_grounding_models_via_grouped_opposition-based_learni.md
!multimodal_vlm/mimic_in-context_learning_for_multimodal_tasks.md
!multimodal_vlm/mimo_a_medical_vision_language_model_with_visual_referring_multimodal_input_and_.md
!multimodal_vlm/mllm-as-a-judge_for_image_safety_without_human_labeling.md
!multimodal_vlm/mmrl_multi-modal_representation_learning_for_vision-language_models.md
!multimodal_vlm/molmo_and_pixmo_open_weights_and_open_data_for_state-of-the-art_vision-language_.md
!multimodal_vlm/mosaic_of_modalities_a_comprehensive_benchmark_for_multimodal_graph_learning.md
!multimodal_vlm/move-kd_knowledge_distillation_for_vlms_with_mixture_of_visual_encoders.md
!multimodal_vlm/multi-layer_visual_feature_fusion_in_multimodal_llms_methods_analysis_and_best_p.md
!multimodal_vlm/multi-modal_contrastive_masked_autoencoders_a_two-stage_progressive_pre-training.md
!multimodal_vlm/multimodal_autoregressive_pre-training_of_large_vision_encoders.md
!multimodal_vlm/multimodal_ocr_parse_anything_from_documents.md
!multimodal_vlm/neighborretr_balancing_hub_centrality_in_cross-modal_retrieval.md
!multimodal_vlm/nlprompt_noise-label_prompt_learning_for_vision-language_models.md
!multimodal_vlm/nvila_efficient_frontier_visual_language_models.md
!multimodal_vlm/on_the_out-of-distribution_generalization_of_large_multimodal_models.md
!multimodal_vlm/opening_a_comprehensive_benchmark_for_judging_open-ended_interleaved_image-text_.md
!multimodal_vlm/optimus-2_multimodal_minecraft_agent_with_goal-observation-action_conditioned_po.md
!multimodal_vlm/parc_a_quantitative_framework_uncovering_the_symmetries_within_vision_language_m.md
!multimodal_vlm/peace_empowering_geologic_map_holistic_understanding_with_mllms.md
!multimodal_vlm/period-llm_extending_the_periodic_capability_of_multimodal_large_language_model.md
!multimodal_vlm/playing_the_fool_jailbreaking_llms_and_multimodal_llms_with_out-of-distribution_.md
!multimodal_vlm/post-pre-training_for_modality_alignment_in_vision-language_foundation_models.md
!multimodal_vlm/rap_retrieval-augmented_personalization_for_multimodal_large_language_models.md
!multimodal_vlm/realistic_test-time_adaptation_of_vision-language_models.md
!multimodal_vlm/reasoning_to_attend_try_to_understand_how_seg_token_works.md
!multimodal_vlm/recognition-synergistic_scene_text_editing.md
!multimodal_vlm/relation-rich_visual_document_generator_for_visual_information_extraction.md
!multimodal_vlm/rethinking_few-shot_adaptation_of_vision-language_models_in_two_stages.md
!multimodal_vlm/rethinking_vision-language_model_in_face_forensics_multi-modal_interpretable_for.md
!multimodal_vlm/rethinking_vlms_for_image_forgery_detection_and_localization.md
!multimodal_vlm/revisionllm_recursive_vision-language_model_for_temporal_grounding_in_hour-long_.md
!multimodal_vlm/revisiting_model_stitching_in_the_foundation_model_era.md
!multimodal_vlm/rlaif-v_open-source_ai_feedback_leads_to_super_gpt-4v_trustworthiness.md
!multimodal_vlm/robospatial_teaching_spatial_understanding_to_2d_and_3d_vision-language_models_f.md
!multimodal_vlm/scalable_video-to-dataset_generation_for_cross-platform_mobile_agents.md
!multimodal_vlm/seeing_the_abstract_translating_the_abstract_language_for_vision_language_models.md
!multimodal_vlm/segagent_exploring_pixel_understanding_capabilities_in_mllms_by_imitating_human_.md
!multimodal_vlm/self-evolving_visual_concept_library_using_vision-language_critics.md
!multimodal_vlm/self-supervised_spatial_correspondence_across_modalities.md
!multimodal_vlm/single_domain_generalization_for_few-shot_counting_via_universal_representation_.md
!multimodal_vlm/sketchagent_language-driven_sequential_sketch_generation.md
!multimodal_vlm/skip_tuning_pre-trained_vision-language_models_are_effective_and_efficient_adapt.md
!multimodal_vlm/sldprtnet_a_large-scale_multimodal_dataset_for_cad_generation_in_language-driven.md
!multimodal_vlm/smartclip_modular_vision-language_alignment_with_identification_guarantees.md
!multimodal_vlm/spa-vl_a_comprehensive_safety_preference_alignment_dataset_for_vision_language_m.md
!multimodal_vlm/sparrow_learning_spatial_precision_and_temporal_referential_consistency_in_pixel.md
!multimodal_vlm/starvector_generating_scalable_vector_graphics_code_from_images_and_text.md
!multimodal_vlm/stealthy_backdoor_attack_in_self-supervised_learning_vision_encoders_for_large_v.md
!multimodal_vlm/sting-bee_towards_vision-language_model_for_real-world_x-ray_baggage_security_in.md
!multimodal_vlm/svlta_benchmarking_vision-language_temporal_alignment_via_synthetic_video_situat.md
!multimodal_vlm/symdpo_boosting_in-context_learning_of_large_multimodal_models_with_symbol_demon.md
!multimodal_vlm/synthetic_data_is_an_elegant_gift_for_continual_vision-language_models.md
!multimodal_vlm/task_preference_optimization_improving_multimodal_large_language_models_with_vis.md
!multimodal_vlm/taxonomy-aware_evaluation_of_vision-language_models.md
!multimodal_vlm/teaching_large_language_models_to_regress_accurate_image_quality_scores_using_sc.md
!multimodal_vlm/topo-r1_detecting_topological_anomalies_via_vision-language_models.md
!multimodal_vlm/towards_understanding_how_knowledge_evolves_in_large_vision-language_models.md
!multimodal_vlm/unem_unrolled_generalized_em_for_transductive_few-shot_learning.md
!multimodal_vlm/unveiling_the_ignorance_of_mllms_seeing_clearly_answering_incorrectly.md
!multimodal_vlm/upme_an_unsupervised_peer_review_framework_for_multimodal_large_language_model_e.md
!multimodal_vlm/v-stylist_video_stylization_via_collaboration_and_reflection_of_mllm_agents.md
!multimodal_vlm/vidcomposition_can_mllms_analyze_compositions_in_compiled_videos.md
!multimodal_vlm/video-xl_extra-long_vision_language_model_for_hour-scale_video_understanding.md
!multimodal_vlm/videoglamm_a_large_multimodal_model_for_pixel-level_visual_grounding_in_videos.md
!multimodal_vlm/vila-m3_enhancing_vision-language_models_with_medical_expert_knowledge.md
!multimodal_vlm/vision-language_model_ip_protection_via_prompt-based_learning.md
!multimodal_vlm/vision-language_models_do_not_understand_negation.md
!multimodal_vlm/visionarena_230k_real_world_user-vlm_conversations_with_preference_labels.md
!multimodal_vlm/visionzip_longer_is_better_but_not_necessary_in_vision_language_models.md
!multimodal_vlm/visual_and_semantic_prompt_collaboration_for_generalized_zero-shot_learning.md
!multimodal_vlm/vladva_discriminative_fine-tuning_of_lvlms.md
!multimodal_vlm/vlsi_verbalized_layers-to-interactions_from_large_to_small_vision_language_model.md
!multimodal_vlm/whats_in_the_image_a_deep-dive_into_the_vision_of_vision_language_models.md
!multimodal_vlm/words_or_vision_do_vision-language_models_have_blind_faith_in_text.md
!multimodal_vlm/your_large_vision-language_model_only_needs_a_few_attention_heads_for_visual_gro.md
!nlp_generation/artformer_controllable_generation_of_diverse_3d_articulated_objects.md
!nlp_generation/dense_match_summarization_for_faster_two-view_estimation.md
!object_detection/aa-clip_enhancing_zero-shot_anomaly_detection_via_anomaly-aware_clip.md
!object_detection/abra_teleporting_fine-tuned_knowledge_across_domains_for_open-vocabulary_object_.md
!object_detection/anomalyncd_towards_novel_anomaly_class_discovery_in_industrial_scenarios.md
!object_detection/bacon_improving_clarity_of_image_captions_via_bag-of-concept_graphs.md
!object_detection/boosting_domain_incremental_learning_selecting_the_optimal_parameters_is_all_you.md
!object_detection/deim_detr_with_improved_matching_for_fast_convergence.md
!object_detection/distribution_prototype_diffusion_learning_for_open-set_supervised_anomaly_detect.md
!object_detection/efficient_event-based_object_detection_a_hybrid_neural_network_with_spatial_and_.md
!object_detection/efficient_test-time_adaptive_object_detection_via_sensitivity-guided_pruning.md
!object_detection/generalized_diffusion_detector_mining_robust_features_from_diffusion_models_for_.md
!object_detection/integration_of_deep_generative_anomaly_detection_algorithm_in_high-speed_industr.md
!object_detection/interpreting_object-level_foundation_models_via_visual_precision_search.md
!object_detection/large_self-supervised_models_bridge_the_gap_in_domain_adaptive_object_detection.md
!object_detection/mi-detr_an_object_detection_model_with_multi-time_inquiries_mechanism.md
!object_detection/mr_detr_instructive_multi-route_training_for_detection_transformers.md
!object_detection/mulsen_ad_multi_sensor_anomaly_detection.md
!object_detection/multi-sensor_object_anomaly_detection_unifying_appearance_geometry_and_internal_.md
!object_detection/multiple_object_tracking_as_id_prediction.md
!object_detection/object_detection_using_event_camera_a_moe_heat_conduction_based_detector_and_a_n.md
!object_detection/odd-one-out_anomaly_detection_by_comparing_with_neighbors.md
!object_detection/one-for-more_continual_diffusion_model_for_anomaly_detection.md
!object_detection/po3ad_predicting_point_offsets_toward_better_3d_point_cloud_anomaly_detection.md
!object_detection/probpose_a_probabilistic_approach_to_2d_human_pose_estimation.md
!object_detection/roictrl_boosting_instance_control_for_visual_generation.md
!object_detection/rsar_restricted_state_angle_resolver_and_rotated_sar_benchmark.md
!object_detection/search_and_detect_training-free_long_tail_object_detection_via_web-image_retriev.md
!object_detection/show_dont_tell_detecting_novel_objects_by_watching_human_videos.md
!object_detection/simltd_simple_supervised_and_semi-supervised_long-tailed_object_detection.md
!object_detection/small_target_detection_based_on_mask-enhanced_attention_fusion_of_visible_and_in.md
!object_detection/t2icount_enhancing_cross-modal_understanding_for_zero-shot_counting.md
!object_detection/tailedcore_few-shot_sampling_for_unsupervised_long-tail_noisy_anomaly_detection.md
!object_detection/test-time_backdoor_detection_for_object_detection_models.md
!object_detection/tornadonet_real-time_building_damage_detection_with_ordinal_supervision.md
!object_detection/towards_raw_object_detection_in_diverse_conditions.md
!object_detection/towards_zero-shot_anomaly_detection_and_reasoning_with_multimodal_large_language.md
!object_detection/univad_a_training-free_unified_model_for_few-shot_visual_anomaly_detection.md
!object_detection/unseen_visual_anomaly_generation.md
!object_detection/vcbench_a_streaming_counting_benchmark_for_spatial-temporal_state_maintenance_in.md
!optimization/automatic_joint_structured_pruning_and_quantization_for_efficient_neural_network.md
!optimization/conformal_prediction_for_zero-shot_models.md
!optimization/convex_relaxation_for_robust_vanishing_point_estimation_in_manhattan_world.md
!optimization/federated_learning_with_domain_shift_eraser.md
!optimization/how_to_merge_your_multimodal_models_over_time.md
!optimization/mind_the_gap_confidence_discrepancy_can_guide_federated_semi-supervised_learning.md
!optimization/model_poisoning_attacks_to_federated_learning_via_multi-round_consistency.md
!optimization/scope_semantic_coreset_with_orthogonal_projection_embeddings_for_federated_learn.md
!optimization/stop_walking_in_circles_bailing_out_early_in_projected_gradient_descent.md
!optimization/test-time_augmentation_improves_efficiency_in_conformal_prediction.md
!optimization/towards_stable_and_storage-efficient_dataset_distillation_matching_convexified_t.md
!others/bendfm_a_taxonomy_and_synthetic_cad_dataset_for_manufacturability_assessment_in_.md
!others/bounds_on_agreement_between_subjective_and_objective_measurements.md
!others/care_transformer_linear_attention.md
!others/deconstructing_the_failure_of_ideal_noise_correction_a_three-pillar_diagnosis.md
!others/do_imagenet-trained_models_learn_shortcuts_the_impact_of_frequency_shortcuts_on_.md
!others/ebs-ekf_accurate_and_high_frequency_event-based_star_tracking.md
!others/edm_equirectangular_projection-oriented_dense_kernelized_feature_matching.md
!others/effortless_active_labeling_for_long-term_test-time_adaptation.md
!others/event_ellipsometer_event-based_mueller-matrix_video_imaging.md
!others/evos_efficient_implicit_neural_training_via_evolutionary_selector.md
!others/exploring_contextual_attribute_density_in_referring_expression_counting.md
!others/feature_selection_for_latent_factor_models.md
!others/fiction_4d_future_interaction_prediction_from_video.md
!others/focal_split_untethered_snapshot_depth_from_differential_defocus.md
!others/foundations_of_the_theory_of_performance-based_ranking.md
!others/full-dof_egomotion_estimation_for_event_cameras_using_geometric_solvers.md
!others/gradient-guided_annealing_for_domain_generalization.md
!others/hotspot_signed_distance_function_optimization_with_an_asymptotically_sufficient_.md
!others/image_reconstruction_from_readout-multiplexed_single-photon_detector_arrays.md
!others/improving_accuracy_and_calibration_via_differentiated_deep_mutual_learning.md
!others/improving_transferable_targeted_attacks_with_feature_tuning_mixup.md
!others/instance-wise_supervision-level_optimization_in_active_learning.md
!others/integral_fast_fourier_color_constancy.md
!others/latte-mv_learning_to_anticipate_table_tennis_hits_from_monocular_videos.md
!others/locally_orderless_images_for_optimization_in_differentiable_rendering.md
!others/magicarticulate_make_your_3d_models_articulation-ready.md
!others/neisf_neural_incident_stokes_field_for_polarized_inverse_rendering_of_conductors.md
!others/on_the_generalization_of_handwritten_text_recognition_models.md
!others/open_set_label_shift_with_test_time_out-of-distribution_reference.md
!others/order-one_rolling_shutter_cameras.md
!others/pleas_-_merging_models_with_permutations_and_least_squares.md
!others/potential_field_based_deep_metric_learning.md
!others/practical_solutions_to_the_relative_pose_of_three_calibrated_cameras.md
!others/progressive_correspondence_regenerator_for_robust_3d_registration.md
!others/radio_frequency_ray_tracing_with_neural_object_representation_for_enhanced_rf_mo.md
!others/removing_reflections_from_raw_photos.md
!others/rethinking_epistemic_and_aleatoric_uncertainty_for_active_open-set_annotation_an.md
!others/rooftop_wind_field_reconstruction_using_sparse_sensors_from_deterministic_to_gen.md
!others/scene-agnostic_pose_regression_for_visual_localization.md
!others/sdf-net_structure-aware_disentangled_feature_learning_for_opticall-sar_ship_re-i.md
!others/strap-vit_segregated_tokens_with_randomized_--_transformations_for_defense_again.md
!others/subnet-aware_dynamic_supernet_training_for_neural_architecture_search.md
!others/sufficient_invariant_learning_for_distribution_shift.md
!others/taet_two-stage_adversarial_equalization_training_on_long-tailed_distributions.md
!others/tensoflow_tensorial_flow-based_sampler_for_inverse_rendering.md
!others/three-view_focal_length_recovery_from_homographies.md
!others/towards_in-the-wild_3d_plane_reconstruction_from_a_single_image.md
!others/towards_million-scale_adversarial_robustness_evaluation_with_stronger_individual.md
!others/traf-align_trajectory-aware_feature_alignment_for_asynchronous_multi-agent_perce.md
!others/training-free_neural_architecture_search_through_variance_of_knowledge_of_deep_n.md
!others/tuning_the_frequencies_robust_training_for_sinusoidal_neural_networks.md
!others/uncertainty_weighted_gradients_for_model_calibration.md
!others/uniphy_learning_a_unified_constitutive_model_for_inverse_physics_simulation.md
!others/vinabench_benchmark_for_faithful_and_consistent_visual_narratives.md
!others/wear_classification_of_abrasive_flap_wheels_using_a_hierarchical_deep_learning_a.md
!others/which_viewpoint_shows_it_best_language_for_weakly_supervising_view_selection_in_.md
!others/zero-shot_head_swapping_in_real-world_scenarios.md
!others/zo-sam_zero-order_sharpness-aware_minimization_for_efficient_sparse_training.md
!physics/accurate_differential_operators_for_hybrid_neural_fields.md
!physics/atp_adaptive_threshold_pruning_for_efficient_data_encoding_in_quantum_neural_net.md
!physics/difffno_diffusion_fourier_neural_operator.md
!physics/improve_representation_for_imbalanced_regression_through_geometric_constraints.md
!physics/kac_kolmogorov-arnold_classifier_for_continual_learning.md
!physics/learning_phase_distortion_with_selective_state_space_models_for_video_turbulence.md
!physics/towards_faithful_multimodal_concept_bottleneck_models.md
!recommender/finevq_fine-grained_user_generated_content_video_quality_assessment.md
!reinforcement_learning/calf_communication_aware_distributed_rl.md
!reinforcement_learning/gazing_at_rewards_eye_movements_as_a_lens_into_human_and_ai_decision-making_in_h.md
!reinforcement_learning/grove_a_generalized_reward_for_learning_open-vocabulary_physical_skill.md
!reinforcement_learning/skillmimic_learning_basketball_interaction_skills_from_demonstrations.md
!reinforcement_learning/thinking_in_streaming_video.md
!remote_sensing/dense_dispersed_structured_light_for_hyperspectral_3d_imaging_of_dynamic_scenes.md
!remote_sensing/disciple_learning_interpretable_programs_for_scientific_visual_discovery.md
!remote_sensing/earthdial_turning_multi-sensory_earth_observations_to_interactive_dialogues.md
!remote_sensing/hierarchical_dual-change_collaborative_learning_for_uav_scene_change_captioning.md
!remote_sensing/joint_and_streamwise_distributed_mimo_satellite_communications_with_multi-antenn.md
!remote_sensing/learning_occlusion-robust_vision_transformers_for_real-time_uav_tracking.md
!remote_sensing/meta-learning_hyperparameters_for_parameter_efficient_fine-tuning.md
!remote_sensing/metaspectra_a_compact_broadband_metasurface_camera_for_snapshot_hyperspectral_im.md
!remote_sensing/mfoghub_bridging_multi-regional_and_multi-satellite_data_for_global_marine_fog_d.md
!remote_sensing/sgformer_satellite-ground_fusion_for_3d_semantic_scene_completion.md
!remote_sensing/think_and_answer_me_benchmarking_and_exploring_multi-entity_reasoning_grounding_.md
!robotics/3d-mvp_3d_multiview_pretraining_for_manipulation.md
!robotics/a_data-centric_revisit_of_pre-trained_vision_models_for_robot_learning.md
!robotics/citywalker_learning_embodied_urban_navigation_from_web-scale_videos.md
!robotics/coordinated_manipulation_hybrid_deformable_rigid_objects.md
!robotics/cot-vla_visual_chain-of-thought_reasoning_for_vision-language-action_models.md
!robotics/decision_spikeformer_spike-driven_transformer_for_decision_making.md
!robotics/dexgrasp_anything_towards_universal_robotic_dexterous_grasping_with_physics_awar.md
!robotics/drawer_digital_reconstruction_and_articulation_with_environment_realism.md
!robotics/g3d-lf_generalizable_3d-language_feature_fields_for_embodied_tasks.md
!robotics/gigahands_a_massive_annotated_dataset_of_bimanual_hand_activities.md
!robotics/hearing_anywhere_in_any_environment.md
!robotics/language-grounded_decoupled_action_representation_for_robotic_manipulation.md
!robotics/learning_physics-based_full-body_human_reaching_and_grasping_from_brief_walking_.md
!robotics/let_humanoids_hike_integrative_skill_development_on_complex_trails.md
!robotics/lift3d_policy_lifting_2d_foundation_models_for_robust_3d_robotic_manipulation.md
!robotics/magma_a_foundation_model_for_multimodal_ai_agents.md
!robotics/maniptrans_efficient_dexterous_bimanual_manipulation_transfer_via_residual_learn.md
!robotics/manivideo_generating_hand-object_manipulation_video_with_dexterous_and_generaliz.md
!robotics/mitigating_the_human-robot_domain_discrepancy_in_visual_pre-training_for_robotic.md
!robotics/momanipvla_transferring_vision-language-action_models_for_general_mobile_manipul.md
!robotics/neural_motion_simulator_pushing_the_limit_of_world_models_in_reinforcement_learn.md
!robotics/overcoming_visual_clutter_in_vision_language_action_models_via_concept-gated_vis.md
!robotics/panoaffordancenet_towards_holistic_affordance_grounding_in_360_indoor_environmen.md
!robotics/perceive_what_matters_relevance-driven_scheduling_for_multimodal_streaming_perce.md
!robotics/phoenix_a_motion-based_self-reflection_framework_for_fine-grained_robotic_action.md
!robotics/prof_robot_differentiable_robot_rendering_without_static_and_self-collisions.md
!robotics/reasoning_in_visual_navigation_of_end-to-end_trained_agents_a_dynamical_systems_.md
!robotics/roboground_robotic_manipulation_with_grounded_vision-language_priors.md
!robotics/robotic_visual_instruction.md
!robotics/robotwin_dual-arm_robot_benchmark_with_generative_digital_twins.md
!robotics/sapave_towards_active_perception_and_manipulation_in_vision-language-action_mode.md
!robotics/showui_one_vision-language-action_model_for_gui_visual_agent.md
!robotics/solami_social_vision-language-action_modeling_for_immersive_interaction_with_3d_.md
!robotics/solving_instance_detection_from_an_open-world_perspective.md
!robotics/sortscrews_a_dataset_and_baseline_for_real-time_screw_classification.md
!robotics/think_small_act_big_primitive_prompt_learning_for_lifelong_robot_manipulation.md
!robotics/tinynav_end-to-end_tinyml_for_real-time_autonomous_navigation_on_microcontroller.md
!robotics/towards_long-horizon_vision-language_navigation_platform_benchmark_and_method.md
!robotics/universal_actions_for_enhanced_embodied_foundation_models.md
!robotics/zerograsp_zero-shot_shape_reconstruction_enabled_robotic_grasping.md
!segmentation/2dmamba_efficient_state_space_model_for_image_representation_with_applications_o.md
!segmentation/a_distractor-aware_memory_for_visual_object_tracking_with_sam2.md
!segmentation/assessing_and_learning_alignment_of_unimodal_vision_and_language_model.md
!segmentation/assessing_and_learning_alignment_of_unimodal_vision_and_language_models.md
!segmentation/audio-visual_instance_segmentation.md
!segmentation/binwang2hfnet_geogran-aware_hierarchical_feature_fusion_network_for_salient_obje.md
!segmentation/comparative_evaluation_of_traditional_methods_and_deep_learning_for_brain_glioma.md
!segmentation/condensing_action_segmentation_datasets_via_generative_network_inversion.md
!segmentation/continuous_locomotive_crowd_behavior_generation.md
!segmentation/cosmos_cross-modality_self-distillation_for_vision_language_pre-training.md
!segmentation/crossearth-sar_a_sar-centric_and_billion-scale_geospatial_foundation_model_for_d.md
!segmentation/da-vpt_semantic-guided_visual_prompt_tuning_for_vision_transformers.md
!segmentation/declip_decoupled_learning_for_open-vocabulary_dense_perception.md
!segmentation/defmamba_deformable_visual_state_space_model.md
!segmentation/dformerv2_geometry_self-attention_for_rgbd_semantic_segmentation.md
!segmentation/dinov2_meets_text_a_unified_framework_for_image-_and_pixel-level_vision-language.md
!segmentation/dpseg_dual-prompt_cost_volume_learning_for_open-vocabulary_semantic_segmentation.md
!segmentation/dual-agent_optimization_framework_for_cross-domain_few-shot_segmentation.md
!segmentation/dynamic_derivation_and_elimination_audio_visual_segmentation_with_enhanced_audio.md
!segmentation/edgetam_on-device_track_anything_model.md
!segmentation/editar_unified_conditional_generation_with_autoregressive_models.md
!segmentation/effective_sam_combination_for_open-vocabulary_semantic_segmentation.md
!segmentation/efficient_rgb-d_scene_understanding_via_multi-task_adaptive_learning_and_cross-d.md
!segmentation/exploiting_temporal_state_space_sharing_for_video_semantic_segmentation.md
!segmentation/exploring_clips_dense_knowledge_for_weakly_supervised_semantic_segmentation.md
!segmentation/exploring_simple_open-vocabulary_semantic_segmentation.md
!segmentation/f-lmm_grounding_frozen_large_multimodal_models.md
!segmentation/fine-grained_image-text_correspondence_with_cost_aggregation_for_open-vocabulary.md
!segmentation/finecaption_compositional_image_captioning_focusing_on_wherever_you_want_at_any_.md
!segmentation/foveated_instance_segmentation.md
!segmentation/fractal_calibration_for_long-tailed_object_detection.md
!segmentation/frequency_dynamic_convolution_for_dense_image_prediction.md
!segmentation/generative_video_propagation.md
!segmentation/glus_global-local_reasoning_unified_into_a_single_large_language_model_for_video.md
!segmentation/golden_cudgel_network_for_real-time_semantic_segmentation.md
!segmentation/groupmamba_efficient_group-based_visual_state_space_model.md
!segmentation/hfp-sam_hierarchical_frequency_prompted_sam_for_efficient_marine_animal_segmenta.md
!segmentation/hierarchical_compact_clustering_attention_coca_for_unsupervised_object-centric_l.md
!segmentation/id-patch_robust_id_association_for_group_photo_personalization.md
!segmentation/image_quality_assessment_from_human_to_machine_preference.md
!segmentation/learning_4d_panoptic_scene_graph_generation_from_rich_2d_visual_scene.md
!segmentation/livos_light_video_object_segmentation_with_gated_linear_matching.md
!segmentation/m3-vos_multi-phase_multi-transition_and_multi-scenery_video_object_segmentation.md
!segmentation/mambaout_do_we_really_need_mamba_for_vision.md
!segmentation/mambavision_a_hybrid_mamba-transformer_vision_backbone.md
!segmentation/mammalps_a_multi-view_video_behavior_monitoring_dataset_of_wild_mammals_in_the_s.md
!segmentation/mask-adapter_the_devil_is_in_the_masks_for_open-vocabulary_segmentation.md
!segmentation/mass13k_a_matting-level_semantic_segmentation_benchmark.md
!segmentation/matanyone_stable_video_matting_with_consistent_memory_propagation.md
!segmentation/mv-ssm_multi-view_state_space_modeling_for_3d_human_pose_estimation.md
!segmentation/overlock_an_overview-first-look-closely-next_convnet_with_context-mixing_dynamic.md
!segmentation/paint_by_inpaint_learning_to_add_image_objects_by_removing_them_first.md
!segmentation/picosam3_real-time_in-sensor_region-of-interest_segmentation.md
!segmentation/posta_a_go-to_framework_for_customized_artistic_poster_generation.md
!segmentation/prompt-driven_lightweight_foundation_model_for_instance_segmentation-based_fault.md
!segmentation/rdnet_region_proportion-aware_dynamic_adaptive_salient_object_detection_network_.md
!segmentation/resclip_residual_attention_for_training-free_dense_vision-language_inference.md
!segmentation/rethinking_query-based_transformer_for_continual_image_segmentation.md
!segmentation/revisiting_audio-visual_segmentation_with_vision-centric_transformer.md
!segmentation/ripvis_rip_currents_video_instance_segmentation_benchmark_for_beach_monitoring_a.md
!segmentation/robust_3d_shape_reconstruction_in_zero-shot_from_a_single_image_in_the_wild.md
!segmentation/robust_audio-visual_segmentation_via_audio-guided_visual_convergent_alignment.md
!segmentation/rocket-1_mastering_open-world_interaction_with_visual-temporal_context_prompting.md
!segmentation/ros-sam_high-quality_interactive_segmentation_for_remote_sensing_moving_object.md
!segmentation/rsonet_region-guided_selective_optimization_network_for_rgb-t_salient_object_det.md
!segmentation/sam2-love_segment_anything_model_2_in_language-aided_audio-visual_scenes.md
!segmentation/samwise_infusing_wisdom_in_sam2_for_text-driven_video_segmentation.md
!segmentation/sap_segment_any_4k_panorama.md
!segmentation/scale_efficient_training_for_large_datasets.md
!segmentation/scene-centric_unsupervised_panoptic_segmentation.md
!segmentation/segment_any-quality_images_with_generative_latent_space_enhancement.md
!segmentation/segment_any_motion_in_videos.md
!segmentation/semantic_library_adaptation_lora_retrieval_and_fusion_for_open-vocabulary_semant.md
!segmentation/sgma_semantic-guided_modality-aware_segmentation_for_remote_sensing_with_incompl.md
!segmentation/shiftwiseconv_small_convolutional_kernel_with_large_kernel_effect.md
!segmentation/show_and_tell_visually_explainable_deep_neural_nets_via_spatially-aware_concept_.md
!segmentation/sketchfusion_learning_universal_sketch_features_through_fusing_foundation_models.md
!segmentation/smarteraser_remove_anything_from_images_using_masked-region_guidance.md
!segmentation/soft_self-labeling_and_potts_relaxations_for_weakly-supervised_segmentation.md
!segmentation/spatio-semantic_expert_routing_architecture_with_mixture-of-experts_for_referrin.md
!segmentation/storygpt-v_large_language_models_as_consistent_story_visualizers.md
!segmentation/style-editor_text-driven_object-centric_style_editing.md
!segmentation/task-driven_image_fusion_with_learnable_fusion_loss.md
!segmentation/the_devil_is_in_low-level_features_for_cross-domain_few-shot_segmentation.md
!segmentation/the_devil_is_in_temporal_token_high_quality_video_reasoning_segmentation.md
!segmentation/the_power_of_context_how_multimodality_improves_image_super-resolution.md
!segmentation/token_cropr_faster_vits_for_quite_a_few_tasks.md
!segmentation/towards_generalizable_scene_change_detection.md
!segmentation/uni4d_unifying_visual_foundation_models_for_4d_modeling_from_a_single_video.md
!segmentation/universal_domain_adaptation_for_semantic_segmentation.md
!segmentation/using_diffusion_priors_for_video_amodal_segmentation.md
!segmentation/v-clr_view-consistent_learning_for_open-world_instance_segmentation.md
!segmentation/visual_consensus_prompting_for_co-salient_object_detection.md
!segmentation/your_vit_is_secretly_an_image_segmentation_model.md
!self_supervised/autossvh_exploring_automated_frame_sampling_for_efficient_self-supervised_video_.md
!self_supervised/boss_a_best-of-strategies_selector_as_an_oracle_for_deep_active_learning.md
!self_supervised/breaking_the_tuning_barrier_zero-hyperparameters_yield_multi-corner_analysis_via.md
!self_supervised/chexworld_exploring_image_world_modeling_for_radiograph_representation_learning.md
!self_supervised/do_your_best_and_get_enough_rest_for_continual_learning.md
!self_supervised/escaping_platos_cave_towards_the_alignment_of_3d_and_text_latent_spaces.md
!self_supervised/few-shot_implicit_function_generation_via_equivariance.md
!self_supervised/from_prototypes_to_general_distributions_an_efficient_curriculum_for_masked_imag.md
!self_supervised/hyperbolic_category_discovery.md
!self_supervised/learning_to_normalize_on_the_spd_manifold_under_bures-wasserstein_geometry.md
!self_supervised/map_unleashing_hybrid_mamba-transformer_vision_backbones_potential_with_masked_a.md
!self_supervised/mari_material_retrieval_integration_across_domains.md
!self_supervised/metawriter_personalized_handwritten_text_recognition_using_meta-learned_prompt_t.md
!self_supervised/mos_modeling_object-scene_associations_in_generalized_category_discovery.md
!self_supervised/ocrt_boosting_foundation_models_in_the_open_world_with_object-concept-relation_t.md
!self_supervised/order-robust_class_incremental_learning_graph-driven_dynamic_similarity_grouping.md
!self_supervised/representation_learning_for_spatiotemporal_physical_systems.md
!self_supervised/sata_spatial_autocorrelation_token_analysis_for_enhancing_the_robustness_of_visi.md
!self_supervised/scalelsd_scalable_deep_line_segment_detection_streamlined.md
!self_supervised/sec-promptsemantic_complementary_prompting_for_few-shot_class-incremental_learni.md
!self_supervised/smile_infusing_spatial_and_motion_semantics_in_masked_video_learning.md
!self_supervised/spectral_state_space_model_for_rotation-invariant_visual_representation_learning.md
!self_supervised/task-agnostic_guided_feature_expansion_for_class-incremental_learning.md
!self_supervised/text-phase_synergy_network_with_dual_priors_for_unsupervised_cross-domain_image_.md
!self_supervised/transformers_without_normalization.md
!self_supervised/unistd_towards_unified_spatio-temporal_learning_across_diverse_disciplines.md
!signal_comm/abc-former_auxiliary_bimodal_cross-domain_transformer_with_interactive_channel_a.md
!signal_comm/breaking_the_low-rank_dilemma_of_linear_attention.md
!signal_comm/continuous_space-time_video_resampling_with_invertible_motion_steganography.md
!signal_comm/ditask_multi-task_fine-tuning_with_diffeomorphic_transformations.md
!signal_comm/neural_video_compression_with_context_modulation.md
!social_computing/as_language_models_scale_low-order_linear_depth_dynamics_emerge.md
!social_computing/classifier-guided_clip_distillation_for_unsupervised_multi-label_classification.md
!social_computing/classifier-to-bias_toward_unsupervised_automatic_bias_detection_for_visual_class.md
!social_computing/learning_from_neighbors_category_extrapolation_for_long-tail_learning.md
!social_computing/let_samples_speak_mitigating_spurious_correlation_by_exploiting_the_clusterness_.md
!social_computing/project-probe-aggregate_efficient_fine-tuning_for_group_robustness.md
!time_series/competition-aware_cpc_forecasting_with_near-market_coverage.md
!time_series/dejavid_encoder-agnostic_learned_temporal_matching_for_video_classification.md
!time_series/flavc_learned_video_compression_with_feature_level_attention.md
!time_series/l2gtx_from_local_to_global_time_series_explanations.md
!time_series/learning_extremely_high_density_crowds_as_active_matters.md
!video_generation/4real-video_learning_generalizable_photo-realistic_4d_video_diffusion.md
!video_generation/animateanything_consistent_and_controllable_animation_for_video_generation.md
!video_generation/articulated_kinematics_distillation_from_video_diffusion_models.md
!video_generation/bf-stvsr_b-splines_and_fourier---best_friends_for_high_fidelity_spatia.md
!video_generation/can_text-to-video_generation_help_video-language_alignment.md
!video_generation/conmo_controllable_motion_disentanglement_and_recomposition_for_zero-shot_motion.md
!video_generation/dynamic_camera_poses_and_where_to_find_them.md
!video_generation/dynamicscaler_panoramic_video.md
!video_generation/dynamicscaler_seamless_and_scalable_video_generation_for_panoramic_scenes.md
!video_generation/exploring_temporally-aware_features_for_point_tracking.md
!video_generation/fade_frequency-aware_diffusion_model_factorization_for_video_editing.md
!video_generation/flashmotion_few-step_controllable_video_generation_with_trajectory_guidance.md
!video_generation/from_slow_bidirectional_to_fast_autoregressive_video_diffusion_models.md
!video_generation/gen3c_3d-informed_world-consistent_video_generation_with_precise_camera_control.md
!video_generation/generative_inbetweening_through_frame-wise_conditions-driven_video_generation.md
!video_generation/geometry-guided_online_3d_video_synthesis_with_multi-view_temporal_consistency.md
!video_generation/hoigen-1m_a_large-scale_dataset_for_human-object_interaction_video_generation.md
!video_generation/hunyuanportrait_implicit_condition_control_for_enhanced_portrait_animation.md
!video_generation/hypernvd_accelerating_neural_video_decomposition_via_hypernetworks.md
!video_generation/identity-preserving_text-to-video_generation_by_frequency_decomposition.md
!video_generation/idol_instant_photorealistic_3d_human_creation_from_a_single_image.md
!video_generation/improved_video_vae_for_latent_video_diffusion_model.md
!video_generation/interdyn_controllable_interactive_dynamics_with_video_diffusion_models.md
!video_generation/learning_from_streaming_video_with_orthogonal_gradients.md
!video_generation/learning_temporally_consistent_video_depth_from_video_diffusion_priors.md
!video_generation/levitor_3d_trajectory_oriented_image-to-video_synthesis.md
!video_generation/long_video_diffusion_generation_with_segmented_cross-attention_and_content-rich_.md
!video_generation/longdiff_training-free_long_video_generation_in_one_go.md
!video_generation/mimir_improving_video_diffusion_models_for_precise_text_understanding.md
!video_generation/mimo_controllable_character_video_synthesis_with_spatial_decomposed_modeling.md
!video_generation/mind_the_time_temporally-controlled_multi-event_video_generation.md
!video_generation/motif_making_text_count_in_image_animation_with_motion_focal_loss.md
!video_generation/motion_modes_what_could_happen_next.md
!video_generation/motion_prompting_controlling_video_generation_with_motion_trajectories.md
!video_generation/motionpro_a_precise_motion_controller_for_image-to-video_generation.md
!video_generation/motionstone_decoupled_motion_intensity_modulation_with_diffusion_transformer_for.md
!video_generation/moviebench_a_hierarchical_movie_level_dataset_for_long_video_generation.md
!video_generation/multi-subject_open-set_personalization_in_video_generation.md
!video_generation/navigation_world_models.md
!video_generation/neuro-symbolic_evaluation_of_text-to-video_models_using_formal_verification.md
!video_generation/one-minute_video_generation_with_test-time_training.md
!video_generation/optical-flow_guided_prompt_optimization_for_coherent_video_generation.md
!video_generation/osv_one_step_is_enough_for_high-quality_image_to_video_generation.md
!video_generation/out_of_sight_out_of_mind_evaluating_state_evolution_in_video_world_models.md
!video_generation/parallelized_autoregressive_visual_generation.md
!video_generation/patchvsr_breaking_video_diffusion_resolution_limits_with_patch-wise_video_super-.md
!video_generation/pathways_on_the_image_manifold_image_editing_via_video_generation.md
!video_generation/phyt2v_llm-guided_iterative_self-refinement_for_physics-grounded_text-to-video_g.md
!video_generation/posetraj_pose-aware_trajectory_control_in_video_diffusion.md
!video_generation/recapture_generative_video_camera_controls_for_user-provided_videos_using_masked.md
!video_generation/saw_toward_a_surgical_action_world_model_via_controllable_and_scalable_video_gen.md
!video_generation/semantic_satellite_communications_for_synchronized_audiovisual_reconstruction.md
!video_generation/shotadapter_text-to-multi-shot_video_generation_with_diffusion_models.md
!video_generation/sketchvideo_sketch-based_video_generation_and_editing.md
!video_generation/spatiotemporal_skip_guidance_for_enhanced_video_diffusion_sampling.md
!video_generation/streamingt2v_consistent_dynamic_and_extendable_long_video_generation_from_text.md
!video_generation/streetcrafter_street_view_synthesis_with_controllable_video_diffusion_models.md
!video_generation/taming_teacher_forcing_for_masked_autoregressive_video_generation.md
!video_generation/teller_real-time_streaming_audio-driven_portrait_animation_with_autoregressive_m.md
!video_generation/the_devil_is_in_the_prompts_retrieval-augmented_prompt_optimization_for_text-to-.md
!video_generation/through-the-mask_mask-based_motion_trajectories_for_image-to-video_generation.md
!video_generation/timestep_embedding_tells_its_time_to_cache_for_video_diffusion_model.md
!video_generation/tokenmotion_decoupled_motion_control_via_token_disentanglement_for_human-centric.md
!video_generation/tora_trajectory-oriented_diffusion_transformer_for_video_generation.md
!video_generation/towards_precise_scaling_laws_for_video_diffusion_transformers.md
!video_generation/tracktention_leveraging_point_tracking_to_attend_videos_faster_and_better.md
!video_generation/transpixeler_advancing_text-to-video_generation_with_transparency.md
!video_generation/unified_dense_prediction_of_video_diffusion.md
!video_generation/veu-bench_towards_comprehensive_understanding_of_video_editing.md
!video_generation/video-bench_human-aligned_video_generation_benchmark.md
!video_generation/video-colbert_contextualized_late_interaction_for_text-to-video_retrieval.md
!video_generation/video_motion_transfer_with_diffusion_transformers.md
!video_generation/videodirector_precise_video_editing_via_text-to-video_models.md
!video_generation/videodpo_omni-preference_alignment_for_video_diffusion_generation.md
!video_generation/videogigagan_towards_detail-rich_video_super-resolution.md
!video_generation/videoguide_improving_video_diffusion_models_without_training_through_a_teachers_.md
!video_generation/videoscene_distilling_video_diffusion_model_to_generate_3d_scenes_in_one_step.md
!video_generation/vidtwin_video_vae_with_decoupled_structure_and_dynamics.md
!video_generation/vires_video_instance_repainting_via_sketch_and_text_guided_generation.md
!video_generation/visual_prompting_for_one-shot_controllable_video_editing_without_inversion.md
!video_generation/wav2sem_plug-and-play_audio_semantic_decoupling_for_3d_speech-driven_facial_anim.md
!video_generation/when_to_lock_attention_training-free_kv_control_in_video_diffusion.md
!video_generation/world-consistent_video_diffusion_with_explicit_3d_modeling.md
!video_generation/world2act_latent_action_post-training_via_skill-compositional_world_models.md
!video_generation/zero-1-to-a_zero-shot_one_image_to_animatable_head_avatars_using_video_diffusion.md
!video_understanding/anomize_better_open_vocabulary_video_anomaly_detection.md
!video_understanding/behaviorvlm_unified_finetuning-free_behavioral_understanding_with_vision-languag.md
!video_understanding/beyond_single-sample_reliable_multi-sample_distillation_for_video_understanding.md
!video_understanding/bim-vfi_bidirectional_motion_field-guided_frame_interpolation_for_video_with_non.md
!video_understanding/bimba_selective-scan_compression_for_long-range_video_question_answering.md
!video_understanding/bootstrap_your_own_views_masked_ego-exo_modeling_for_fine-grained_view-invariant.md
!video_understanding/context-enhanced_memory-refined_transformer_for_online_action_detection.md
!video_understanding/cross-modal_causal_relation_alignment_for_video_question_grounding.md
!video_understanding/decafnet_delegate_and_conquer_for_efficient_temporal_grounding_in_long_videos.md
!video_understanding/divprune_diversity-based_visual_token_pruning_for_large_multimodal_models.md
!video_understanding/dpflow_adaptive_optical_flow_estimation_with_a_dual-pyramid_framework.md
!video_understanding/dpu_dynamic_prototype_updating_for_multimodal_out-of-distribution_detection.md
!video_understanding/drvideo_document_retrieval_based_long_video_understanding.md
!video_understanding/dynfocus_dynamic_cooperative_network_empowers_llms_with_video_understanding.md
!video_understanding/edcflow_exploring_temporally_dense_difference_maps_for_event-based_optical_flow_.md
!video_understanding/efficient_transfer_learning_for_video-language_foundation_models.md
!video_understanding/egolife_towards_egocentric_life_assistant.md
!video_understanding/egotextvqa_towards_egocentric_scene-text_aware_video_question_answering.md
!video_understanding/etap_event-based_tracking_of_any_point.md
!video_understanding/expertaf_expert_actionable_feedback_from_video.md
!video_understanding/fc-track_overlap-aware_post-association_correction_for_online_multi-object_track.md
!video_understanding/frame_floor-aligned_representation_for_avatar_motion_from_egocentric_video.md
!video_understanding/fsbench_a_figure_skating_benchmark_for_advancing_artistic_sports_understanding.md
!video_understanding/gg-ssms_graph-generating_state_space_models.md
!video_understanding/h-more_learning_human-centric_motion_representation_for_action_analysis.md
!video_understanding/heterogeneous_skeleton-based_action_representation_learning.md
!video_understanding/hierarq_task-aware_hierarchical_q-former_for_enhanced_video_understanding.md
!video_understanding/holmes-vau_towards_long-term_video_anomaly_understanding_at_any_granularity.md
!video_understanding/humocon_concept_discovery_for_human_motion_understanding.md
!video_understanding/hyperglm_hypergraph_for_video_scene_graph_generation_and_anticipation.md
!video_understanding/learning_audio-guided_video_representation_with_gated_attention_for_video-text_r.md
!video_understanding/lion-fs_fast_slow_video-language_thinker_as_online_video_assistant.md
!video_understanding/llavidal_a_large_language_vision_model_for_daily_activities_of_living.md
!video_understanding/localizing_events_in_videos_with_multimodal_queries.md
!video_understanding/m-llm_based_video_frame_selection_for_efficient_video_understanding.md
!video_understanding/mambavlt_time-evolving_multimodal_state_space_model_for_vision-language_tracking.md
!video_understanding/mlvu_benchmarking_multi-task_long_video_understanding.md
!video_understanding/mmvu_measuring_expert-level_multi-discipline_video_understanding.md
!video_understanding/must_the_first_dataset_and_unified_framework_for_multispectral_uav_single_object.md
!video_understanding/number_it_temporal_grounding_videos_like_flipping_manga.md
!video_understanding/object-shot_enhanced_grounding_network_for_egocentric_video.md
!video_understanding/omni-rgpt_unifying_image_and_video_region-level_understanding_via_token_marks.md
!video_understanding/omnidirectional_multi-object_tracking.md
!video_understanding/on_the_consistency_of_video_large_language_models_in_temporal_comprehension.md
!video_understanding/ovo-bench_how_far_is_your_video-llms_from_real-world_online_video_understanding.md
!video_understanding/pave_patching_and_adapting_video_large_language_models.md
!video_understanding/progress-aware_video_frame_captioning.md
!video_understanding/q-bench-video_benchmark_the_video_quality_understanding_of_lmms.md
!video_understanding/question-aware_gaussian_experts_for_audio-visual_question_answering.md
!video_understanding/re-thinking_temporal_search_for_long-form_video_understanding.md
!video_understanding/rewind_understanding_long_videos_with_instructed_learnable_memory.md
!video_understanding/seal_semantic_attention_learning_for_long_video_representation.md
!video_understanding/seq2time_sequential_knowledge_transfer_for_video_llm_temporal_grounding.md
!video_understanding/seriesbench_a_benchmark_for_narrative-driven_drama_series_understanding.md
!video_understanding/similarity-guided_layer-adaptive_vision_transformer_for_uav_tracking.md
!video_understanding/stop_integrated_spatial-temporal_dynamic_prompting_for_video_understanding.md
!video_understanding/tamt_temporal-aware_model_tuning_for_cross-domain_few-shot_action_recognition.md
!video_understanding/temporal_alignment-free_video_matching_for_few-shot_action_recognition.md
!video_understanding/temporally_consistent_object-centric_learning_by_contrasting_slots.md
!video_understanding/towards_universal_soccer_video_understanding.md
!video_understanding/unbiasing_through_textual_descriptions_mitigating_representation_bias_in_video_b.md
!video_understanding/video-panda_parameter-efficient_alignment_for_encoder-free_video-language_models.md
!video_understanding/video_streaming_thinking_videollms_can_watch_and_think_simultaneously.md
!video_understanding/video_summarization_with_large_language_models.md
!video_understanding/videogem_training-free_action_grounding_in_videos.md
!video_understanding/videorefer_suite_advancing_spatial-temporal_object_understanding_with_video_llm.md
!video_understanding/vista_enhancing_long-duration_and_high-resolution_video_understanding_by_video_s.md
!video_understanding/vited_video_temporal_evidence_distillation.md
!video_understanding/voco-llama_towards_vision_compression_with_large_language_models.md
!vlm_efficiency/coap_memory-efficient_training_with_correlation-aware_gradient_projection.md
!vlm_efficiency/mbq_modality-balanced_quantization_for_large_vision-language_models.md
!vlm_efficiency/quantization_without_tears.md
!vlm_reasoning/beyond_final_answers_crystal_benchmark_for_transparent_multimodal_reasoning_eval.md
!vlm_reasoning/coarse_correspondences_boost_spatial-temporal_reasoning_in_multimodal_language_m.md
!vlm_reasoning/critic-v_vlm_critics_help_catch_vlm_errors_in_multimodal_reasoning.md
!vlm_reasoning/document_haystacks_vision-language_reasoning_over_piles_of_1000_documents.md
!vlm_reasoning/espire_a_diagnostic_benchmark_for_embodied_spatial_reasoning_of_vision-language_.md
!vlm_reasoning/insight-v_exploring_long-chain_visual_reasoning_with_multimodal_large_language_m.md
!vlm_reasoning/mm-condchain_a_programmatically_verified_benchmark_for_visually_grounded_deep_co.md
!vlm_reasoning/mv-math_evaluating_multimodal_math_reasoning_in_multi-visual_contexts.md
!vlm_reasoning/reasoning_over_video_evaluating_how_mllms_extract_integrate_and_reconstruct_spat.md
!vlm_reasoning/seqafford_sequential_3d_affordance_reasoning_via_multimodal_large_language_model.md
!vlm_reasoning/spatial_reasoning_is_not_a_free_lunch_a_controlled_study_on_llava.md
!vlm_reasoning/thinking_in_dynamics_how_multimodal_large_language_models_perceive_track_and_rea.md
!vlm_reasoning/thinking_in_space_how_multimodal_large_language_models_see_remember_and_recall_s.md
