The representation learning the other two threads are built on.
The problem. Scientific data is structured, sparse, expensive to label and captured through several complementary sensors or assays at once. Learning useful representations from it means confronting all four properties together, whether the object is a molecule or a street.
What I build. Methods for learning from structured 3D data under weak supervision. (Yin et al., 2022) pre-trains detectors on unlabelled point clouds by contrasting region proposals; (Yin et al., 2022) and (Wang et al., 2023) push the same objective into the semi-supervised and domain-adaptive settings, where labels exist but not for the distribution you care about; (Li et al., 2023) supervises segmentation from cheaper modalities instead of dense masks. On the multimodal side, (Yin et al., 2024) fuses camera and LiDAR at both the instance and the scene level, and (Yin et al., 2023) models temporal structure with graph message passing and spatiotemporal attention.
Where it is going. These techniques were developed for perception, and the transfer to molecular systems is direct: pre-training when labels are scarce, adapting across distributions, and fusing modalities that each see part of the object. That transfer is what made the move into protein design a continuation rather than a restart.
@inproceedings{yin2024isfusion,title={IS-Fusion: Instance-Scene Collaborative Fusion for Multimodal 3D Object Detection},author={Yin, Junbo and Shen, Jianbing and Chen, Runnan and Li, Wei and Yang, Ruigang and Frossard, Pascal and Wang, Wenguan},booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},year={2024}}
@inproceedings{wang2023ssda3d,title={SSDA3D: Semi-supervised Domain Adaptation for 3D Object Detection from Point Cloud},author={Wang, Yan and Yin, Junbo and Li, Wei and Frossard, Pascal and Yang, Ruigang and Shen, Jianbing},note={* Equal contribution (Y. Wang, J. Yin).},booktitle={AAAI Conference on Artificial Intelligence (AAAI)},pages={2707--2715},year={2023}}
@inproceedings{li2023lwsis,title={LWSIS: LiDAR-Guided Weakly Supervised Instance Segmentation for Autonomous Driving},author={Li, Xiang and Yin, Junbo and Shi, Botian and Li, Yikang and Yang, Ruigang and Shen, Jianbing},note={* Equal contribution (X. Li, J. Yin).},booktitle={AAAI Conference on Artificial Intelligence (AAAI)},pages={1433--1441},year={2023}}
@article{yin2023tpami,title={Graph Neural Network and Spatiotemporal Transformer Attention for 3D Video Object Detection from Point Clouds},author={Yin, Junbo and Shen, Jianbing and Gao, Xin and Crandall, David and Yang, Ruigang},journal={IEEE Transactions on Pattern Analysis and Machine Intelligence},volume={45},number={8},pages={9822--9835},year={2023}}
@inproceedings{yin2022proposalcontrast,title={ProposalContrast: Unsupervised Pre-training for LiDAR-based 3D Object Detection},author={Yin, Junbo and Zhou, Dingfu and Zhang, Liangjun and Fang, Jin and Xu, Cheng-Zhong and Shen, Jianbing and Wang, Wenguan},booktitle={European Conference on Computer Vision (ECCV)},pages={17--33},year={2022}}
@inproceedings{yin2022proficient,title={Semi-supervised 3D Object Detection with Proficient Teachers},author={Yin, Junbo and Fang, Jin and Zhou, Dingfu and Zhang, Liangjun and Xu, Cheng-Zhong and Shen, Jianbing and Wang, Wenguan},booktitle={European Conference on Computer Vision (ECCV)},pages={727--743},year={2022}}