{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T15:38:45Z","timestamp":1773157125012,"version":"3.50.1"},"reference-count":56,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,11,1]],"date-time":"2025-11-01T00:00:00Z","timestamp":1761955200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62403201"],"award-info":[{"award-number":["62403201"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai Municipality","doi-asserted-by":"publisher","award":["23ZR1414900"],"award-info":[{"award-number":["23ZR1414900"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai Municipality","doi-asserted-by":"publisher","award":["24ZR1415200"],"award-info":[{"award-number":["24ZR1415200"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100007219","name":"Natural Science Foundation of Shanghai Municipality","doi-asserted-by":"publisher","award":["22ZR1416500"],"award-info":[{"award-number":["22ZR1416500"]}],"id":[{"id":"10.13039\/100007219","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003021","name":"East China University of Science and Technology","doi-asserted-by":"publisher","award":["JKH01251821"],"award-info":[{"award-number":["JKH01251821"]}],"id":[{"id":"10.13039\/501100003021","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Engineering Applications of Artificial Intelligence"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1016\/j.engappai.2025.111919","type":"journal-article","created":{"date-parts":[[2025,8,20]],"date-time":"2025-08-20T13:08:28Z","timestamp":1755695308000},"page":"111919","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":1,"special_numbering":"PC","title":["Multi-modal self-supervised contrastive representation learning for three-dimensional point cloud understanding"],"prefix":"10.1016","volume":"160","author":[{"given":"Weichao","family":"Ding","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zehao","family":"Yang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fei","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chunhua","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9451-8502","authenticated-orcid":false,"given":"Wenbo","family":"Dong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.engappai.2025.111919_b1","doi-asserted-by":"crossref","unstructured":"Afham, M., Dissanayake, I., Dissanayake, D., Dharmasiri, A., Thilakarathna, K., Rodrigo, R., 2022. Crosspoint: Self-supervised cross-modal contrastive learning for 3D point cloud understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 9902\u20139912.","DOI":"10.1109\/CVPR52688.2022.00967"},{"key":"10.1016\/j.engappai.2025.111919_b2","series-title":"Advances in Neural Information Processing Systems","first-page":"24206","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","author":"Akbari","year":"2021"},{"key":"10.1016\/j.engappai.2025.111919_b3","doi-asserted-by":"crossref","first-page":"770","DOI":"10.1080\/13588265.2022.2130608","article-title":"Vehicle collisions analysis on highways based on multi-user driving simulator and multinomial logistic regression model on US highways in michigan","volume":"28","author":"Almadi","year":"2023","journal-title":"Int. J. Crashworthiness"},{"key":"10.1016\/j.engappai.2025.111919_b4","doi-asserted-by":"crossref","first-page":"1280","DOI":"10.1108\/CI-10-2021-0196","article-title":"Structural performance of buried reinforced concrete pipelines under deep embankment soil","volume":"24","author":"Almasabha","year":"2024","journal-title":"Constr. Innov."},{"key":"10.1016\/j.engappai.2025.111919_b5","doi-asserted-by":"crossref","unstructured":"Behley, J., Garbade, M., Milioto, A., Quenzel, J., Behnke, S., Stachniss, C., Gall, J., 2019. Semantickitti: A dataset for semantic scene understanding of lidar sequences. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9297\u20139307.","DOI":"10.1109\/ICCV.2019.00939"},{"key":"10.1016\/j.engappai.2025.111919_b6","series-title":"Advances in Neural Information Processing Systems","first-page":"1","article-title":"Pointgpt: Auto-regressively generative pre-training from point clouds","author":"Chen","year":"2024"},{"key":"10.1016\/j.engappai.2025.111919_b7","doi-asserted-by":"crossref","DOI":"10.1016\/j.engappai.2023.107529","article-title":"Self-supervised rotation-equivariant spherical vector network for learning canonical 3D point cloud orientation","volume":"128","author":"Chen","year":"2024","journal-title":"Eng. Appl. Artif. Intell."},{"key":"10.1016\/j.engappai.2025.111919_b8","series-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2018"},{"key":"10.1016\/j.engappai.2025.111919_b9","doi-asserted-by":"crossref","unstructured":"Gao, Y., Wang, Z., Zheng, W.-S., Xie, C., Zhou, Y., 2024. Sculpting holistic 3D representation in contrastive language-image-3D pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 22998\u201323008.","DOI":"10.1109\/CVPR52733.2024.02170"},{"key":"10.1016\/j.engappai.2025.111919_b10","doi-asserted-by":"crossref","first-page":"187","DOI":"10.1007\/s41095-021-0229-5","article-title":"Pct: Point cloud transformer","volume":"7","author":"Guo","year":"2021","journal-title":"Comput. Vis. Media"},{"issue":"5","key":"10.1016\/j.engappai.2025.111919_b11","first-page":"5436","article-title":"Beyond self-attention: External attention using two linear layers for visual tasks","volume":"45","author":"Guo","year":"2022","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2025.111919_b12","doi-asserted-by":"crossref","unstructured":"Hamdi, A., Giancola, S., Ghanem, B., 2021. Mvtn: Multi-view transformation network for 3D shape recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 1\u201311.","DOI":"10.1109\/ICCV48922.2021.00007"},{"key":"10.1016\/j.engappai.2025.111919_b13","doi-asserted-by":"crossref","unstructured":"He, S., Ding, H., Jiang, X., Wen, B., 2025. Segpoint: Segment any point cloud via large language model. In: European Conference on Computer Vision. pp. 349\u2013367.","DOI":"10.1007\/978-3-031-72670-5_20"},{"key":"10.1016\/j.engappai.2025.111919_b14","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J., 2016. Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.engappai.2025.111919_b15","doi-asserted-by":"crossref","unstructured":"Huang, T., Dong, B., Yang, Y., Huang, X., Lau, R.W., Ouyang, W., Zuo, W., 2023a. CLIP2Point: Transfer CLIP to Point Cloud Classification with Image-Depth Pre-Training. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 22157\u201322167.","DOI":"10.1109\/ICCV51070.2023.02025"},{"key":"10.1016\/j.engappai.2025.111919_b16","doi-asserted-by":"crossref","unstructured":"Huang, Z., Ren, Y., Pu, X., Huang, S., Xu, Z., He, L., 2023b. Self-supervised graph attention networks for deep weighted multi-view clustering. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 7936\u20137943.","DOI":"10.1609\/aaai.v37i7.25960"},{"key":"10.1016\/j.engappai.2025.111919_b17","doi-asserted-by":"crossref","unstructured":"Huang, S., Xie, Y., Zhu, S.-C., Zhu, Y., 2021. Spatio-temporal self-supervised representation learning for 3D point clouds. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 6535\u20136545.","DOI":"10.1109\/ICCV48922.2021.00647"},{"key":"10.1016\/j.engappai.2025.111919_b18","series-title":"Advances in Neural Information Processing Systems","first-page":"8174","article-title":"Rotation-invariant local-to-global representation learning for 3D point cloud","author":"Kim","year":"2020"},{"key":"10.1016\/j.engappai.2025.111919_b19","doi-asserted-by":"crossref","unstructured":"Li, L., Heizmann, M., 2022. A closer look at invariances in self-supervised pre-training for 3D vision. In: European Conference on Computer Vision. pp. 656\u2013673.","DOI":"10.1007\/978-3-031-20056-4_38"},{"key":"10.1016\/j.engappai.2025.111919_b20","doi-asserted-by":"crossref","unstructured":"Li, X., Xu, Q., Zhang, J., Zhang, T., Yu, Q., Sheng, L., Xu, D., 2024. Multi-Modality Affinity Inference for Weakly Supervised 3D Semantic Segmentation. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 3216\u20133224.","DOI":"10.1609\/aaai.v38i4.28106"},{"key":"10.1016\/j.engappai.2025.111919_b21","series-title":"Advances in Neural Information Processing Systems","first-page":"17612","article-title":"Mind the gap: Understanding the modality gap in multi-modal contrastive representation learning","author":"Liang","year":"2022"},{"key":"10.1016\/j.engappai.2025.111919_b22","doi-asserted-by":"crossref","first-page":"3897","DOI":"10.1109\/TMM.2023.3317998","article-title":"Inter-modal masked autoencoder for self-supervised learning on point clouds","volume":"26","author":"Liu","year":"2023","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2025.111919_b23","doi-asserted-by":"crossref","unstructured":"Meng, H.-Y., Gao, L., Lai, Y.-K., Manocha, D., 2019. Vv-net: Voxel vae net with group convolutions for point cloud segmentation. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 8500\u20138508.","DOI":"10.1109\/ICCV.2019.00859"},{"key":"10.1016\/j.engappai.2025.111919_b24","doi-asserted-by":"crossref","unstructured":"Pang, Y., Wang, W., Tay, F.E., Liu, W., Tian, Y., Yuan, L., 2022. Masked autoencoders for point cloud self-supervised learning. In: European Conference on Computer Vision. pp. 604\u2013621.","DOI":"10.1007\/978-3-031-20086-1_35"},{"key":"10.1016\/j.engappai.2025.111919_b25","doi-asserted-by":"crossref","unstructured":"Paul, S., Patterson, Z., Bouguila, N., 2023. CrossMoCo: Multi-Modal momentum contrastive learning for point cloud. In: 2023 Conference on Robots and Vision. pp. 273\u2013280.","DOI":"10.1109\/CRV60082.2023.00042"},{"key":"10.1016\/j.engappai.2025.111919_b26","unstructured":"Qi, Z., Dong, R., Fan, G., Ge, Z., Zhang, X., Ma, K., Yi, L., 2023. Contrast with reconstruct: Contrastive 3d representation learning guided by generative pretraining. In: International Conference on Machine Learning. pp. 28223\u201328243."},{"key":"10.1016\/j.engappai.2025.111919_b27","unstructured":"Qi, C.R., Su, H., Mo, K., Guibas, L.J., 2017a. PointNet: Deep learning on point sets for 3D classification and segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 652\u2013660."},{"key":"10.1016\/j.engappai.2025.111919_b28","series-title":"Advances in Neural Information Processing Systems","first-page":"1","article-title":"Pointnet++: Deep hierarchical feature learning on point sets in a metric space","author":"Qi","year":"2017"},{"key":"10.1016\/j.engappai.2025.111919_b29","series-title":"Advances in Neural Information Processing Systems","first-page":"23192","article-title":"Pointnext: Revisiting pointnet++ with improved training and scaling strategies","author":"Qian","year":"2022"},{"key":"10.1016\/j.engappai.2025.111919_b30","series-title":"Advances in Neural Information Processing Systems","first-page":"1","article-title":"Self-supervised deep learning on point clouds by reconstructing space","author":"Sauder","year":"2019"},{"key":"10.1016\/j.engappai.2025.111919_b31","series-title":"Advances in Neural Information Processing Systems","first-page":"7212","article-title":"Self-supervised few-shot learning on point clouds","author":"Sharma","year":"2020"},{"key":"10.1016\/j.engappai.2025.111919_b32","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2023.121505","article-title":"Slope displacement detection in construction: An automated management algorithm for disaster prevention","volume":"237","author":"Shehadeh","year":"2024","journal-title":"Expert Syst. Appl."},{"key":"10.1016\/j.engappai.2025.111919_b33","doi-asserted-by":"crossref","first-page":"23224","DOI":"10.1109\/JSEN.2024.3405079","article-title":"DCPoint: Global-local dual contrast for self-supervised representation learning of 3D point clouds","volume":"24","author":"Shi","year":"2024","journal-title":"IEEE Sensors J."},{"key":"10.1016\/j.engappai.2025.111919_b34","doi-asserted-by":"crossref","unstructured":"Uy, M.A., Pham, Q.-H., Hua, B.-S., Nguyen, T., Yeung, S.-K., 2019. Revisiting point cloud classification: A new benchmark dataset and classification model on real-world data. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 1588\u20131597.","DOI":"10.1109\/ICCV.2019.00167"},{"key":"10.1016\/j.engappai.2025.111919_b35","doi-asserted-by":"crossref","unstructured":"Wang, C., Jiang, L., Wu, X., Tian, Z., Peng, B., Zhao, H., Jia, J., 2024. GroupContrast: Semantic-aware Self-supervised Representation Learning for 3D Understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4917\u20134928.","DOI":"10.1109\/CVPR52733.2024.00470"},{"key":"10.1016\/j.engappai.2025.111919_b36","doi-asserted-by":"crossref","unstructured":"Wang, H., Liu, Q., Yue, X., Lasenby, J., Kusner, M.J., 2021. Unsupervised Point Cloud Pre-Training via Occlusion Completion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 9782\u20139792.","DOI":"10.1109\/ICCV48922.2021.00964"},{"issue":"5","key":"10.1016\/j.engappai.2025.111919_b37","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3326362","article-title":"Dynamic graph CNN for learning on point clouds","volume":"38","author":"Wang","year":"2019","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.engappai.2025.111919_b38","doi-asserted-by":"crossref","unstructured":"Wang, H., Tang, J., Ji, J., Sun, X., Zhang, R., Ma, Y., Zhao, M., Li, L., Zhao, Z., Lv, T., et al., 2023. Beyond first impressions: Integrating joint multi-modal cues for comprehensive 3D representation. In: Proceedings of the ACM International Conference on Multimedia. pp. 3403\u20133414.","DOI":"10.1145\/3581783.3611767"},{"key":"10.1016\/j.engappai.2025.111919_b39","doi-asserted-by":"crossref","unstructured":"Wei, X., Yu, R., Sun, J., 2020. View-GCN: View-based graph convolutional network for 3D shape analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1850\u20131859.","DOI":"10.1109\/CVPR42600.2020.00192"},{"key":"10.1016\/j.engappai.2025.111919_b40","doi-asserted-by":"crossref","unstructured":"Wu, C., Huang, Q., Jin, K., Pfrommer, J., Beyerer, J., 2024a. A cross branch fusion-based contrastive learning framework for point cloud self-supervised learning. In: 2024 International Conference on 3D Vision. pp. 528\u2013538.","DOI":"10.1109\/3DV62453.2024.00012"},{"key":"10.1016\/j.engappai.2025.111919_b41","doi-asserted-by":"crossref","unstructured":"Wu, X., Jiang, L., Wang, P.-S., Liu, Z., Liu, X., Qiao, Y., Ouyang, W., He, T., Zhao, H., 2024b. Point Transformer V3: Simpler Faster Stronger. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 4840\u20134851.","DOI":"10.1109\/CVPR52733.2024.00463"},{"key":"10.1016\/j.engappai.2025.111919_b42","doi-asserted-by":"crossref","first-page":"1626","DOI":"10.1109\/TMM.2023.3284591","article-title":"Self-supervised intra-modal and cross-modal contrastive learning for point cloud understanding","volume":"26","author":"Wu","year":"2023","journal-title":"IEEE Trans. Multimed."},{"issue":"9","key":"10.1016\/j.engappai.2025.111919_b43","doi-asserted-by":"crossref","first-page":"11321","DOI":"10.1109\/TPAMI.2023.3262786","article-title":"Unsupervised point cloud representation learning with deep neural networks: A survey","volume":"45","author":"Xiao","year":"2023","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.engappai.2025.111919_b44","doi-asserted-by":"crossref","unstructured":"Xie, S., Gu, J., Guo, D., Qi, C.R., Guibas, L., Litany, O., 2020. Pointcontrast: Unsupervised pre-training for 3D point cloud understanding. In: European Conference on Computer Vision. pp. 574\u2013591.","DOI":"10.1007\/978-3-030-58580-8_34"},{"key":"10.1016\/j.engappai.2025.111919_b45","doi-asserted-by":"crossref","unstructured":"Xu, J., Yang, S., Li, X., Tang, Y., Hao, Y., Hu, L., Chen, M., 2024a. A Probability-Driven Framework for Open World 3D Point Cloud Semantic Segmentation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 5977\u20135986.","DOI":"10.1109\/CVPR52733.2024.00571"},{"key":"10.1016\/j.engappai.2025.111919_b46","doi-asserted-by":"crossref","first-page":"8799","DOI":"10.1109\/TMM.2024.3382512","article-title":"CP-Net: Contour-perturbed reconstruction network for self-supervised point cloud learning","volume":"26","author":"Xu","year":"2024","journal-title":"IEEE Trans. Multimed."},{"key":"10.1016\/j.engappai.2025.111919_b47","doi-asserted-by":"crossref","unstructured":"Xue, L., Gao, M., Xing, C., Mart\u00edn-Mart\u00edn, R., Wu, J., Xiong, C., Xu, R., Niebles, J.C., Savarese, S., 2023. Ulip: Learning a unified representation of language, images, and point clouds for 3D understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 1179\u20131189.","DOI":"10.1109\/CVPR52729.2023.00120"},{"key":"10.1016\/j.engappai.2025.111919_b48","doi-asserted-by":"crossref","unstructured":"Xue, L., Yu, N., Zhang, S., Panagopoulou, A., Li, J., Mart\u00edn-Mart\u00edn, R., Wu, J., Xiong, C., Xu, R., Niebles, J.C., et al., 2024. Ulip-2: Towards scalable multimodal pre-training for 3D understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 27091\u201327101.","DOI":"10.1109\/CVPR52733.2024.02558"},{"key":"10.1016\/j.engappai.2025.111919_b49","doi-asserted-by":"crossref","unstructured":"Yu, H.-T., Song, M., 2024. MM-Point: Multi-View information-Enhanced multi-Modal self-Supervised 3D point cloud understanding. In: Proceedings of the AAAI Conference on Artificial Intelligence. pp. 6773\u20136781.","DOI":"10.1609\/aaai.v38i7.28501"},{"key":"10.1016\/j.engappai.2025.111919_b50","doi-asserted-by":"crossref","unstructured":"Yu, X., Tang, L., Rao, Y., Huang, T., Zhou, J., Lu, J., 2022. Point-bert: Pre-training 3D point cloud transformers with masked point modeling. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 19313\u201319322.","DOI":"10.1109\/CVPR52688.2022.01871"},{"key":"10.1016\/j.engappai.2025.111919_b51","doi-asserted-by":"crossref","unstructured":"Zhang, R., Guo, Z., Zhang, W., Li, K., Miao, X., Cui, B., Qiao, Y., Gao, P., Li, H., 2022a. Pointclip: Point cloud understanding by clip. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 8552\u20138562.","DOI":"10.1109\/CVPR52688.2022.00836"},{"key":"10.1016\/j.engappai.2025.111919_b52","series-title":"Advances in Neural Information Processing Systems","first-page":"53076","article-title":"Hednet: A hierarchical encoder-decoder network for 3D object detection in point clouds","author":"Zhang","year":"2024"},{"issue":"12","key":"10.1016\/j.engappai.2025.111919_b53","doi-asserted-by":"crossref","first-page":"11985","DOI":"10.1002\/int.23073","article-title":"PVT: Point-voxel transformer for point cloud learning","volume":"37","author":"Zhang","year":"2022","journal-title":"Int. J. Intell. Syst."},{"key":"10.1016\/j.engappai.2025.111919_b54","unstructured":"Zhang, Q., Wu, H., Zhang, C., Hu, Q., Fu, H., Zhou, J.T., Peng, X., 2023. Provable Dynamic Fusion for Low-Quality Multimodal Data. In: International Conference on Machine Learning. pp. 41753\u201341769."},{"key":"10.1016\/j.engappai.2025.111919_b55","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jiang, L., Jia, J., Torr, P.H., Koltun, V., 2021. Point Transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 16259\u201316268.","DOI":"10.1109\/ICCV48922.2021.01595"},{"key":"10.1016\/j.engappai.2025.111919_b56","doi-asserted-by":"crossref","unstructured":"Zhu, X., Zhang, R., He, B., Guo, Z., Zeng, Z., Qin, Z., Zhang, S., Gao, P., 2023. PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world Learning. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2639\u20132650.","DOI":"10.1109\/ICCV51070.2023.00249"}],"container-title":["Engineering Applications of Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197625019219?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0952197625019219?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T19:18:53Z","timestamp":1772824733000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0952197625019219"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11]]},"references-count":56,"alternative-id":["S0952197625019219"],"URL":"https:\/\/doi.org\/10.1016\/j.engappai.2025.111919","relation":{},"ISSN":["0952-1976"],"issn-type":[{"value":"0952-1976","type":"print"}],"subject":[],"published":{"date-parts":[[2025,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Multi-modal self-supervised contrastive representation learning for three-dimensional point cloud understanding","name":"articletitle","label":"Article Title"},{"value":"Engineering Applications of Artificial Intelligence","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.engappai.2025.111919","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2025 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"111919"}}