{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T03:41:10Z","timestamp":1742960470578,"version":"3.40.3"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783319464657"},{"type":"electronic","value":"9783319464664"}],"license":[{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2016,1,1]],"date-time":"2016-01-01T00:00:00Z","timestamp":1451606400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2016]]},"DOI":"10.1007\/978-3-319-46466-4_17","type":"book-chapter","created":{"date-parts":[[2016,9,16]],"date-time":"2016-09-16T09:31:58Z","timestamp":1474018318000},"page":"275-293","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Weakly Supervised Learning of Heterogeneous Concepts in Videos"],"prefix":"10.1007","author":[{"given":"Sohil","family":"Shah","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kuldeep","family":"Kulkarni","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arijit","family":"Biswas","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ankit","family":"Gandhi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Om","family":"Deshmukh","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Larry S.","family":"Davis","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2016,9,17]]},"reference":[{"key":"17_CR1","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"472","DOI":"10.1007\/978-3-319-10605-2_31","volume-title":"Computer Vision \u2013 ECCV 2014","author":"Z Shi","year":"2014","unstructured":"Shi, Z., Yang, Y., Hospedales, T.M., Xiang, T.: Weakly supervised learning of objects, attributes and their associations. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8690, pp. 472\u2013487. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10605-2_31"},{"key":"17_CR2","doi-asserted-by":"crossref","unstructured":"Leung, T., Song, Y., Zhang, J.: Handling label noise in video classification via multiple instance learning. In: 2011 IEEE International Conference on Computer Vision (ICCV), pp. 2056\u20132063. IEEE (2011)","DOI":"10.1109\/ICCV.2011.6126479"},{"key":"17_CR3","doi-asserted-by":"crossref","unstructured":"Oquab, M., Bottou, L., Laptev, I., Sivic, J.: Is object localization for free? weakly-supervised learning with convolutional neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2015)","DOI":"10.1109\/CVPR.2015.7298668"},{"key":"17_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"628","DOI":"10.1007\/978-3-319-10602-1_41","volume-title":"Computer Vision \u2013 ECCV 2014","author":"P Bojanowski","year":"2014","unstructured":"Bojanowski, P., Lajugie, R., Bach, F., Laptev, I., Ponce, J., Schmid, C., Sivic, J.: Weakly supervised action labeling in videos under ordering constraints. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 628\u2013643. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10602-1_41"},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Bojanowski, P., Bach, F., Laptev, I., Ponce, J., Schmid, C., Sivic, J.: Finding actors and actions in movies. In: 2013 IEEE International Conference on Computer Vision (ICCV), pp. 2280\u20132287. IEEE (2013)","DOI":"10.1109\/ICCV.2013.283"},{"key":"17_CR6","first-page":"475","volume":"18","author":"Z Ghahramani","year":"2005","unstructured":"Ghahramani, Z., Griffiths, T.L.: Infinite latent feature models and the indian buffet process. Adv. Neural Inf. Proces. Syst. 18, 475\u2013482 (2005)","journal-title":"Adv. Neural Inf. Proces. Syst."},{"key":"17_CR7","first-page":"1185","volume":"12","author":"TL Griffiths","year":"2011","unstructured":"Griffiths, T.L., Ghahramani, Z.: The indian buffet process: an introduction and review. J. Mach. Learn. Res. 12, 1185\u20131224 (2011)","journal-title":"J. Mach. Learn. Res."},{"key":"17_CR8","unstructured":"Ozdemir, B., Davis, L.S.: A probabilistic framework for multimodal retrieval using integrative indian buffet process. In: Advances in Neural Information Processing Systems, pp. 2384\u20132392 (2014)"},{"issue":"2","key":"17_CR9","doi-asserted-by":"publisher","first-page":"305","DOI":"10.1111\/j.1551-6709.2011.01216.x","volume":"36","author":"I Yildirim","year":"2012","unstructured":"Yildirim, I., Jacobs, R.A.: A rational analysis of the acquisition of multisensory representations. Cogn. Sci. 36(2), 305\u2013332 (2012)","journal-title":"Cogn. Sci."},{"issue":"1","key":"17_CR10","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.jmp.2011.08.004","volume":"56","author":"SJ Gershman","year":"2012","unstructured":"Gershman, S.J., Blei, D.M.: A tutorial on bayesian nonparametric models. J. Math. Psychol. 56(1), 1\u201312 (2012)","journal-title":"J. Math. Psychol."},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Xu, C., Hsieh, S.H., Xiong, C., Corso, J.J.: Can humans fly? action understanding with multiple classes of actors. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2264\u20132273 (2015)","DOI":"10.1109\/CVPR.2015.7298839"},{"key":"17_CR12","first-page":"1417","volume":"7","author":"C Zhang","year":"2005","unstructured":"Zhang, C., Platt, J.C., Viola, P.A.: Multiple instance boosting for object detection. Adv. Neural Inf. Process. Syst. 7, 1417\u20131424 (2005)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"17_CR13","unstructured":"Andrews, S., Tsochantaridis, I., Hofmann, T.: Support vector machines for multiple-instance learning. In: Proceedings of Advances in Neural Information Processing Systems, pp. 561\u2013568 (2002)"},{"issue":"2","key":"17_CR14","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1109\/TPAMI.2008.284","volume":"32","author":"S Ali","year":"2010","unstructured":"Ali, S., Shah, M.: Human action recognition in videos using kinematic features and multiple instance learning. Pattern Anal. Mach. Intell. IEEE Trans. 32(2), 288\u2013303 (2010)","journal-title":"Pattern Anal. Mach. Intell. IEEE Trans."},{"key":"17_CR15","doi-asserted-by":"crossref","unstructured":"Babenko, B., Yang, M.H., Belongie, S.: Visual tracking with online multiple instance learning. In: IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2009, pp. 983\u2013990. IEEE (2009)","DOI":"10.1109\/CVPR.2009.5206737"},{"key":"17_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"158","DOI":"10.1007\/978-3-540-88693-8_12","volume-title":"Computer Vision \u2013 ECCV 2008","author":"T Cour","year":"2008","unstructured":"Cour, T., Jordan, C., Miltsakaki, E., Taskar, B.: Movie\/Script: alignment and parsing of video and text transcription. In: Forsyth, D., Torr, P., Zisserman, A. (eds.) ECCV 2008. LNCS, vol. 5305, pp. 158\u2013171. Springer, Heidelberg (2008). doi: 10.1007\/978-3-540-88693-8_12"},{"key":"17_CR17","doi-asserted-by":"crossref","unstructured":"Bojanowski, P., Lagugie, R., Grave, E., Bach, F., Laptev, I., Ponce, J., Schmid, C.: Weakly-supervised alignment of video with text. In: ICCV, IEEE (2015)","DOI":"10.1109\/ICCV.2015.507"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"Prest, A., Leistner, C., Civera, J., Schmid, C., Ferrari, V.: Learning object class detectors from weakly annotated video. In: 2012 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 3282\u20133289. IEEE (2012)","DOI":"10.1109\/CVPR.2012.6248065"},{"issue":"3","key":"17_CR19","doi-asserted-by":"publisher","first-page":"237","DOI":"10.1007\/s11263-013-0646-8","volume":"106","author":"H Bilen","year":"2014","unstructured":"Bilen, H., Namboodiri, V.P., Van Gool, L.J.: Object and action classification with latent window parameters. Int. J. Comput. Vis. 106(3), 237\u2013251 (2014)","journal-title":"Int. J. Comput. Vis."},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Tapaswi, M., Bauml, M., Stiefelhagen, R.: Book2movie: Aligning video scenes with book chapters. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1827\u20131835 (2015)","DOI":"10.1109\/CVPR.2015.7298792"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Kiros, R., Zemel, R., Salakhutdinov, R., Urtasun, R., Torralba, A., Fidler, S.: Aligning books and movies: towards story-like visual explanations by watching movies and reading books. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 19\u201327 (2015)","DOI":"10.1109\/ICCV.2015.11"},{"key":"17_CR22","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"95","DOI":"10.1007\/978-3-319-10590-1_7","volume-title":"Computer Vision \u2013 ECCV 2014","author":"V Ramanathan","year":"2014","unstructured":"Ramanathan, V., Joulin, A., Liang, P., Fei-Fei, L.: Linking people in videos with \u201ctheir\u201d names using coreference resolution. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8689, pp. 95\u2013110. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10590-1_7"},{"key":"17_CR23","doi-asserted-by":"crossref","unstructured":"Karpathy, A., Fei-Fei, L.: Deep visual-semantic alignments for generating image descriptions. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 3128\u20133137 (2015)","DOI":"10.1109\/CVPR.2015.7298932"},{"key":"17_CR24","unstructured":"Xu, K., Ba, J., Kiros, R., Cho, K., Courville, A., Salakhudinov, R., Zemel, R., Bengio, Y.: Show, attend and tell: Neural image caption generation with visual attention. In: Proceedings of The 32nd International Conference on Machine Learning, pp. 2048\u20132057 (2015)"},{"key":"17_CR25","doi-asserted-by":"crossref","unstructured":"Fang, H., Gupta, S., Iandola, F., Srivastava, R.K., Deng, L., Doll\u00e1r, P., Gao, J., He, X., Mitchell, M., Platt, J.C., et al.: From captions to visual concepts and back. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1473\u20131482 (2015)","DOI":"10.1109\/CVPR.2015.7298754"},{"key":"17_CR26","doi-asserted-by":"crossref","unstructured":"Sun, C., Gan, C., Nevatia, R.: Automatic concept discovery from parallel text and visual corpora. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2596\u20132604 (2015)","DOI":"10.1109\/ICCV.2015.298"},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Venugopalan, S., Rohrbach, M., Donahue, J., Mooney, R., Darrell, T., Saenko, K.: Sequence to sequence-video to text. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 4534\u20134542 (2015)","DOI":"10.1109\/ICCV.2015.515"},{"key":"17_CR28","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"209","DOI":"10.1007\/978-3-319-24947-6_17","volume-title":"Pattern Recognition","author":"A Rohrbach","year":"2015","unstructured":"Rohrbach, A., Rohrbach, M., Schiele, B.: The long-short story of movie description. In: Gall, J., Gehler, P., Leibe, B. (eds.) GCPR 2015. LNCS, vol. 9358, pp. 209\u2013221. Springer, Heidelberg (2015). doi: 10.1007\/978-3-319-24947-6_17"},{"issue":"11","key":"17_CR29","doi-asserted-by":"publisher","first-page":"1875","DOI":"10.1109\/TMM.2015.2477044","volume":"17","author":"K Cho","year":"2015","unstructured":"Cho, K., Courville, A., Bengio, Y.: Describing multimedia content using attention-based encoder-decoder networks. Multimedia IEEE Trans. 17(11), 1875\u20131886 (2015)","journal-title":"Multimedia IEEE Trans."},{"key":"17_CR30","unstructured":"Doshi, F., Miller, K., Gael, J.V., Teh, Y.W.: Variational inference for the indian buffet process. In: International Conference on Artificial Intelligence and Statistics, pp. 137\u2013144 (2009)"},{"issue":"1\u20132","key":"17_CR31","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1561\/2200000001","volume":"1","author":"MJ Wainwright","year":"2008","unstructured":"Wainwright, M.J., Jordan, M.I.: Graphical models, exponential families, and variational inference. Found. Trends Mach. Learn. 1(1\u20132), 1\u2013305 (2008)","journal-title":"Found. Trends Mach. Learn."},{"issue":"4","key":"17_CR32","doi-asserted-by":"crossref","first-page":"278","DOI":"10.1080\/00031305.1988.10475585","volume":"42","author":"A Zellner","year":"1988","unstructured":"Zellner, A.: Optimal information processing and bayes\u2019s theorem. Am. Stat. 42(4), 278\u2013280 (1988)","journal-title":"Am. Stat."},{"issue":"1","key":"17_CR33","first-page":"1799","volume":"15","author":"J Zhu","year":"2014","unstructured":"Zhu, J., Chen, N., Xing, E.P.: Bayesian inference with posterior regularization and applications to infinite latent svms. J. Mach. Learn. Res. 15(1), 1799\u20131847 (2014)","journal-title":"J. Mach. Learn. Res."},{"key":"17_CR34","first-page":"2001","volume":"11","author":"K Ganchev","year":"2010","unstructured":"Ganchev, K., Gra\u00e7a, J., Gillenwater, J., Taskar, B.: Posterior regularization for structured latent variable models. J. Mach. Learn. Res. 11, 2001\u20132049 (2010)","journal-title":"J. Mach. Learn. Res."},{"key":"17_CR35","unstructured":"Zhu, X., Ramanan, D.: Face detection, pose estimation and landmark estimation in the wild. In: IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2012)"},{"key":"17_CR36","doi-asserted-by":"crossref","unstructured":"Everingham, M., Sivic, J., Zisserman, A.: Hello! my name is buffy-automatic naming of characters in tv video. In: BMVC. vol. 2. 6 (2006)","DOI":"10.5244\/C.20.92"},{"key":"17_CR37","doi-asserted-by":"crossref","unstructured":"Parkhi, O.M., Simonyan, K., Vedaldi, A., Zisserman, A.: A compact and discriminative face track descriptor. In: 2014 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 1693\u20131700. IEEE (2014)","DOI":"10.1109\/CVPR.2014.219"},{"key":"17_CR38","doi-asserted-by":"crossref","unstructured":"Wang, H., Schmid, C.: Action recognition with improved trajectories. In: 2013 IEEE International Conference on Computer Vision (ICCV), pp. 3551\u20133558. IEEE (2013)","DOI":"10.1109\/ICCV.2013.441"},{"key":"17_CR39","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"737","DOI":"10.1007\/978-3-319-10578-9_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"D Oneata","year":"2014","unstructured":"Oneata, D., Revaud, J., Verbeek, J., Schmid, C.: Spatio-temporal object detection proposals. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8691, pp. 737\u2013752. Springer, Heidelberg (2014). doi: 10.1007\/978-3-319-10578-9_48"},{"key":"17_CR40","doi-asserted-by":"crossref","unstructured":"Chatfield, K., Simonyan, K., Vedaldi, A., Zisserman, A.: Return of the devil in the details: delving deep into convolutional nets. In: British Machine Vision Conference (2014)","DOI":"10.5244\/C.28.6"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2016"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-319-46466-4_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,6,19]],"date-time":"2024-06-19T11:37:29Z","timestamp":1718797049000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/978-3-319-46466-4_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2016]]},"ISBN":["9783319464657","9783319464664"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-319-46466-4_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2016]]},"assertion":[{"value":"17 September 2016","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Amsterdam","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2016","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"8 October 2016","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"16 October 2016","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2016","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.eccv2016.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"This content has been made available to all.","name":"free","label":"Free to read"}]}}