{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T17:48:20Z","timestamp":1777657700632,"version":"3.51.4"},"reference-count":112,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2021,2,22]],"date-time":"2021-02-22T00:00:00Z","timestamp":1613952000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,2,22]],"date-time":"2021-02-22T00:00:00Z","timestamp":1613952000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/501100006595","name":"UEFISCDI","doi-asserted-by":"crossref","award":["PN-III-P2-2.1-SOL-2016-02-0002"],"award-info":[{"award-number":["PN-III-P2-2.1-SOL-2016-02-0002"]}],"id":[{"id":"10.13039\/501100006595","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/100010661","name":"Horizon 2020 Framework Programme","doi-asserted-by":"publisher","award":["951911"],"award-info":[{"award-number":["951911"]}],"id":[{"id":"10.13039\/100010661","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Ministerul Fondurilor Europene","award":["51675\/09.07.2019"],"award-info":[{"award-number":["51675\/09.07.2019"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2021,5]]},"DOI":"10.1007\/s11263-021-01443-1","type":"journal-article","created":{"date-parts":[[2021,2,22]],"date-time":"2021-02-22T05:03:11Z","timestamp":1613970191000},"page":"1526-1550","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Visual Interestingness Prediction: A Benchmark Framework and Literature Review"],"prefix":"10.1007","volume":"129","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2312-6672","authenticated-orcid":false,"given":"Mihai Gabriel","family":"Constantin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liviu-Daniel","family":"\u015etefan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bogdan","family":"Ionescu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ngoc Q. K.","family":"Duong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Claire-H\u00e9l\u00e9ne","family":"Demarty","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mats","family":"Sj\u00f6berg","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,2,22]]},"reference":[{"key":"1443_CR1","unstructured":"Abdi H.(2007). \u201cThe kendall rank correlation coefficient,\u201d Encyclopedia of measurement and statistics. Sage, pp.\u00a0508\u2013510."},{"key":"1443_CR2","unstructured":"Ahmed, O.\u00a0B., Wacker, J., Gaballo, A., & Huet, B. (2017). Eurecom@mediaeval 2017: Media genre inference for predicting media interestingness. In MediaEval workshop, Dublin, Ireland, September 13-15., (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR3","unstructured":"Almeida, J. (2016) UNIFESP at mediaeval 2016: Predicting media interestingness task. In MediaEval workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org."},{"key":"1443_CR4","unstructured":"Almeida, J., & Savii, R.\u00a0M. (2017). GIBIS at mediaeval 2017: Predicting media interestingness task. In: MediaEval workshop, Dublin, Ireland, September 13-15., (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR5","doi-asserted-by":"crossref","unstructured":"Almeida, J., Leite, N.\u00a0J., & Torres, R.\u00a0d.\u00a0S. (2011). Comparison of video sequences with histograms of motion patterns. In 18th IEEE international conference on image processing, pp.\u00a03673\u20133676, IEEE.","DOI":"10.1109\/ICIP.2011.6116516"},{"key":"1443_CR6","doi-asserted-by":"crossref","unstructured":"Almeida, J., Valem, L.\u00a0P., & Pedronette, D.\u00a0C. (2017) A rank aggregation framework for video interestingness prediction. In International conference on image analysis and processing, pp.\u00a03\u201314, Springer.","DOI":"10.1007\/978-3-319-68560-1_1"},{"issue":"3","key":"1443_CR7","doi-asserted-by":"publisher","first-page":"14","DOI":"10.1145\/2629531","volume":"32","author":"G Awad","year":"2014","unstructured":"Awad, G., Over, P., & Kraaij, W. (2014). Content-based video copy detection benchmarking at trecvid. ACM Transactions on Information Systems (TOIS), 32(3), 14.","journal-title":"ACM Transactions on Information Systems (TOIS)"},{"key":"1443_CR8","unstructured":"Aytar, Y., Vondrick, C., & Torralba, A. (2016). Soundnet: Learning sound representations from unlabeled video, In Advances in neural information processing systems 29: annual conference on neural information processing systems, December 5\u201310 (pp. 892\u2013900). Spain: Barcelona."},{"key":"1443_CR9","doi-asserted-by":"crossref","unstructured":"Bakhshi, S., Shamma, D.\u00a0A., Kennedy, L., Song, Y., De\u00a0Juan, P., & Kaye, J. (2016) Fast, cheap, and good: Why animated gifs engage us. In Proceedings of the chi conference on human factors in computing systems, pp.\u00a0575\u2013586, ACM","DOI":"10.1145\/2858036.2858532"},{"issue":"4","key":"1443_CR10","doi-asserted-by":"publisher","first-page":"184","DOI":"10.1111\/j.2044-8295.1949.tb00219.x","volume":"39","author":"DE Berlyne","year":"1949","unstructured":"Berlyne, D. E. (1949). Interest as a psychological concept. British Journal of Psychology. General Section, 39(4), 184\u2013195.","journal-title":"British Journal of Psychology. General Section"},{"key":"1443_CR11","doi-asserted-by":"publisher","DOI":"10.1037\/11164-000","volume-title":"Conflict, arousal, and curiosity","author":"DE Berlyne","year":"1960","unstructured":"Berlyne, D. E. (1960). Conflict, arousal, and curiosity. New York: McGraw-Hill Book Company."},{"issue":"5","key":"1443_CR12","doi-asserted-by":"publisher","first-page":"279","DOI":"10.3758\/BF03212593","volume":"8","author":"DE Berlyne","year":"1970","unstructured":"Berlyne, D. E. (1970). Novelty, complexity, and hedonic value. Perception & Psychophysics, 8(5), 279\u2013286.","journal-title":"Perception & Psychophysics"},{"key":"1443_CR13","unstructured":"Berson, E., Demarty, C., & Duong, N.\u00a0Q.\u00a0K. (2017). Multimodality and deep learning when predicting media interestingness. In MediaEval workshop, Dublin, Ireland, September 13-15. (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR14","doi-asserted-by":"crossref","unstructured":"Borth, D., Chen, T., Ji, R., & Chang S.-F. (2013). Sentibank: large-scale ontology and classifiers for detecting sentiment and emotions in visual content. In Proceedings of the 21st ACM international conference on Multimedia, pp.\u00a0459\u2013460, ACM.","DOI":"10.1145\/2502081.2502268"},{"issue":"3\u20134","key":"1443_CR15","first-page":"324","volume":"39","author":"RA Bradley","year":"1952","unstructured":"Bradley, R. A., & Terry, M. E. (1952). Rank analysis of incomplete block designs: the method of paired comparisons. Biometrika, 39(3\u20134), 324\u2013345.","journal-title":"Biometrika"},{"issue":"2","key":"1443_CR16","doi-asserted-by":"publisher","first-page":"235","DOI":"10.1145\/3130348.3130373","volume":"51","author":"C Buckley","year":"2017","unstructured":"Buckley, C., & Voorhees, E. M. (2017). Evaluating evaluation measure stability. SIGIR Forum, 51(2), 235\u2013242.","journal-title":"SIGIR Forum"},{"key":"1443_CR17","doi-asserted-by":"crossref","unstructured":"Carballal, A., Fernandez-Lozano, C., Heras, J., & Romero, J. (2019). Transfer learning features for predicting aesthetics through a novel hybrid machine learning method. Neural Computing and Applications, 1\u201312.","DOI":"10.1007\/s00521-019-04065-4"},{"issue":"16","key":"1443_CR18","doi-asserted-by":"publisher","first-page":"1","DOI":"10.2352\/ISSN.2470-1173.2016.16.HVEI-139","volume":"2016","author":"C Chamaret","year":"2016","unstructured":"Chamaret, C., Demarty, C.-H., Demoulin, V., & Marquant, G. (2016). Experiencing the interestingness concept within and between pictures. Electronic Imaging, 2016(16), 1\u201312.","journal-title":"Electronic Imaging"},{"key":"1443_CR19","unstructured":"Constantin, M.\u00a0G., Boteanu, B.\u00a0A., & Ionescu, B. (2017). Lapi at mediaeval 2017-predicting media interestingness. In MediaEval workshop, Dublin, Ireland, September 13-15. (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR20","doi-asserted-by":"crossref","unstructured":"Constantin, M. G., Redi, M., Zen, G., & Ionescu, B. (2019). Computational understanding of visual interestingness beyond semantics: Literature survey and analysis of covariates. ACM Computing Surveys.","DOI":"10.1145\/3301299"},{"key":"1443_CR21","doi-asserted-by":"crossref","unstructured":"Constantin, M.\u00a0G., & Ionescu, B. (2017). Content description for predicting image interestingness. In 2017 international symposium on signals, circuits and systems (ISSCS), pp.\u00a01\u20134, IEEE, 13\u201314 July.","DOI":"10.1109\/ISSCS.2017.8034914"},{"key":"1443_CR22","doi-asserted-by":"crossref","unstructured":"Dalal, N., & Triggs, B. (2005). Histograms of oriented gradients for human detection. In International conference on computer vision & pattern recognition, (Vol.\u00a01), pp.\u00a0886\u2013893, IEEE Computer Society.","DOI":"10.1109\/CVPR.2005.177"},{"key":"1443_CR23","doi-asserted-by":"crossref","unstructured":"Danelljan, M., H\u00e4ger, G., Khan, F., Felsberg, M. (2014). Accurate scale estimation for robust visual tracking. In British machine vision conference, nottingham, September 1-5, BMVA Press.","DOI":"10.5244\/C.28.65"},{"key":"1443_CR24","doi-asserted-by":"crossref","unstructured":"Datta, R., Joshi, D., Li, J., & Wang, J.Z. (2006). Studying aesthetics in photographic images using a computational approach. In European conference on computer vision, pp.\u00a0288\u2013301, Springer.","DOI":"10.1007\/11744078_23"},{"key":"1443_CR25","doi-asserted-by":"crossref","unstructured":"Demarty, C.-H., Sj\u00f6berg, M., Constantin, M.\u00a0G., Duong, N.\u00a0Q., Ionescu, B., Do, T.-T., & Wang, H. (2017). Predicting interestingness of visual content. In Visual content indexing and retrieval with psycho-visual models, pp.\u00a0233\u2013265, Cham: Springer.","DOI":"10.1007\/978-3-319-57687-9_10"},{"key":"1443_CR26","unstructured":"Demarty, C.-H., Sj\u00f6berg, M., Ionescu, B., Do, T.-T., Gygli, M., & Duong, N.\u00a0Q.\u00a0K. (2017). Mediaeval 2017 predicting media interestingness task. In MediaEval Workshop, Dublin, Ireland, September 13-15. (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR27","unstructured":"Demarty, C.-H., Sj\u00f6berg, M., Ionescu, B., Do, T.-T., Wang, H., Duong, N.\u00a0Q.\u00a0K., & Lefebvre, F. (2016). Mediaeval 2016 predicting media interestingness task. In MediaEval workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org"},{"issue":"15","key":"1443_CR28","doi-asserted-by":"publisher","first-page":"1988","DOI":"10.1016\/j.patrec.2008.03.001","volume":"29","author":"T Deselaers","year":"2008","unstructured":"Deselaers, T., Deserno, T. M., & M\u00fcller, H. (2008). Automatic medical image annotation in imageclef 2007: Overview, results, and discussion. Pattern Recognition Letters, 29(15), 1988\u20131995.","journal-title":"Pattern Recognition Letters"},{"key":"1443_CR29","unstructured":"Erdogan, G., Erdem, A., & Erdem, E. (2016). HUCVL at mediaeval 2016: Predicting interesting key frames with deep models. In MediaEval workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org."},{"issue":"1","key":"1443_CR30","doi-asserted-by":"publisher","first-page":"98","DOI":"10.1007\/s11263-014-0733-5","volume":"111","author":"M Everingham","year":"2015","unstructured":"Everingham, M., Eslami, S. A., Van Gool, L., Williams, C. K., Winn, J., & Zisserman, A. (2015). The pascal visual object classes challenge: A retrospective. International Journal of Computer Vision, 111(1), 98\u2013136.","journal-title":"International Journal of Computer Vision"},{"key":"1443_CR31","doi-asserted-by":"crossref","unstructured":"Eyben, F., W\u00f6llmer, M., & Schuller, B. (2010) Opensmile: the munich versatile and fast open-source audio feature extractor. In Proceedings of the 18th ACM international conference on multimedia, pp.\u00a01459\u20131462, ACM.","DOI":"10.1145\/1873951.1874246"},{"issue":"1","key":"1443_CR32","doi-asserted-by":"publisher","first-page":"119","DOI":"10.1006\/jcss.1997.1504","volume":"55","author":"Y Freund","year":"1997","unstructured":"Freund, Y., & Schapire, R. E. (1997). A decision-theoretic generalization of on-line learning and an application to boosting. Journal of Computer and System Sciences, 55(1), 119\u2013139.","journal-title":"Journal of Computer and System Sciences"},{"key":"1443_CR33","doi-asserted-by":"crossref","unstructured":"Friedman, J. H. (2001). Greedy function approximation: a gradient boosting machine. Annals of Statistics, 1189\u20131232.","DOI":"10.1214\/aos\/1013203451"},{"key":"1443_CR34","doi-asserted-by":"crossref","unstructured":"Ghadiyaram, D., Tran, D., & Mahajan, D. (2019). Large-scale weakly-supervised pre-training for video action recognition. In: Proceedings of the IEEE conference on computer vision and pattern recognition, 12046\u201312055.","DOI":"10.1109\/CVPR.2019.01232"},{"key":"1443_CR35","doi-asserted-by":"crossref","unstructured":"Goyal, R., Kahou, S.\u00a0E., Michalski, V., Materzynska, J., Westphal, S., Kim, H., Haenel, V., Fruend, I., Yianilos, P., & Mueller-Freitag, M. et\u00a0al., (2017). The something something video database for learning and evaluating visual common sense. In ICCV, (Vol.\u00a01), p.\u00a05","DOI":"10.1109\/ICCV.2017.622"},{"key":"1443_CR36","doi-asserted-by":"crossref","unstructured":"Grabner, H., Nater, F., Druey, M., & Van\u00a0Gool, L. (2013). Visual interestingness in image sequences. In Proceedings of the 21st ACM international conference on Multimedia, pp.\u00a01017\u20131026, ACM.","DOI":"10.1145\/2502081.2502109"},{"key":"1443_CR37","doi-asserted-by":"crossref","unstructured":"Gygli, M., & Soleymani, M. (2016). Analyzing and predicting gif interestingness. In Proceedings of the 24th ACM international conference on Multimedia, pp.\u00a0122\u2013126, ACM.","DOI":"10.1145\/2964284.2967195"},{"key":"1443_CR38","doi-asserted-by":"crossref","unstructured":"Gygli, M., Grabner, H., Riemenschneider, H., Nater, F., & Van\u00a0Gool, L. (2013) The interestingness of images. In Proceedings of the IEEE international conference on computer vision, pp.\u00a01633\u20131640, IEEE.","DOI":"10.1109\/ICCV.2013.205"},{"key":"1443_CR39","doi-asserted-by":"crossref","unstructured":"Gygli, M., Song, Y., & Cao, L. (2016). Video2gif: Automatic generation of animated gifs from video,\u201d In Proceedings of the IEEE conference on computer vision and pattern recognition, pp.\u00a01001\u20131009, IEEE.","DOI":"10.1109\/CVPR.2016.114"},{"key":"1443_CR40","unstructured":"Han, S., Meng, Z., Khan, A.-S., & Tong, Y. (2016). Incremental boosting convolutional neural network for facial action unit recognition. In Advances in neural information processing systems, 109\u2013117."},{"issue":"1","key":"1443_CR41","doi-asserted-by":"publisher","first-page":"77","DOI":"10.1080\/19312450709336664","volume":"1","author":"AF Hayes","year":"2007","unstructured":"Hayes, A. F., & Krippendorff, K. (2007). Answering the call for a standard reliability measure for coding data. Communication Methods and Measures, 1(1), 77\u201389.","journal-title":"Communication Methods and Measures"},{"key":"1443_CR42","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition, 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"1443_CR43","first-page":"213","volume":"11","author":"S Hidi","year":"1992","unstructured":"Hidi, S., & Anderson, V. (1992). Situational interest and its impact on reading and expository writing. The Role of Interest in Learning and Development, 11, 213\u2013214.","journal-title":"The Role of Interest in Learning and Development"},{"issue":"8","key":"1443_CR44","doi-asserted-by":"publisher","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","volume":"9","author":"S Hochreiter","year":"1997","unstructured":"Hochreiter, S., & Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735\u20131780.","journal-title":"Neural Computation"},{"key":"1443_CR45","doi-asserted-by":"crossref","unstructured":"Hsieh, L.-C., Hsu, W.\u00a0H., & Wang, H.-C. (2014). Investigating and predicting social and visual image interestingness on social media by crowdsourcing. In IEEE international conference on acoustics, speech and signal processing (ICASSP), pp.\u00a04309\u20134313, IEEE.","DOI":"10.1109\/ICASSP.2014.6854415"},{"key":"1443_CR46","doi-asserted-by":"crossref","unstructured":"Hua, X.-S., Yang, L., Wang, J., Wang, J., Ye, M., Wang, K., Rui, Y., & Li, J. (2013). Clickage: Towards bridging semantic and intent gaps via mining click logs of search engines. In Proceedings of the 21st ACM international conference on Multimedia, pp.\u00a0243\u2013252.","DOI":"10.1145\/2502081.2502283"},{"key":"1443_CR47","doi-asserted-by":"crossref","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., & Darrell, T. (2014). Caffe: Convolutional architecture for fast feature embedding. In Proceedings of the 22nd ACM international conference on Multimedia, pp.\u00a0675\u2013678, ACM.","DOI":"10.1145\/2647868.2654889"},{"key":"1443_CR48","doi-asserted-by":"crossref","unstructured":"Jiang, Y.-G., Wang, Y., Feng, R., Xue, X., Zheng, Y., & Yang, H. (2013). Understanding and predicting interestingness of videos. In Twenty-Seventh AAAI conference on artificial intelligence, pp.\u00a01\u20137.","DOI":"10.1609\/aaai.v27i1.8457"},{"issue":"8","key":"1443_CR49","doi-asserted-by":"publisher","first-page":"1174","DOI":"10.1109\/TMM.2015.2436813","volume":"17","author":"Y-G Jiang","year":"2015","unstructured":"Jiang, Y.-G., Dai, Q., Mei, T., Rui, Y., & Chang, S.-F. (2015). Super fast event recognition in internet videos. IEEE Transactions on Multimedia, 17(8), 1174\u20131186.","journal-title":"IEEE Transactions on Multimedia"},{"key":"1443_CR50","doi-asserted-by":"publisher","first-page":"55","DOI":"10.1016\/j.compmedimag.2014.03.004","volume":"39","author":"J Kalpathy-Cramer","year":"2015","unstructured":"Kalpathy-Cramer, J., de Herrera, A. G. S., Demner-Fushman, D., Antani, S., Bedrick, S., & M\u00fcller, H. (2015). Evaluating performance of biomedical image retrieval systems-an overview of the medical image retrieval task at imageclef 2004\u20132013. Computerized Medical Imaging and Graphics, 39, 55\u201361.","journal-title":"Computerized Medical Imaging and Graphics"},{"key":"1443_CR51","unstructured":"Kay, W., Carreira, J., Simonyan, K., Zhang, B., Hillier, C., Vijayanarasimhan, S., Viola, F., Green, T., Back, T., Natsev, P. et\u00a0al., (2017). The kinetics human action video dataset. arXiv preprint arXiv:1705.06950."},{"key":"1443_CR52","unstructured":"Ke, Y., Tang, X., & Jing, F. (2006). The design of high-level features for photo quality assessment. In IEEE computer society conference on computer vision and pattern recognition (Vol.\u00a01), pp.\u00a0419\u2013426, IEEE."},{"key":"1443_CR53","doi-asserted-by":"crossref","unstructured":"Khosla, A., Raju, A. S., Torralba, A., & Oliva, A. (2015). Understanding and predicting image memorability at a large scale. Proceedings of the IEEE international conference on computer vision, 2390\u20132398.","DOI":"10.1109\/ICCV.2015.275"},{"key":"1443_CR54","unstructured":"Kingma, D.\u00a0P., & Ba, J. (2015). Adam: A method for stochastic optimization. In 3rd International conference on learning representations, San Diego, CA, USA, May 7-9, conference track proceedings."},{"key":"1443_CR55","unstructured":"Kiros, R., Salakhutdinov, R., & Zemel, R.\u00a0S. (2014). \u201cUnifying visual-semantic embeddings with multimodal neural language models,\u201d arXiv preprint arXiv:1411.2539."},{"key":"1443_CR56","doi-asserted-by":"crossref","unstructured":"Kittler, J., Hater, M., Duin, R.\u00a0P. (1996). Combining classifiers. In Proceedings of 13th international conference on pattern recognition, (Vol.\u00a02), pp.\u00a0897\u2013901, IEEE.","DOI":"10.1109\/ICPR.1996.547205"},{"key":"1443_CR57","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. Advances in Neural Information Processing Systems, 1097\u20131105."},{"key":"1443_CR58","unstructured":"Lam, V., Do, T., Phan, S., Le, D.-D., Satoh, S., & Duong, D.\u00a0A. (2016). Nii-uit at mediaeval 2016 predicting media interestingness task. In MediaEval Workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org."},{"key":"1443_CR59","doi-asserted-by":"crossref","unstructured":"Lazebnik, S., Schmid, C., & Ponce, J. (2006). Beyond bags of features: Spatial pyramid matching for recognizing natural scene categories. In IEEE computer society conference on computer vision and pattern recognition, (Vol.\u00a02), pp.\u00a02169\u20132178, IEEE.","DOI":"10.1109\/CVPR.2006.68"},{"key":"1443_CR60","doi-asserted-by":"crossref","unstructured":"Li, J., Barkowsky, M., & Callet, P.\u00a0L. (2013). Boosting paired comparison methodology in measuring visual discomfort of 3dtv: performances of three different designs. In Proceedings of SPIE electronic imaging, stereoscopic displays and applications (Vol.\u00a08648).","DOI":"10.1117\/12.2002075"},{"key":"1443_CR61","doi-asserted-by":"crossref","unstructured":"Li, X., Huo, Y., Jin, Q., & Xu, J. (2016). Detecting violence in video using subclasses. In Proceedings of the 2016 ACM conference on multimedia conference, MM 2016, pp.\u00a0586\u2013590, ACM, October 15-19.","DOI":"10.1145\/2964284.2967289"},{"issue":"2","key":"1443_CR62","doi-asserted-by":"publisher","first-page":"236","DOI":"10.1109\/JSTSP.2009.2015077","volume":"3","author":"C Li","year":"2009","unstructured":"Li, C., & Chen, T. (2009). Aesthetic visual quality assessment of paintings. IEEE Journal of Selected Topics in Signal Processing, 3(2), 236\u2013252.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"1443_CR63","unstructured":"Liem, C. (2016). \u201cTUD-MMC at mediaeval 2016: Predicting media interestingness task. In MediaEval workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org."},{"key":"1443_CR64","unstructured":"Liu, Y., Gu, Z., & Ko, T.\u00a0H. (2017). Predicting media interestingness via biased discriminant embedding and supervised manifold regression. In MediaEval workshop, Dublin, Ireland, September 13-15. (Vol.\u00a01984), CEUR-WS.org."},{"key":"1443_CR65","doi-asserted-by":"crossref","unstructured":"Liu, Y., Gu, Z., Ko, T.\u00a0H., & Hua, K.\u00a0A. (2018). Learning perceptual embeddings with two related tasks for joint predictions of media interestingness and emotions. In Proceedings of the ACM on international conference on multimedia retrieval, pp.\u00a0420\u2013427, ACM.","DOI":"10.1145\/3206025.3206071"},{"key":"1443_CR66","unstructured":"Liu, F., Niu, Y., & Gleicher M. (2009). Using web photos for measuring video frame interestingness. In Twenty-First international joint conference on artificial intelligence."},{"key":"1443_CR67","doi-asserted-by":"crossref","unstructured":"Liu, C., Zoph, B., Neumann, M., Shlens, J., Hua, W., Li, L.-J., et al. (2018). Progressive neural architecture search. In Proceedings of the European conference on computer vision (ECCV), 19\u201334.","DOI":"10.1007\/978-3-030-01246-5_2"},{"issue":"2","key":"1443_CR68","doi-asserted-by":"publisher","first-page":"91","DOI":"10.1023\/B:VISI.0000029664.99615.94","volume":"60","author":"DG Lowe","year":"2004","unstructured":"Lowe, D. G. (2004). Distinctive image features from scale-invariant keypoints. International Journal of Computer Vision, 60(2), 91\u2013110.","journal-title":"International Journal of Computer Vision"},{"key":"1443_CR69","doi-asserted-by":"crossref","unstructured":"Mann, H. B., & Whitney, D. R. (1947). On a test of whether one of two random variables is stochastically larger than the other. The Annals of Mathematical Statistics, pp. 50\u201360.","DOI":"10.1214\/aoms\/1177730491"},{"issue":"1","key":"1443_CR70","doi-asserted-by":"publisher","first-page":"5","DOI":"10.1007\/s11031-007-9053-1","volume":"31","author":"RR McCrae","year":"2007","unstructured":"McCrae, R. R. (2007). Aesthetic chills as a universal marker of openness to experience. Motivation and Emotion, 31(1), 5\u201311.","journal-title":"Motivation and Emotion"},{"key":"1443_CR71","doi-asserted-by":"publisher","first-page":"11","DOI":"10.1016\/j.neucom.2018.02.052","volume":"291","author":"S Mo","year":"2018","unstructured":"Mo, S., Niu, J., Su, Y., & Das, S. K. (2018). A novel feature set for video emotion recognition. Neurocomputing, 291, 11\u201320.","journal-title":"Neurocomputing"},{"key":"1443_CR72","doi-asserted-by":"publisher","first-page":"971","DOI":"10.1109\/TPAMI.2002.1017623","volume":"7","author":"T Ojala","year":"2002","unstructured":"Ojala, T., Pietik\u00e4inen, M., & M\u00e4enp\u00e4\u00e4, T. (2002). Multiresolution gray-scale and rotation invariant texture classification with local binary patterns. IEEE Transactions on Pattern Analysis & Machine Intelligence, 7, 971\u2013987.","journal-title":"IEEE Transactions on Pattern Analysis & Machine Intelligence"},{"issue":"3","key":"1443_CR73","doi-asserted-by":"publisher","first-page":"145","DOI":"10.1023\/A:1011139631724","volume":"42","author":"A Oliva","year":"2001","unstructured":"Oliva, A., & Torralba, A. (2001). Modeling the shape of the scene: A holistic representation of the spatial envelope. International Journal of Computer Vision, 42(3), 145\u2013175.","journal-title":"International Journal of Computer Vision"},{"key":"1443_CR74","doi-asserted-by":"crossref","unstructured":"Opitz, M., Waltner, G., Possegger, H., & Bischof, H. (2017). Bier-boosting independent embeddings robustly. In Proceedings of the IEEE international conference on computer vision, 5189\u20135198.","DOI":"10.1109\/ICCV.2017.555"},{"issue":"5","key":"1443_CR75","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1080\/1364557032000081654","volume":"7","author":"S Ovadia","year":"2004","unstructured":"Ovadia, S. (2004). Ratings and rankings: reconsidering the structure of values and their measurement. International Journal of Social Research Methodology, 7(5), 403\u2013414.","journal-title":"International Journal of Social Research Methodology"},{"key":"1443_CR76","doi-asserted-by":"crossref","unstructured":"Parekh, J., Tibrewal, H., & Parekh, S. (2018). Deep pairwise classification and ranking for predicting media interestingness. In Proceedings of the 2018 ACM on international conference on multimedia retrieval, ICMR, Yokohama, Japan, June 11-14., pp.\u00a0428\u2013433, ACM.","DOI":"10.1145\/3206025.3206078"},{"key":"1443_CR77","unstructured":"Permadi, R.\u00a0A., Putra, S.\u00a0G.\u00a0P., Helmiriawan, & Liem C.\u00a0C.\u00a0S. (2017). DUT-MMSR at mediaeval 2017: Predicting media interestingness task. In MediaEval workshop, Dublin, Ireland, September 13-15. (Vol.\u00a01984), CEUR-WS.org."},{"issue":"21","key":"1443_CR78","doi-asserted-by":"publisher","first-page":"22547","DOI":"10.1007\/s11042-017-4730-x","volume":"76","author":"J Poignant","year":"2017","unstructured":"Poignant, J., Bredin, H., & Barras, C. (2017). Multimodal person discovery in broadcast tv: lessons learned from mediaeval 2015. Multimedia Tools and Applications, 76(21), 22547\u201322567.","journal-title":"Multimedia Tools and Applications"},{"key":"1443_CR79","unstructured":"Randolph, J. J. (2005). \u201cFree-marginal multirater kappa (multirater k free): an alternative to fleiss\u2019 fixed-marginal multirater kappa\u201d, In Joensuu learning and instruction symposium. Finland: Joensuu."},{"key":"1443_CR80","unstructured":"Rayatdoost, S., & Soleymani, M. (2016). Ranking images and videos on visual interestingness by visual sentiment features. In MediaEval workshop, Hilversum, The Netherlands, October 20-21. (Vol.\u00a01739), CEUR-WS.org."},{"issue":"3","key":"1443_CR81","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., et al. (2015). Imagenet large scale visual recognition challenge. International Journal of Computer Vision, 115(3), 211\u2013252.","journal-title":"International Journal of Computer Vision"},{"issue":"3","key":"1443_CR82","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., et al. (2015). ImageNet large scale visual recognition challenge. International Journal of Computer Vision (IJCV), 115(3), 211\u2013252.","journal-title":"International Journal of Computer Vision (IJCV)"},{"key":"1443_CR83","doi-asserted-by":"crossref","unstructured":"Salesses, P., Schechtner, K., & Hidalgo, C.\u00a0A. (2013). The collaborative image of the city: mapping the inequality of urban perception PloS one 8(7).","DOI":"10.1371\/journal.pone.0068400"},{"key":"1443_CR84","doi-asserted-by":"crossref","unstructured":"Sanderson, M., & Zobel, J. (2005). Information retrieval system evaluation: effort, sensitivity, and reliability. In Proceedings of the 28th annual international ACM SIGIR conference on research and development in information retrieval, pp.\u00a0162\u2013169, ACM, August 15-19.","DOI":"10.1145\/1076034.1076064"},{"key":"1443_CR85","doi-asserted-by":"crossref","unstructured":"Selvaraju, R. R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., & Batra, D. (2017). Grad-cam: Visual explanations from deep networks via gradient-based localization. In Proceedings of the IEEE international conference on computer vision, 618\u2013626.","DOI":"10.1109\/ICCV.2017.74"},{"key":"1443_CR86","doi-asserted-by":"crossref","unstructured":"Shen, Y., Demarty, C.-H., & Duong, N.\u00a0Q.\u00a0K. (2017). Deep learning for multimodal-based video interestingness prediction. In IEEE international conference on multimedia and expo (ICME), pp.\u00a01003\u20131008, IEEE.","DOI":"10.1109\/ICME.2017.8019300"},{"key":"1443_CR87","unstructured":"Shen, Y., Demarty, C., Duong, N.\u00a0Q.\u00a0K. (2016). Technicolor@mediaeval 2016 predicting media interestingness task. In MediaEval workshop, Hilversum, The Netherlands, October 20-21., (Vol.\u00a01739), CEUR-WS.org."},{"issue":"1","key":"1443_CR88","doi-asserted-by":"publisher","first-page":"89","DOI":"10.1037\/1528-3542.5.1.89","volume":"5","author":"PJ Silvia","year":"2005","unstructured":"Silvia, P. J. (2005). What is interesting? exploring the appraisal structure of interest. Emotion, 5(1), 89.","journal-title":"Emotion"},{"issue":"1","key":"1443_CR89","doi-asserted-by":"publisher","first-page":"48","DOI":"10.1037\/a0014632","volume":"3","author":"PJ Silvia","year":"2009","unstructured":"Silvia, P. J. (2009). Looking past pleasure: anger, confusion, disgust, pride, surprise, and other unusual aesthetic emotions. Psychology of Aesthetics, Creativity, and the Arts, 3(1), 48.","journal-title":"Psychology of Aesthetics, Creativity, and the Arts"},{"key":"1443_CR90","unstructured":"Simonyan, K., & Zisserman, A. (2014). Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556."},{"key":"1443_CR91","unstructured":"Sivaraman, K., & Somappa, G. (2016). Moviescope: Movie trailer classification using deep neural networks. University of Virginia."},{"issue":"4","key":"1443_CR92","doi-asserted-by":"publisher","first-page":"411","DOI":"10.1016\/j.cviu.2009.03.011","volume":"114","author":"AF Smeaton","year":"2010","unstructured":"Smeaton, A. F., Over, P., & Doherty, A. R. (2010). Video shot boundary detection: Seven years of trecvid activity. Computer Vision and Image Understanding, 114(4), 411\u2013418.","journal-title":"Computer Vision and Image Understanding"},{"key":"1443_CR93","doi-asserted-by":"crossref","unstructured":"Soleymani, M. (2015) The quest for visual interest. In Proceedings of the 23rd ACM international conference on multimedia, pp.\u00a0919\u2013922, ACM.","DOI":"10.1145\/2733373.2806364"},{"key":"1443_CR94","doi-asserted-by":"crossref","unstructured":"Son, J., Jung, I., Park, K., & Han, B. (2015). Tracking-by-segmentation with online gradient boosting decision tree. In Proceedings of the IEEE international conference on computer vision, 3056\u20133064.","DOI":"10.1109\/ICCV.2015.350"},{"key":"1443_CR95","unstructured":"Springenberg, J.\u00a0T., Dosovitskiy, A., Brox, T., Riedmiller, M. (2014). Striving for simplicity: The all convolutional net. arXiv preprint arXiv:1412.6806."},{"key":"1443_CR96","doi-asserted-by":"crossref","unstructured":"Squalli-Houssaini, H., Duong, N.\u00a0Q.\u00a0K., Gwena\u00eblle, M., & Demarty, C.-H. (2018). Deep learning for predicting image memorability. In IEEE international conference on acoustics, speech and signal processing (ICASSP), pp.\u00a02371\u20132375, IEEE","DOI":"10.1109\/ICASSP.2018.8462292"},{"key":"1443_CR97","doi-asserted-by":"crossref","unstructured":"Sudhakaran, S., Escalera, S., & Lanz, O. (2020). Gate-shift networks for video action recognition. In Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR42600.2020.00118"},{"key":"1443_CR98","unstructured":"Touvron, H., Vedaldi, A., Douze, M., & J\u00e9gou, H. (2019). Fixing the train-test resolution discrepancy. Advances in Neural Information Processing Systems, 8250\u20138260."},{"key":"1443_CR99","doi-asserted-by":"crossref","unstructured":"Tran, D., Bourdev, L., Fergus, R., Torresani, L., & Paluri, M. (2015). Learning spatiotemporal features with 3d convolutional networks. In Proceedings of the IEEE international conference on computer vision, pp.\u00a04489\u20134497, IEEE.","DOI":"10.1109\/ICCV.2015.510"},{"key":"1443_CR100","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., & Feiszli, M. (2019). Video classification with channel-separated convolutional networks. In Proceedings of the IEEE international conference on computer vision, 5552\u20135561.","DOI":"10.1109\/ICCV.2019.00565"},{"key":"1443_CR101","doi-asserted-by":"crossref","unstructured":"Tran, D., Wang, H., Torresani, L., Ray, J., LeCun, Y., & Paluri, M. (2018). A closer look at spatiotemporal convolutions for action recognition. Proceedings of the IEEE conference on computer vision and pattern recognition, 6450\u20136459.","DOI":"10.1109\/CVPR.2018.00675"},{"key":"1443_CR102","doi-asserted-by":"crossref","unstructured":"Urbano, J., Marrero, M., & Mart\u00edn, D. (2013) On the measurement of test collection reliability. In The 36th International ACM SIGIR conference on research and development in information retrieval, pp.\u00a0393\u2013402, ACM, July 28 - August 1.","DOI":"10.1145\/2484028.2484038"},{"key":"1443_CR103","unstructured":"Vasudevan, A.\u00a0B., Gygli, M., Volokitin, A., & Van\u00a0Gool, L. (2016). Eth-cvl@ mediaeval 2016: Textual-visual embeddings and video2gif for video interestingness. In MediaEval workshop, Hilversum, The Netherlands, October 20-21., (Vol.\u00a01739), CEUR-WS.org."},{"key":"1443_CR104","doi-asserted-by":"crossref","unstructured":"Vigna, S. (2015). A weighted correlation index for rankings with ties. In Proceedings of the 24th international conference on World Wide Web, WWW Eds. A.\u00a0Gangemi, S.\u00a0Leonardi, and A.\u00a0Panconesi, pp.\u00a01166\u20131176, ACM, May 18-22.","DOI":"10.1145\/2736277.2741088"},{"key":"1443_CR105","doi-asserted-by":"crossref","unstructured":"Voorhees, E.\u00a0M. (1998). Variations in relevance judgments and the measurement of retrieval effectiveness In Proceedings of the 21st annual international ACM SIGIR conference on research and development in information retrieval. Eds. W.\u00a0B. Croft, A.\u00a0Moffat, C.\u00a0J. van Rijsbergen, R.\u00a0Wilkinson, and J.\u00a0Zobel, pp.\u00a0315\u2013323, ACM, August 24-28.","DOI":"10.1145\/290941.291017"},{"key":"1443_CR106","doi-asserted-by":"crossref","unstructured":"Wang, S., Chen, S., Zhao, J., & Jin, Q. (2018). Video interestingness prediction based on ranking model. In Proceedings of the joint workshop of the 4th workshop on affective social multimedia computing and first multi-modal affective computing of large-scale multimedia data, ASMMC-MMAC\u201918, pp.\u00a055\u201361, ACM.","DOI":"10.1145\/3267935.3267952"},{"key":"1443_CR107","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.\u00a0A., Oliva, A., & Torralba, A. (2010). Sun database: Large-scale scene recognition from abbey to zoo. In IEEE computer society conference on computer vision and pattern recognition, pp.\u00a03485\u20133492, IEEE.","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"1443_CR108","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., Tu, Z., & He, K. (2017). Aggregated residual transformations for deep neural networks. In Proceedings of the IEEE conference on computer vision and pattern recognition, 1492\u20131500.","DOI":"10.1109\/CVPR.2017.634"},{"key":"1443_CR109","unstructured":"Xu, B., Fu, Y., & Jiang, Y. (2016). Bigvid at mediaeval 2016: Predicting interestingness in images and videos. In MediaEval workshop, Hilversum, The Netherlands, October 20-21 (Vol.\u00a01739), CEUR-WS.org."},{"key":"1443_CR110","unstructured":"Yalniz, I.\u00a0Z., J\u00e9gou, H., Chen, K., Paluri, M., & Mahajan, D. (2019). Billion-scale semi-supervised learning for image classification. arXiv preprint arXiv:1905.00546."},{"issue":"4","key":"1443_CR111","doi-asserted-by":"publisher","first-page":"762","DOI":"10.1109\/TASL.2010.2064164","volume":"19","author":"Y-H Yang","year":"2011","unstructured":"Yang, Y.-H., & Chen, H. H. (2011). Ranking-based emotion recognition for music organization and retrieval. IEEE Transactions on Audio, Speech, and Language Processing, 19(4), 762\u2013774.","journal-title":"IEEE Transactions on Audio, Speech, and Language Processing"},{"key":"1443_CR112","doi-asserted-by":"crossref","unstructured":"Yannakakis, G.\u00a0N., & Hallam, J. (2011). Ranking vs. preference: a comparative study of self-reporting In: International conference on affective computing and intelligent interaction, pp.\u00a0437\u2013446, Springer.","DOI":"10.1007\/978-3-642-24600-5_47"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-021-01443-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-021-01443-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,18]],"date-time":"2022-12-18T12:50:51Z","timestamp":1671367851000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-021-01443-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,2,22]]},"references-count":112,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2021,5]]}},"alternative-id":["1443"],"URL":"https:\/\/doi.org\/10.1007\/s11263-021-01443-1","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,2,22]]},"assertion":[{"value":"13 May 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 January 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"22 February 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}