{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:19:29Z","timestamp":1750220369462,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":28,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3481543","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T06:57:34Z","timestamp":1634540254000},"page":"1167-1175","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Distantly Supervised Semantic Text Detection and Recognition for Broadcast Sports Videos Understanding"],"prefix":"10.1145","author":[{"given":"Avijit","family":"Shah","sequence":"first","affiliation":[{"name":"Yahoo! Research, Sunnyvale, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Topojoy","family":"Biswas","sequence":"additional","affiliation":[{"name":"Yahoo! Research, Sunnyvale, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sathish","family":"Ramadoss","sequence":"additional","affiliation":[{"name":"Yahoo! Research, Sunnyvale, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Deven Santosh","family":"Shah","sequence":"additional","affiliation":[{"name":"Microsoft, Mountain View, CA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2015.100"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"crossref","unstructured":"S. Giancola M. Amine T. Dghaily and B. Ghanem. 2018. SoccerNet: A Scalable Dataset for Action Spotting in Soccer Videos. (2018) 1792--179210. https:\/\/doi.org\/10.1109\/CVPRW.2018.00223  S. Giancola M. Amine T. Dghaily and B. Ghanem. 2018. SoccerNet: A Scalable Dataset for Action Spotting in Soccer Videos. (2018) 1792--179210. https:\/\/doi.org\/10.1109\/CVPRW.2018.00223","DOI":"10.1109\/CVPRW.2018.00223"},{"volume-title":"Harnessing deep neural networks with logic rules. arXiv preprint arXiv:1603.06318","year":"2016","author":"Hu Zhiting","key":"e_1_3_2_2_3_1"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.5555\/3327546.3327711"},{"key":"e_1_3_2_2_5_1","unstructured":"Max Jaderberg Karen Simonyan Andrea Vedaldi and Andrew Zisserman. 2014. Synthetic Data and Artificial Neural Networks for Natural Scene Text Recognition. (2014).  Max Jaderberg Karen Simonyan Andrea Vedaldi and Andrew Zisserman. 2014. Synthetic Data and Artificial Neural Networks for Natural Scene Text Recognition. (2014)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0823-z"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333942"},{"volume-title":"The Robust Reading Competition Annotation and Evaluation Platform. In 2018 13th IAPR International Workshop on Document Analysis Systems (DAS). 61--66","year":"2018","author":"Karatzas D.","key":"e_1_3_2_2_8_1"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2011.295"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2013.221"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1117\/12.2575834"},{"volume-title":"A Single-Shot Oriented Scene Text Detector. CoRR abs\/1801.02765","year":"2018","author":"Liao Minghui","key":"e_1_3_2_2_12_1"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"crossref","unstructured":"H. Liu and B. Bhanu. 2019. Pose-Guided R-CNN for Jersey Number Recognition in Sports. (2019) 2457--2466. https:\/\/doi.org\/10.1109\/CVPRW.2019.00301  H. Liu and B. Bhanu. 2019. Pose-Guided R-CNN for Jersey Number Recognition in Sports. (2019) 2457--2466. https:\/\/doi.org\/10.1109\/CVPRW.2019.00301","DOI":"10.1109\/CVPRW.2019.00301"},{"volume-title":"Berg","year":"2015","author":"Liu Wei","key":"e_1_3_2_2_14_1"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2012.242"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-004-0134-3"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-011-0878-y"},{"volume-title":"Hemayed","year":"2021","author":"Nady Ahmed","key":"e_1_3_2_2_18_1"},{"volume-title":"Ryoo","year":"2018","author":"Piergiovanni AJ","key":"e_1_3_2_2_19_1"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"crossref","unstructured":"Vignesh Ramanathan Jonathan Huang Sami Abu-El-Haija Alexander Gorban Kevin Murphy and Li Fei-Fei. 2016. Detecting events and key actors in multiperson videos. arXiv:1511.02917 [cs.CV]  Vignesh Ramanathan Jonathan Huang Sami Abu-El-Haija Alexander Gorban Kevin Murphy and Li Fei-Fei. 2016. Detecting events and key actors in multiperson videos. arXiv:1511.02917 [cs.CV]","DOI":"10.1109\/CVPR.2016.332"},{"volume-title":"Predictive Biases in Natural Language Processing Models: A Conceptual Framework and Overview. Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics (2020","year":"2020","author":"Shah Deven Santosh","key":"e_1_3_2_2_21_1"},{"volume-title":"Belongie","year":"2017","author":"Shi Baoguang","key":"e_1_3_2_2_22_1"},{"key":"e_1_3_2_2_23_1","unstructured":"Baoguang Shi Xiang Bai and Cong Yao. 2015. An End-to-End Trainable Neural Network for Image-based Sequence Recognition and Its Application to Scene Text Recognition. arXiv:1507.05717 [cs.CV]  Baoguang Shi Xiang Bai and Cong Yao. 2015. An End-to-End Trainable Neural Network for Image-based Sequence Recognition and Its Application to Scene Text Recognition. arXiv:1507.05717 [cs.CV]"},{"volume-title":"Detecting Text in Natural Image with Connectionist Text Proposal Network. CoRR abs\/1609.03605","year":"2016","author":"Tian Zhi","key":"e_1_3_2_2_24_1"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1117\/12.2588177"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.31193\/ssap.01.9787520107174"},{"volume-title":"Fine-Grained Video Captioning for Sports Narrative. (June","year":"2018","author":"Yu Huanyu","key":"e_1_3_2_2_27_1"},{"volume-title":"EAST: An Efficient and Accurate Scene Text Detector. arXiv:1704.03155 [cs.CV]","year":"2017","author":"Zhou Xinyu","key":"e_1_3_2_2_28_1"}],"event":{"name":"MM '21: ACM Multimedia Conference","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Virtual Event China","acronym":"MM '21"},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3481543","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3481543","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:17:35Z","timestamp":1750191455000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3481543"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":28,"alternative-id":["10.1145\/3474085.3481543","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3481543","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}