{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,27]],"date-time":"2025-03-27T13:00:48Z","timestamp":1743080448125,"version":"3.40.3"},"publisher-location":"Cham","reference-count":43,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031434204"},{"type":"electronic","value":"9783031434211"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-43421-1_6","type":"book-chapter","created":{"date-parts":[[2023,9,17]],"date-time":"2023-09-17T20:37:24Z","timestamp":1694983044000},"page":"90-106","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Improving Autoregressive NLP Tasks via\u00a0Modular Linearized Attention"],"prefix":"10.1007","author":[{"given":"Victor","family":"Agostinelli","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lizhong","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,9,18]]},"reference":[{"key":"6_CR1","doi-asserted-by":"crossref","unstructured":"Ashtari, P., Sima, D.M., Lathauwer, L.D., Sappey-Marinier, D., Maes, F., Huffel, S.V.: Factorizer: a scalable interpretable approach to context modeling for medical image segmentation. Med. Image Anal. 84, 102706 (2022)","DOI":"10.1016\/j.media.2022.102706"},{"key":"6_CR2","unstructured":"Baevski, A., Auli, M.: Adaptive input representations for neural language modeling (2019)"},{"key":"6_CR3","doi-asserted-by":"publisher","unstructured":"Beltagy, I., Peters, M.E., Cohan, A.: Longformer: The long-document transformer (2020). https:\/\/doi.org\/10.48550\/ARXIV.2004.05150","DOI":"10.48550\/ARXIV.2004.05150"},{"key":"6_CR4","doi-asserted-by":"publisher","unstructured":"Bentivogli, L., et al.: Cascade versus direct speech translation: Do the differences still make a difference? In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing. vol. 1: Long Papers), pp. 2873\u20132887. Association for Computational Linguistics, Online (Aug 2021). https:\/\/doi.org\/10.18653\/v1\/2021.acl-long.224, https:\/\/aclanthology.org\/2021.acl-long.224","DOI":"10.18653\/v1\/2021.acl-long.224"},{"key":"6_CR5","unstructured":"Berndt, D.J., Clifford, J.: Using dynamic time warping to find patterns in time series. In: Proceedings of the 3rd International Conference on Knowledge Discovery and Data Mining, pp. 359\u2013370. AAAIWS 1994, AAAI Press (1994)"},{"key":"6_CR6","doi-asserted-by":"publisher","unstructured":"Child, R., Gray, S., Radford, A., Sutskever, I.: Generating long sequences with sparse transformers (2019). https:\/\/doi.org\/10.48550\/ARXIV.1904.10509, https:\/\/arxiv.org\/abs\/1904.10509","DOI":"10.48550\/ARXIV.1904.10509"},{"key":"6_CR7","doi-asserted-by":"publisher","unstructured":"Choromanski, K., et al.: Rethinking attention with performers (2020). https:\/\/doi.org\/10.48550\/ARXIV.2009.14794, https:\/\/arxiv.org\/abs\/2009.14794","DOI":"10.48550\/ARXIV.2009.14794"},{"key":"6_CR8","doi-asserted-by":"publisher","unstructured":"Clark, K., Khandelwal, U., Levy, O., Manning, C.D.: What does BERT look at? An analysis of BERT\u2019s attention (2019). https:\/\/doi.org\/10.48550\/ARXIV.1906.04341, https:\/\/arxiv.org\/abs\/1906.04341","DOI":"10.48550\/ARXIV.1906.04341"},{"key":"6_CR9","doi-asserted-by":"publisher","unstructured":"Dai, Z., Yang, Z., Yang, Y., Carbonell, J., Le, Q.V., Salakhutdinov, R.: Transformer-XL: Attentive language models beyond a fixed-length context (2019). https:\/\/doi.org\/10.48550\/ARXIV.1901.02860, https:\/\/arxiv.org\/abs\/1901.02860","DOI":"10.48550\/ARXIV.1901.02860"},{"key":"6_CR10","doi-asserted-by":"publisher","unstructured":"Dosovitskiy, A., et al.: An image is worth 16x16 words: Transformers for image recognition at scale (2020). https:\/\/doi.org\/10.48550\/ARXIV.2010.11929, https:\/\/arxiv.org\/abs\/2010.11929","DOI":"10.48550\/ARXIV.2010.11929"},{"key":"6_CR11","doi-asserted-by":"crossref","unstructured":"Hu, C., et al.: RankNAS: Efficient neural architecture search by pairwise ranking (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.191"},{"key":"6_CR12","doi-asserted-by":"publisher","unstructured":"Huang, C.Z.A., et al.: Music transformer (2018). https:\/\/doi.org\/10.48550\/ARXIV.1809.04281, https:\/\/arxiv.org\/abs\/1809.04281","DOI":"10.48550\/ARXIV.1809.04281"},{"key":"6_CR13","unstructured":"Ito, K., Johnson, L.: The LJ speech dataset (2017)"},{"key":"6_CR14","doi-asserted-by":"publisher","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are RNNs: Fast autoregressive transformers with linear attention (2020). https:\/\/doi.org\/10.48550\/ARXIV.2006.16236, https:\/\/arxiv.org\/abs\/2006.16236","DOI":"10.48550\/ARXIV.2006.16236"},{"key":"6_CR15","doi-asserted-by":"publisher","unstructured":"Kitaev, N., Kaiser, L., Levskaya, A.: Reformer: The efficient transformer (2020). https:\/\/doi.org\/10.48550\/ARXIV.2001.04451, https:\/\/arxiv.org\/abs\/2001.04451","DOI":"10.48550\/ARXIV.2001.04451"},{"key":"6_CR16","doi-asserted-by":"publisher","unstructured":"Kovaleva, O., Romanov, A., Rogers, A., Rumshisky, A.: Revealing the dark secrets of BERT (2019). https:\/\/doi.org\/10.48550\/ARXIV.1908.08593, https:\/\/arxiv.org\/abs\/1908.08593","DOI":"10.48550\/ARXIV.1908.08593"},{"key":"6_CR17","doi-asserted-by":"crossref","unstructured":"Kubichek, R.F.: Mel-cepstral distance measure for objective speech quality assessment. In: Proceedings of IEEE Pacific Rim Conference on Communications Computers and Signal Processing. vol. 1, pp. 125\u2013128 (1993)","DOI":"10.1109\/PACRIM.1993.407206"},{"key":"6_CR18","doi-asserted-by":"publisher","unstructured":"Lee, J., Lee, Y., Kim, J., Kosiorek, A.R., Choi, S., Teh, Y.W.: Set transformer: A framework for attention-based permutation-invariant neural networks (2018). https:\/\/doi.org\/10.48550\/ARXIV.1810.00825, https:\/\/arxiv.org\/abs\/1810.00825","DOI":"10.48550\/ARXIV.1810.00825"},{"key":"6_CR19","doi-asserted-by":"publisher","unstructured":"Li, N., Liu, S., Liu, Y., Zhao, S., Liu, M., Zhou, M.: Neural speech synthesis with transformer network (2018). https:\/\/doi.org\/10.48550\/ARXIV.1809.08895, https:\/\/arxiv.org\/abs\/1809.08895","DOI":"10.48550\/ARXIV.1809.08895"},{"key":"6_CR20","unstructured":"Liu, Y., et al.: RoBERTa: A robustly optimized BERT pretraining approach (2019)"},{"key":"6_CR21","unstructured":"Liu, Z., et al.: Neural architecture search on efficient transformers and beyond (2022)"},{"key":"6_CR22","doi-asserted-by":"crossref","unstructured":"Ma, M., et al.: STACL: simultaneous translation with implicit anticipation and controllable latency using prefix-to-prefix framework. In: Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics, pp. 3025\u20133036. Association for Computational Linguistics (ACL), Florence, Italy (2019)","DOI":"10.18653\/v1\/P19-1289"},{"key":"6_CR23","unstructured":"Ma, X., Pino, J., Cross, J., Puzon, L., Gu, J.: Monotonic multi head attention. In: International Conference on Learning Representations (2020)"},{"key":"6_CR24","unstructured":"Ma, X., Pino, J., Cross, J., Puzon, L., Gu, J.: SimulMT to simulST: adapting simultaneous text translation to end-to-end simultaneous speech translation. In: Proceedings of 2020 Asia-Pacific Chapter of the Association for Computational Linguistics and the International Joint Conference on Natural Language Processing (2020)"},{"key":"6_CR25","doi-asserted-by":"publisher","unstructured":"Ma, X., Dousti, M.J., Wang, C., Gu, J., Pino, J.: SimulEval: An evaluation toolkit for simultaneous translation (2020). https:\/\/doi.org\/10.48550\/ARXIV.2007.16193, https:\/\/arxiv.org\/abs\/2007.16193","DOI":"10.48550\/ARXIV.2007.16193"},{"key":"6_CR26","doi-asserted-by":"publisher","unstructured":"Madani, A., et al.: ProGen: Language modeling for protein generation (2020). https:\/\/doi.org\/10.48550\/ARXIV.2004.03497, https:\/\/arxiv.org\/abs\/2004.03497","DOI":"10.48550\/ARXIV.2004.03497"},{"key":"6_CR27","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M.: Montreal forced aligner: trainable text-speech alignment using kaldi. In: Interspeech (2017)","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"6_CR28","doi-asserted-by":"publisher","unstructured":"Ott, M., et al.: fairseq: A fast, extensible toolkit for sequence modeling (2019). https:\/\/doi.org\/10.48550\/ARXIV.1904.01038, https:\/\/arxiv.org\/abs\/1904.01038","DOI":"10.48550\/ARXIV.1904.01038"},{"key":"6_CR29","doi-asserted-by":"publisher","unstructured":"Parmar, N., et al.: Image transformer (2018). https:\/\/doi.org\/10.48550\/ARXIV.1802.05751, https:\/\/arxiv.org\/abs\/1802.05751","DOI":"10.48550\/ARXIV.1802.05751"},{"key":"6_CR30","doi-asserted-by":"publisher","unstructured":"Peng, H., Pappas, N., Yogatama, D., Schwartz, R., Smith, N.A., Kong, L.: Random feature attention (2021). https:\/\/doi.org\/10.48550\/ARXIV.2103.02143, https:\/\/arxiv.org\/abs\/2103.02143","DOI":"10.48550\/ARXIV.2103.02143"},{"key":"6_CR31","doi-asserted-by":"crossref","unstructured":"Post, M.: A call for clarity in reporting BLEU scores. In: Proceedings of the Third Conference on Machine Translation: Research Papers, pp. 186\u2013191. Association for Computational Linguistics, Belgium, Brussels (Oct 2018), https:\/\/www.aclweb.org\/anthology\/W18-6319","DOI":"10.18653\/v1\/W18-6319"},{"key":"6_CR32","unstructured":"Qin, Z., et al.: cosFormer: Rethinking softmax in attention. In: International Conference on Learning Representations (2022), https:\/\/openreview.net\/forum?id=Bl8CQrx2Up4"},{"key":"6_CR33","doi-asserted-by":"publisher","unstructured":"Ren, Y., et al.: FastSpeech 2: Fast and high-quality end-to-end text to speech (2020). https:\/\/doi.org\/10.48550\/ARXIV.2006.04558, https:\/\/arxiv.org\/abs\/2006.04558","DOI":"10.48550\/ARXIV.2006.04558"},{"key":"6_CR34","doi-asserted-by":"publisher","unstructured":"Skerry-Ryan, R., et al.: Towards end-to-end prosody transfer for expressive speech synthesis with Tacotron (2018). https:\/\/doi.org\/10.48550\/ARXIV.1803.09047, https:\/\/arxiv.org\/abs\/1803.09047","DOI":"10.48550\/ARXIV.1803.09047"},{"key":"6_CR35","doi-asserted-by":"publisher","unstructured":"Sukhbaatar, S., Grave, E., Bojanowski, P., Joulin, A.: Adaptive attention span in transformers (2019). https:\/\/doi.org\/10.48550\/ARXIV.1905.07799, https:\/\/arxiv.org\/abs\/1905.07799","DOI":"10.48550\/ARXIV.1905.07799"},{"key":"6_CR36","unstructured":"Tay, Y., et al.: Long range arena : a benchmark for efficient transformers. In: International Conference on Learning Representations (2021). https:\/\/openreview.net\/forum?id=qVyeW-grC2k"},{"key":"6_CR37","unstructured":"Vaswani, A., et al.: Attention is all you need. In: Proceedings of 31st Conference on Neural Information Processing Systems (NIPS 2017) (2017)"},{"key":"6_CR38","doi-asserted-by":"publisher","unstructured":"Wang, C., et al.: fairseq $$s^2$$: A scalable and integrable speech synthesis toolkit (2021). https:\/\/doi.org\/10.48550\/ARXIV.2109.06912, https:\/\/arxiv.org\/abs\/2109.06912","DOI":"10.48550\/ARXIV.2109.06912"},{"key":"6_CR39","doi-asserted-by":"publisher","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: Self-attention with linear complexity (2020). https:\/\/doi.org\/10.48550\/ARXIV.2006.04768, https:\/\/arxiv.org\/abs\/2006.04768","DOI":"10.48550\/ARXIV.2006.04768"},{"key":"6_CR40","doi-asserted-by":"publisher","unstructured":"Weiss, R.J., Skerry-Ryan, R., Battenberg, E., Mariooryad, S., Kingma, D.P.: Wave-tacotron: Spectrogram-free end-to-end text-to-speech synthesis (2020). https:\/\/doi.org\/10.48550\/ARXIV.2011.03568, https:\/\/arxiv.org\/abs\/2011.03568","DOI":"10.48550\/ARXIV.2011.03568"},{"key":"6_CR41","doi-asserted-by":"publisher","unstructured":"Wu, Q., Lan, Z., Qian, K., Gu, J., Geramifard, A., Yu, Z.: Memformer: A memory-augmented transformer for sequence modeling (2020). https:\/\/doi.org\/10.48550\/ARXIV.2010.06891, https:\/\/arxiv.org\/abs\/2010.06891","DOI":"10.48550\/ARXIV.2010.06891"},{"key":"6_CR42","doi-asserted-by":"crossref","unstructured":"Xiong, Y., et al.: Nystr\u00f6mformer: A nystr\u00f6m-based algorithm for approximating self-attention (2021)","DOI":"10.1609\/aaai.v35i16.17664"},{"key":"6_CR43","doi-asserted-by":"publisher","unstructured":"Zaheer, M., et al.: Big bird: Transformers for longer sequences (2020). https:\/\/doi.org\/10.48550\/ARXIV.2007.14062, https:\/\/arxiv.org\/abs\/2007.14062","DOI":"10.48550\/ARXIV.2007.14062"}],"container-title":["Lecture Notes in Computer Science","Machine Learning and Knowledge Discovery in Databases: Research Track"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-43421-1_6","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T06:56:25Z","timestamp":1730098585000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-43421-1_6"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031434204","9783031434211"],"references-count":43,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-43421-1_6","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"18 September 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECML PKDD","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Joint European Conference on Machine Learning and Knowledge Discovery in Databases","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Turin","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18 September 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22 September 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecml2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/2023.ecmlpkdd.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"829","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"196","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"24% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.63","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Applied Data Science Track: 239 submissions, 58 accepted papers; Demo Track: 31 submissions, 16 accepted papers.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}