{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T15:11:17Z","timestamp":1786115477515,"version":"3.56.0"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032045577","type":"print"},{"value":"9783032045584","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,9,12]],"date-time":"2025-09-12T00:00:00Z","timestamp":1757635200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,9,12]],"date-time":"2025-09-12T00:00:00Z","timestamp":1757635200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-04558-4_10","type":"book-chapter","created":{"date-parts":[[2025,9,11]],"date-time":"2025-09-11T11:16:33Z","timestamp":1757589393000},"page":"119-127","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Small Transformer Architectures for\u00a0Task Switching"],"prefix":"10.1007","author":[{"given":"Claudius","family":"Gros","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,9,12]]},"reference":[{"key":"10_CR1","doi-asserted-by":"crossref","unstructured":"Bietti, A., Cabannes, V., Bouchacourt, D., Jegou, H., Bottou, L.: Birth of a transformer: A memory viewpoint. Advances in Neural Information Processing Systems 36 (2024)","DOI":"10.52202\/075280-0077"},{"issue":"12","key":"10_CR2","first-page":"1","volume":"56","author":"S Chen","year":"2024","unstructured":"Chen, S., Zhang, Y., Yang, Q.: Multi-task learning in natural language processing: an overview. ACM Comput. Surv. 56(12), 1\u201332 (2024)","journal-title":"ACM Comput. Surv."},{"key":"10_CR3","unstructured":"Deletang, G., et\u00a0al.: Neural networks and the chomsky hierarchy. In: The Eleventh International Conference on Learning Representations (2022)"},{"key":"10_CR4","unstructured":"Elhage, N., et\u00a0al.: Toy models of superposition. arXiv preprint arXiv:2209.10652 (2022)"},{"key":"10_CR5","unstructured":"Gros, C.: Reorganizing attention-space geometry with expressive attention. arXiv preprint arXiv:2407.18601 (2024)"},{"key":"10_CR6","unstructured":"Gu, A., Dao, T.: Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)"},{"key":"10_CR7","doi-asserted-by":"crossref","unstructured":"Gupta, A., Yu, J., Zhao, T.Z., Kumar, V., Rovinsky, A., Xu, K., Devlin, T., Levine, S.: Reset-free reinforcement learning via multi-task learning: Learning dexterous manipulation behaviors without human intervention. In: 2021 IEEE International Conference on Robotics and Automation (ICRA), pp. 6664\u20136671, IEEE (2021)","DOI":"10.1109\/ICRA48506.2021.9561384"},{"key":"10_CR8","unstructured":"Hoffmann, J., et\u00a0al.: Training compute-optimal large language models. arXiv preprint arXiv:2203.15556 (2022)"},{"key":"10_CR9","doi-asserted-by":"crossref","unstructured":"Ivgi, M., Carmon, Y., Berant, J.: Scaling laws under the microscope: Predicting transformer performance from small scale experiments. In: Findings of the Association for Computational Linguistics: EMNLP 2022, pp. 7354\u20137371 (2022)","DOI":"10.18653\/v1\/2022.findings-emnlp.544"},{"key":"10_CR10","unstructured":"Kaplan, J., et al.: Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)"},{"key":"10_CR11","unstructured":"Katharopoulos, A., Vyas, A., Pappas, N., Fleuret, F.: Transformers are rnns: Fast autoregressive transformers with linear attention. In: International conference on machine learning, pp. 5156\u20135165, PMLR (2020)"},{"key":"10_CR12","doi-asserted-by":"crossref","unstructured":"Knight, P., Duan, R.: Multi-task learning with summary statistics. Advances in Neural Information Processing Systems 36 (2024)","DOI":"10.52202\/075280-2350"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Kumar, V., et al.: Robohive: A unified framework for robot learning. Advances in Neural Information Processing Systems 36 (2024)","DOI":"10.52202\/075280-1918"},{"key":"10_CR14","doi-asserted-by":"publisher","first-page":"157","DOI":"10.1162\/tacl_a_00638","volume":"12","author":"NF Liu","year":"2024","unstructured":"Liu, N.F., Lin, K., Hewitt, J., Paranjape, A., Bevilacqua, M., Petroni, F., Liang, P.: Lost in the middle: How language models use long contexts. Trans. Assoc. Comput. Linguist. 12, 157\u2013173 (2024)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10_CR15","unstructured":"Naveed, H., et al.: A comprehensive overview of large language models. arXiv preprint arXiv:2307.06435 (2023)"},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"Neumann, O., Gros, C.: Scaling laws for a multi-agent reinforcement learning model. In: Deep Reinforcement Learning Workshop NeurIPS 2022 (2022)","DOI":"10.14428\/esann\/2022.ES2022-53"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Neumann, O., Gros, C.: Alphazero neural scaling and zipf\u2019s law: a tale of board games and power laws. arXiv preprint arXiv:2412.11979 (2024)","DOI":"10.52202\/085713-1957"},{"key":"10_CR18","unstructured":"Press, O., Smith, N.A., Lewis, M.: Train short, test long: Attention with linear biases enables input length extrapolation. arXiv preprint arXiv:2108.12409 (2021)"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"S\u00e1ndor, B., Nowak, M., Koglin, T., Martin, L., Gros, C.: Kick control: using the attracting states arising within the sensorimotor loop of self-organized robots as motor primitives. Frontiers in neurorobotics 12 (2018)","DOI":"10.3389\/fnbot.2018.00040"},{"issue":"13","key":"10_CR20","doi-asserted-by":"publisher","first-page":"1133","DOI":"10.1177\/02783649231201196","volume":"42","author":"M Saveriano","year":"2023","unstructured":"Saveriano, M., Abu-Dakka, F.J., Kramberger, A., Peternel, L.: Dynamic movement primitives in robotics: a tutorial survey. Int. J. Robot. Res. 42(13), 1133\u20131184 (2023)","journal-title":"Int. J. Robot. Res."},{"key":"10_CR21","doi-asserted-by":"crossref","unstructured":"Shen, X., Li, D., Leng, R., Qin, Z., Sun, W., Zhong, Y.: Scaling laws for linear complexity language models. arXiv preprint arXiv:2406.16690 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.916"},{"key":"10_CR22","doi-asserted-by":"publisher","first-page":"543","DOI":"10.1162\/tacl_a_00663","volume":"12","author":"L Strobl","year":"2024","unstructured":"Strobl, L., Merrill, W., Weiss, G., Chiang, D., Angluin, D.: What formal languages can transformers express? a survey. Trans. Assoc. Comput. Linguist. 12, 543\u2013561 (2024)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10_CR23","unstructured":"Tay, Y., et al.: Long range arena: A benchmark for efficient transformers. arXiv preprint arXiv:2011.04006 (2020)"},{"key":"10_CR24","unstructured":"Tong, W.L., Pehlevan, C.: Mlps learn in-context. arXiv preprint arXiv:2405.15618 (2024)"},{"key":"10_CR25","unstructured":"Vaswani, A., et al.: Attention is all you need. Advances in neural information processing systems 30 (2017)"},{"key":"10_CR26","unstructured":"Wang, S., Li, B.Z., Khabsa, M., Fang, H., Ma, H.: Linformer: Self-attention with linear complexity. arXiv preprint arXiv:2006.04768 (2020)"},{"key":"10_CR27","unstructured":"Wu, H., Wu, J., Xu, J., Wang, J., Long, M.: Flowformer: linearizing transformers with conservation flows. arXiv preprint arXiv:2202.06258 (2022)"},{"key":"10_CR28","unstructured":"Zhang, Y., Backurs, A., Bubeck, S., Eldan, R., Gunasekar, S., Wagner, T.: Unveiling transformers with lego: a synthetic reasoning task. arXiv preprint arXiv:2206.04301 (2022)"},{"issue":"12","key":"10_CR29","doi-asserted-by":"publisher","first-page":"5586","DOI":"10.1109\/TKDE.2021.3070203","volume":"34","author":"Y Zhang","year":"2021","unstructured":"Zhang, Y., Yang, Q.: A survey on multi-task learning. IEEE Trans. Knowl. Data Eng. 34(12), 5586\u20135609 (2021)","journal-title":"IEEE Trans. Knowl. Data Eng."}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks and Machine Learning \u2013 ICANN 2025"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-04558-4_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,7]],"date-time":"2026-08-07T14:26:02Z","timestamp":1786112762000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-04558-4_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,9,12]]},"ISBN":["9783032045577","9783032045584"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-04558-4_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,9,12]]},"assertion":[{"value":"12 September 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that\u00a0are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"ICANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kaunas","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lithuania","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 September 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"12 September 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"34","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icann2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/e-nns.org\/icann2025\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}