{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,8]],"date-time":"2026-03-08T01:38:51Z","timestamp":1772933931772,"version":"3.50.1"},"reference-count":79,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T00:00:00Z","timestamp":1765152000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,12,8]]},"DOI":"10.1109\/bigdata66926.2025.11402021","type":"proceedings-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T20:57:57Z","timestamp":1772830677000},"page":"1034-1043","source":"Crossref","is-referenced-by-count":0,"title":["A Theoretical Framework Bridging Attention and SVM Optimization Dynamics and Sparsity"],"prefix":"10.1109","author":[{"given":"Zhihang","family":"Li","sequence":"first","affiliation":[{"name":"University of California, San Diego,California,US"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhao","family":"Song","sequence":"additional","affiliation":[{"name":"University of California, Berkeley,California,US"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiale","family":"Zhao","sequence":"additional","affiliation":[{"name":"Guangdong University of Technology,Guangdong,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref2","author":"Devlin","year":"2018","journal-title":"Bert: Pre-training of deep bidirectional transformers for language understanding"},{"key":"ref3","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown","year":"2020","journal-title":"NeurIPS"},{"key":"ref4","author":"Dosovitskiy","year":"2020","journal-title":"An image is worth 16 x 16 words: Transformers for image recognition at scale"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"ref6","volume-title":"Language models are unsupervised multitask learners","author":"Radford","year":"2019"},{"key":"ref7","article-title":"OpenAI","year":"2022","journal-title":"Openai: Introducing chatgpt"},{"key":"ref8","article-title":"OpenAI","year":"2023","journal-title":"Gpt-4 technical report"},{"key":"ref9","author":"Kaplan","year":"2020","journal-title":"Scaling laws for neural language models"},{"key":"ref10","author":"Sardana","year":"2023","journal-title":"Beyond chinchilla-optimal: Accounting for inference in language model scaling laws"},{"key":"ref11","author":"Wang","year":"2020","journal-title":"Linformer: Self-attention with linear complexity"},{"key":"ref12","author":"Peng","year":"2021","journal-title":"Random feature attention"},{"key":"ref13","first-page":"5156","article-title":"Transformers are rnns: Fast autoregressive transformers with linear attention","volume-title":"International conference on machine learning","author":"Katharopoulos","year":"2020"},{"key":"ref14","first-page":"2793","article-title":"Attention is not all you need: Pure attention loses rank doubly exponentially with depth","volume-title":"NeurIPS","author":"Dong","year":"2021"},{"key":"ref15","author":"Gu","year":"2021","journal-title":"Efficiently modeling long sequences with structured state spaces"},{"key":"ref16","author":"Beltagy","year":"2020","journal-title":"Longformer: The long-document transformer"},{"key":"ref17","author":"Choromanski","year":"2020","journal-title":"Rethinking attention with performers"},{"key":"ref18","first-page":"17283","article-title":"Big bird: Transformers for longer sequences","volume":"33","author":"Zaheer","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.sustainlp-1.5"},{"key":"ref20","author":"Xiao","year":"2023","journal-title":"Efficient streaming language models with attention sinks"},{"key":"ref21","article-title":"Deja vu: Contextual sparsity for efficient 11 ms at inference time","author":"Liu","year":"2023","journal-title":"Manuscript"},{"key":"ref22","author":"Tarzanagh","year":"2023","journal-title":"Transformers as support vector machines"},{"key":"ref23","author":"Nanda","year":"2023","journal-title":"Progress measures for grokking via mechanistic interpretability"},{"key":"ref24","author":"Morwani","year":"2023","journal-title":"Feature emergence via margin maxi-mization: case studies in algebraic tasks"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1282"},{"key":"ref26","first-page":"2733","article-title":"Designing and interpreting probes with control tasks","volume-title":"Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP)","author":"Hewitt"},{"key":"ref27","author":"Tenney","year":"2019","journal-title":"What do you learn from context? probing for sentence structure in contextualized word representations"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1162\/coli_a_00422"},{"key":"ref29","doi-asserted-by":"crossref","first-page":"455","DOI":"10.18653\/v1\/2020.conll-1.37","article-title":"On the computational power of transformers and its implications in sequence modeling","volume-title":"Proceedings of the 24th Conference on Computational Natural Language Learning. Online: Association for Computational Linguistics","author":"Bhattamishra","year":"2020"},{"key":"ref30","article-title":"Are transformers universal approximators of sequence-to-sequence functions?","volume-title":"International Conference on Learning Representations","author":"Yun","year":"2020"},{"key":"ref31","first-page":"17413","article-title":"Scatterbrain: Unifying sparse and low-rank attention","volume":"34","author":"Chen","year":"2021","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref32","author":"Hu","year":"2025","journal-title":"Universal approximation with softmax attention"},{"key":"ref33","author":"Liu","year":"2025","journal-title":"Attention mechanism, max-affine partition, and universal approximation"},{"key":"ref34","author":"Hu","year":"2025","journal-title":"Minimalist softmax attention provably learns constrained boolean functions"},{"key":"ref35","doi-asserted-by":"crossref","first-page":"7096","DOI":"10.18653\/v1\/2020.emnlp-main.576","article-title":"On the Ability and Limitations of Transformers to Recognize Formal Languages","volume-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)","author":"Bhattamishra","year":"2020"},{"key":"ref36","first-page":"4301","article-title":"How can self-attention networks recognize Dyck-n languages?","author":"Ebrahimi","year":"2020","journal-title":"Findings of the Association for Computational Linguistics: EMNLP 2020"},{"key":"ref37","first-page":"3770","article-title":"Self-attention networks can process bounded hierarchical languages","volume-title":"Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)","author":"Yao","year":"2021"},{"key":"ref38","volume-title":"Unveiling transformers with lego: a synthetic reasoning task","author":"Zhang","year":"2022"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.52202\/079017-2001"},{"key":"ref40","author":"Gao","year":"2023","journal-title":"In-context learning for attention scheme: from single softmax regression to multiple softmax regression via a tensor trick"},{"key":"ref41","article-title":"In-context deep learning via transformer models","volume-title":"International Conference on Machine Learning","author":"Wu","year":"2025"},{"key":"ref42","article-title":"In-context learning as conditioned associative memory retrieval","volume-title":"Forty-second International Conference on Machine Learning","author":"Wu","year":"2025"},{"key":"ref43","author":"Hu","year":"2025","journal-title":"In-context algorithm emulation in fixed-weight transformers"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2755"},{"key":"ref45","article-title":"How to capture higher-order correlations? generalizing matrix softmax attention to kronecker computation","volume-title":"The Twelfth International Conference on Learning Representations","year":"2024"},{"key":"ref46","article-title":"The fine-grained complexity of gradient computation for training large language models","volume-title":"The Thirty-eighth Annual Conference on Neural Information Processing Systems","year":"2024"},{"key":"ref47","year":"2025","journal-title":"Fast rope attention: Combining the polynomial method and fast fourier transform"},{"key":"ref48","year":"2025","journal-title":"Only large weights (and not skip connections) can prevent the perils of rank collapse"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1287\/moor.25.3.365.12212"},{"key":"ref50","volume-title":"A direct formulation for sparse pca using semidefinite programming","author":"d\u2019Aspremont","year":"2006"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1214\/08-aos664"},{"key":"ref52","doi-asserted-by":"crossref","DOI":"10.1137\/17M1126680","volume-title":"Robust estimators in high dimensions without the computational intractability","author":"Diakonikolas","year":"2019"},{"key":"ref53","volume-title":"Quantum entropy scoring for fast robust mean estimation and improved outlier detection","author":"Dong","year":"2019"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1145\/3313276.3316303"},{"key":"ref55","volume-title":"Robust sub-gaussian principal component analysis and width-independent schatten packing","author":"Jambulapati","year":"2020"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/FOCS46700.2020.00089"},{"key":"ref57","volume-title":"A faster small treewidth sdp solver","author":"Gu","year":"2022"},{"key":"ref58","volume-title":"Speeding up optimizations via data structures: Faster search, sample and maintenance","author":"Zhang","year":"2022"},{"key":"ref59","author":"Song","year":"2023","journal-title":"Streaming semidefinite programs: o(\\{sqrt{n}) passes, small space and fast runtime"},{"key":"ref60","article-title":"Faster algorithms for structured john ellipsoid computation","author":"Cao","year":"2025","journal-title":"NeurIPS"},{"key":"ref61","first-page":"2140","article-title":"Solving empirical risk minimization in the current matrix multiplication time","volume-title":"Conference on Learning Theory","author":"Lee","year":"2019"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611975994.16"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1145\/3357713.3384284"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1145\/3357713.3384309"},{"key":"ref65","article-title":"Faster dynamic matrix inverse for faster lps","author":"Jiang","year":"2021","journal-title":"STOC"},{"key":"ref66","article-title":"Oblivious sketching-based central path method for solving linear programming problems","author":"Song","year":"2021","journal-title":"ICML"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1137\/1.9781611976496.1"},{"key":"ref68","volume-title":"Solving sdp faster: A robust ipm framework and efficient implementation","author":"Huang","year":"2021"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1145\/1150402.1150429"},{"key":"ref70","author":"Platt","year":"1998","journal-title":"Making large-scale support vector machine learning practical"},{"key":"ref71","author":"John","year":"1998","journal-title":"Sequential minimal optimization: A fast algorithm for training support vector machines"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1145\/1961189.1961199"},{"key":"ref73","first-page":"143","article-title":"Svmtorch: Support vector machines for large-scale regression problems","volume":"1","author":"Collobert","year":"2001","journal-title":"Journal of machine learning research"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2096"},{"key":"ref75","article-title":"A primal-dual framework for transformers and neural networks","volume-title":"The Eleventh International Conference on Learning Representations","author":"Nguyen","year":"2022"},{"key":"ref76","first-page":"1299","article-title":"Sketching for kronecker product regression and p-splines","volume-title":"International Conference on Artificial Intelligence and Statistics","author":"Diao","year":"2018"},{"key":"ref77","article-title":"Optimal sketching for kronecker product regression and low rank approximation","volume":"32","author":"Diao","year":"2019","journal-title":"Advances in neural information processing systems"},{"key":"ref78","volume-title":"Speeding up optimizations via data structures: Faster search, sample and maintenance","author":"Zhang","year":"2022"},{"key":"ref79","volume-title":"A fast optimization view: Reformulating single layer attention in 11 m based on tensor and svm trick, and solving it in matrix multiplication time","author":"Gao","year":"2023"}],"event":{"name":"2025 IEEE International Conference on Big Data (BigData)","location":"Macau, China","start":{"date-parts":[[2025,12,8]]},"end":{"date-parts":[[2025,12,11]]}},"container-title":["2025 IEEE International Conference on Big Data (BigData)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11400704\/11400712\/11402021.pdf?arnumber=11402021","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,7]],"date-time":"2026-03-07T06:55:32Z","timestamp":1772866532000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11402021\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,8]]},"references-count":79,"URL":"https:\/\/doi.org\/10.1109\/bigdata66926.2025.11402021","relation":{},"subject":[],"published":{"date-parts":[[2025,12,8]]}}}