{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,21]],"date-time":"2026-03-21T19:23:28Z","timestamp":1774121008952,"version":"3.50.1"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2012,8,9]],"date-time":"2012-08-09T00:00:00Z","timestamp":1344470400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2013,2]]},"DOI":"10.1007\/s10766-012-0213-x","type":"journal-article","created":{"date-parts":[[2012,8,8]],"date-time":"2012-08-08T12:13:03Z","timestamp":1344427983000},"page":"137-159","source":"Crossref","is-referenced-by-count":17,"title":["B-Queue: Efficient and Practical Queuing for Fast Core-to-Core Communication"],"prefix":"10.1007","volume":"41","author":[{"given":"Junchang","family":"Wang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xinan","family":"Tang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bei","family":"Hua","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,8,9]]},"reference":[{"key":"213_CR1","doi-asserted-by":"crossref","unstructured":"Allen, M.D., Sridharan, S., Sohi, G.S.: Serialization sets: a dynamic dependence-based parallel execution model. In: Proceedings of the 14th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPoPP \u201909, pp. 85\u201396 (2009)","DOI":"10.1145\/1504176.1504190"},{"key":"213_CR2","doi-asserted-by":"crossref","unstructured":"Dobrescu, M., Egi, N., Argyraki, K., Chun, B.G., Fall, K., Iannaccone, G., Knies, A., Manesh, M., Ratnasamy, S.: Routebricks: exploiting parallelism to scale software routers. In: Proceedings of the ACM SIGOPS 22nd Symposium on Operating Systems Principles, SOSP \u201909, pp. 15\u201328 (2009)","DOI":"10.1145\/1629575.1629578"},{"key":"213_CR3","unstructured":"Gcc manual http:\/\/gcc.gnu.org\/onlinedocs\/gcc-4.5.4\/gcc\/"},{"key":"213_CR4","doi-asserted-by":"crossref","unstructured":"Giacomoni, J., Bennett, J.K., Carzaniga, A., Sicker, D.C., Vachharajani, M., Wolf, A.L.: Frame shared memory: line-rate networking on commodity hardware. In: Proceedings of the 3rd ACM\/IEEE Symposium on Architecture for Networking and Communications Systems, ANCS \u201907, pp. 27\u201336 (2007)","DOI":"10.1145\/1323548.1323553"},{"key":"213_CR5","doi-asserted-by":"crossref","unstructured":"Giacomoni, J., Moseley, T., Vachharajani, M.: Fastforward for efficient pipeline parallelism: a cache-optimized concurrent lock-free queue. In: Proceedings of the 13th ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, PPoPP \u201908, pp. 43\u201352 (2008)","DOI":"10.1145\/1345206.1345215"},{"key":"213_CR6","doi-asserted-by":"crossref","unstructured":"Han, S., Jang, K., Park, K., Moon, S.: (2010) Packetshader: a gpu-accelerated software router. In: SIGCOMM\u201910, pp. 195\u2013206","DOI":"10.1145\/1851182.1851207"},{"key":"213_CR7","volume-title":"The Art of Multiprocessor Programming","author":"M. Herlihy","year":"2008","unstructured":"Herlihy M., Shavit N.: The Art of Multiprocessor Programming. Morgan Kaufmann, Massachusetts (2008)"},{"key":"213_CR8","doi-asserted-by":"crossref","first-page":"463","DOI":"10.1145\/78969.78972","volume":"12","author":"M.P. Herlihy","year":"1990","unstructured":"Herlihy M.P., Wing J.M.: Linearizability: a correctness condition for concurrent objects. ACM Trans. Program. Lang. Syst. 12, 463\u2013492 (1990)","journal-title":"ACM Trans. Program. Lang. Syst."},{"key":"213_CR9","unstructured":"Intel Corporation: Intel\u00ae 64 and IA-32 Architectures Software Developer\u2019s Manual. vol 2B. 253669-033US (2009)"},{"key":"213_CR10","unstructured":"Jablin, T.B., Zhang, Y., Jablin, J.A., et\u00a0al.: Liberty queues for epic architectures. In: Proceedings of the Eigth Workshop on Explicitly Parallel Instruction Computer Architectures and Compiler Technology, EPIC\u201910 (2010)"},{"key":"213_CR11","doi-asserted-by":"crossref","unstructured":"Keltcher, C.N., McGrath, K.J.A.: The amd opteron processor for multiprocessor servers. IEEE Micro (2003)","DOI":"10.1109\/MM.2003.1196116"},{"key":"213_CR12","doi-asserted-by":"crossref","unstructured":"Kogan, A., Petrank, E.: Wait-free queues with multiple enqueuers and dequeuers. In: Proceedings of the 16th ACM Symposium on Principles and Practice of Parallel Programming, PPoPP \u201911, pp. 223\u2013234. New York, NY, USA","DOI":"10.1145\/2038037.1941585"},{"key":"213_CR13","doi-asserted-by":"crossref","unstructured":"Kumar, S., Hughes, C.J., Nguyen, A.: Carbon: architectural support for fine-grained parallelism on chip multiprocessors. In: Proceedings of the 34th Annual International Symposium on Computer Architecture, ISCA \u201907, pp. 162\u2013173 (2007)","DOI":"10.1145\/1250662.1250683"},{"key":"213_CR14","doi-asserted-by":"crossref","first-page":"323","DOI":"10.1007\/s00446-007-0050-0","volume":"20","author":"E. Ladan-Mozes","year":"2008","unstructured":"Ladan-Mozes E., Shavit N.: An optimistic approach to lock-free fifo queues. Distrib. Comput. 20, 323\u2013341 (2008)","journal-title":"Distrib. Comput."},{"key":"213_CR15","doi-asserted-by":"crossref","unstructured":"Lamport, L.: Specifying concurrent program modules. ACM Trans. Program. Lang. Syst. 190\u2013222 (1983)","DOI":"10.1145\/69624.357207"},{"key":"213_CR16","doi-asserted-by":"crossref","unstructured":"Lee, P., Bu, T., Chandranmenon, G.: A lock-free, cache-efficient multi-core synchronization mechanism for line-rate network traffic monitoring. In: 2010 IEEE International Symposium on Parallel and Distributed Processing, IPDPS \u201910, pp. 1\u201312 (2010)","DOI":"10.1109\/IPDPS.2010.5470368"},{"key":"213_CR17","doi-asserted-by":"crossref","unstructured":"Lee, S., Tiwari, D., Solihin, Y., Tuck, J.: Haqu: hardware accelerated queueing for fine-grained threading on a chip multiprocessor. In: The 17th IEEE International Symposium on High Performance Computer Architecture, HPCA-17 (2011)","DOI":"10.1109\/HPCA.2011.5749720"},{"key":"213_CR18","unstructured":"Libnids http:\/\/libnids.sourceforge.net\/"},{"key":"213_CR19","doi-asserted-by":"crossref","first-page":"687","DOI":"10.1007\/s11390-009-9251-2","volume":"24","author":"H. Luan","year":"2009","unstructured":"Luan H., Du X.Y., Wang S.: Prefetching j+-tree: a cache-optimized main memory database index structure. J. Comput. Sci. Technol. 24, 687\u2013707 (2009)","journal-title":"J. Comput. Sci. Technol."},{"key":"213_CR20","doi-asserted-by":"crossref","unstructured":"Michael, M.M., Scott, M.L.: Simple, fast, and practical non-blocking and blocking concurrent queue algorithms. In: Proceedings of the Fifteenth Annual ACM Symposium on Principles of Distributed Computing, PODC \u201996, pp. 267\u2013275","DOI":"10.1145\/248052.248106"},{"key":"213_CR21","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1006\/jpdc.1998.1446","volume":"51","author":"M.M. Michael","year":"1998","unstructured":"Michael M.M., Scott M.L.: Non-blocking algorithms and preemption-safe locking on multiprogrammed shared memory multiprocessors. J. Parallel Distrib. Comput. 51, 1\u201326 (1998)","journal-title":"J. Parallel Distrib. Comput."},{"key":"213_CR22","doi-asserted-by":"crossref","unstructured":"Moir, M., Nussbaum, D., Shalev, O., Shavit, N.: Using elimination to implement scalable and lock-free fifo queues. In: Proceedings of the Seventeenth Annual ACM Symposium on Parallelism in Algorithms and Architectures, SPAA \u201905, pp. 253\u2013262 (2005)","DOI":"10.1145\/1073970.1074013"},{"key":"213_CR23","doi-asserted-by":"crossref","unstructured":"Molka, D., Hackenberg, D., Schone, R., Muller, M.S.: Memory performance and cache coherency effects on an intel nehalem multiprocessor system. In: Proceedings of the 2009 18th International Conference on Parallel Architectures and Compilation Techniques (2009)","DOI":"10.1109\/PACT.2009.22"},{"key":"213_CR24","doi-asserted-by":"crossref","unstructured":"Navarro, A., Asenjo, R., Tabik, S., Cascaval, C.: Analytical modeling of pipeline parallelism. In: 18th International Conference on Parallel Architectures and Compilation Techniques, PACT \u201909, pp. 281\u2013290 (2009)","DOI":"10.1109\/PACT.2009.28"},{"key":"213_CR25","unstructured":"Oprofile http:\/\/oprofile.sourceforge.net\/"},{"key":"213_CR26","doi-asserted-by":"crossref","unstructured":"Ottoni, G., Rangan, R., Stoler, A., August, D.I.: Automatic thread extraction with decoupled software pipelining. In: In Proceedings of the 38th IEEE\/ACM International Symposium on Microarchitecture, pp. 105\u2013118. IEEE Computer Society (2005)","DOI":"10.1109\/MICRO.2005.13"},{"key":"213_CR27","doi-asserted-by":"crossref","unstructured":"Preud\u2019homme, T., Sopena, J., Thomas, G., Folliot, B.: Batchqueue: Fast and memory-thrifty core to core communication. In: 2010 22nd International Symposium on Computer Architecture and High Performance Computing, SBAC-PAD \u201910, pp. 215\u2013222 (2010)","DOI":"10.1109\/SBAC-PAD.2010.34"},{"key":"213_CR28","doi-asserted-by":"crossref","unstructured":"Price, G.D., Giacomoni, J., Vachharajani, M.: Visualizing potential parallelism in sequential programs. In: Proceedings of the 17th International Conference on Parallel Architectures and Compilation Techniques, PACT \u201908, pp. 82\u201390 (2008)","DOI":"10.1145\/1454115.1454129"},{"key":"213_CR29","doi-asserted-by":"crossref","unstructured":"Sanchez, D., Lo, D., Yoo, R.M., Sugerman, J., Kozyrakis, C.: Dynamic fine-grain scheduling of pipeline parallelism. In: International Conference on Parallel Architectures and Compilation Techniques, PACT \u201911, pp. 22\u201332 (2011)","DOI":"10.1109\/PACT.2011.9"},{"key":"213_CR30","doi-asserted-by":"crossref","unstructured":"Thies, W., Chandrasekhar, V., Amarasinghe, S.: A practical approach to exploiting coarse-grained pipeline parallelism in c programs. In: 40th IEEE\/ACM International Symposium on Microarchitecture, MICRO \u201907, pp. 356\u2013369 (2007)","DOI":"10.1109\/MICRO.2007.4408268"},{"key":"213_CR31","unstructured":"Torquati, M.: Single-producer\/single-consumer queues on shared cache multi-core systems. CoRR (2010)"},{"key":"213_CR32","doi-asserted-by":"crossref","unstructured":"Transier, F., Sanders, P.: Engineering basic algorithms of an in-memory text search engine. ACM Trans. Inf. Syst. pp. 2:1\u20132:37 (2010)","DOI":"10.1145\/1877766.1877768"},{"key":"213_CR33","doi-asserted-by":"crossref","unstructured":"Tsigas, P., Zhang, Y.: A simple, fast and scalable non-blocking concurrent fifo queue for shared memory multiprocessor systems. In: Proceedings of the Thirteenth Annual ACM symposium on Parallel Algorithms and Architectures, SPAA \u201901, pp. 134\u2013143 (2001)","DOI":"10.1145\/378580.378611"},{"key":"213_CR34","doi-asserted-by":"crossref","unstructured":"Upadhyaya, G., Pai, V.S., Midkiff, S.P.: Expressing and exploiting concurrency in networked applications in aspen. In: In Proceedings of ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 13\u201323 (2007)","DOI":"10.1145\/1229428.1229433"},{"key":"213_CR35","doi-asserted-by":"crossref","unstructured":"Wang, J., Cheng, H., Hua, B., Tang, X.: Practice of parallelizing network applications on multi-core architectures. In: Proceedings of the 23rd International Conference on Supercomputing, ICS \u201909, pp. 204\u2013213 (2009)","DOI":"10.1145\/1542275.1542307"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-012-0213-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-012-0213-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-012-0213-x","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,6]],"date-time":"2025-04-06T17:55:56Z","timestamp":1743962156000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-012-0213-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,8,9]]},"references-count":35,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2013,2]]}},"alternative-id":["213"],"URL":"https:\/\/doi.org\/10.1007\/s10766-012-0213-x","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"value":"0885-7458","type":"print"},{"value":"1573-7640","type":"electronic"}],"subject":[],"published":{"date-parts":[[2012,8,9]]}}}