{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2023,3,6]],"date-time":"2023-03-06T05:37:47Z","timestamp":1678081067124},"reference-count":24,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2010,3,23]],"date-time":"2010-03-23T00:00:00Z","timestamp":1269302400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["J Sign Process Syst"],"published-print":{"date-parts":[[2011,3]]},"DOI":"10.1007\/s11265-010-0465-x","type":"journal-article","created":{"date-parts":[[2010,3,22]],"date-time":"2010-03-22T11:35:13Z","timestamp":1269257713000},"page":"325-340","source":"Crossref","is-referenced-by-count":1,"title":["Loop Distribution and Fusion with Timing and Code Size Optimization"],"prefix":"10.1007","volume":"62","author":[{"given":"Meilin","family":"Liu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Edwin H. -M.","family":"Sha","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qingfeng","family":"Zhuge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yi","family":"He","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meikang","family":"Qiu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2010,3,23]]},"reference":[{"key":"465_CR1","doi-asserted-by":"crossref","first-page":"340","DOI":"10.1109\/HPCA.2005.27","volume-title":"HPCA \u201905: Proceedings of the 11th international symposium on high-performance computer architecture","author":"D Chandra","year":"2005","unstructured":"Chandra, D., Guo, F., Kim, S., & Solihin, Y. (2005). Predicting inter-thread cache contention on a chip multi-processor architecture. In HPCA \u201905: Proceedings of the 11th international symposium on high-performance computer architecture (pp. 340\u2013351). Washington, DC: IEEE Computer Society."},{"key":"465_CR2","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1109\/71.577265","volume":"8","author":"N Manjikian","year":"1997","unstructured":"Manjikian, N., & Abdelrahman, T. S. (1997). Fusion of loops for parallelism and locality. IEEE Transactions on Parallel and Distributed System, 8, 193\u2013209.","journal-title":"IEEE Transactions on Parallel and Distributed System"},{"issue":"4","key":"465_CR3","doi-asserted-by":"crossref","first-page":"424","DOI":"10.1145\/233561.233564","volume":"18","author":"KS McKinley","year":"1996","unstructured":"McKinley, K.S., Carr, S., & Tseng, C.-W. (1996). Improving data locality with loop transformations. ACM Transactions on Programming Languages and Systems (TOPLAS), 18(4), 424\u2013453.","journal-title":"ACM Transactions on Programming Languages and Systems (TOPLAS)"},{"key":"465_CR4","doi-asserted-by":"crossref","unstructured":"Kennedy, K., & Mckinley, K. S. (1992). Optimizing for parallelism and data locality. In Proc. of the 6th conference on supercomputing (pp. 323\u2013334).","DOI":"10.1145\/143369.143427"},{"issue":"4","key":"465_CR5","doi-asserted-by":"crossref","first-page":"452","DOI":"10.1109\/71.97902","volume":"2","author":"M Wolf","year":"1991","unstructured":"Wolf, M. & Lam, M. (1991). A loop transformation theory and an algorithm to maximize parallelism. IEEE Transactions on Parallel and Distributed Systems, 2(4), 452\u2013471.","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"465_CR6","doi-asserted-by":"crossref","first-page":"1242","DOI":"10.1109\/TC.2005.167","volume":"54","author":"A Darte","year":"2005","unstructured":"Darte, A., Schreiber, R., & Villard, G. (2005). Lattice-based memory allocation. IEEE Transactions on Computers, 54, 1242\u20131257.","journal-title":"IEEE Transactions on Computers"},{"key":"465_CR7","doi-asserted-by":"crossref","first-page":"78","DOI":"10.1145\/232973.232983","volume-title":"ISCA \u201996: Proceedings of the 23rd annual international symposium on computer architecture","author":"D Burger","year":"1996","unstructured":"Burger, D., Goodman, J. R., & K\u00e4gi, A. (1996). Memory bandwidth limitations of future microprocessors. In ISCA \u201996: Proceedings of the 23rd annual international symposium on computer architecture (pp. 78\u201389). New York: ACM."},{"key":"465_CR8","first-page":"272","volume-title":"DSD \u201904: Proceedings of the digital system design, EUROMICRO systems","author":"Q Hu","year":"2004","unstructured":"Hu, Q., Palkovic, M., & Kjeldsberg, P. G. (2004). Memory requirement optimization with loop fusion and loop shifting. In DSD \u201904: Proceedings of the digital system design, EUROMICRO systems (pp. 272\u2013278). Washington, DC: IEEE Computer Society."},{"issue":"2","key":"465_CR9","doi-asserted-by":"crossref","first-page":"149","DOI":"10.1145\/375977.375978","volume":"6","author":"PR Panda","year":"2001","unstructured":"Panda, P. R., Catthoor, F., Dutt, N. D., Danckaert, K., Brockmeyer, E., Kulkarni, C., et al. (2001). Data and memory optimization techniques for embedded systems. ACM Transactions on Design Automation of Electronic Systems (TODAES), 6(2), 149\u2013206.","journal-title":"ACM Transactions on Design Automation of Electronic Systems (TODAES)"},{"key":"465_CR10","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4757-2849-1","volume-title":"Custom memory management methodology: Exploration of memory organisation for embedded multimedia system design","author":"F Catthoor","year":"1998","unstructured":"Catthoor, F., de Greef, E., & Suytack, S. (1998). Custom memory management methodology: Exploration of memory organisation for embedded multimedia system design. Norwell: Kluwer Academic."},{"key":"465_CR11","doi-asserted-by":"crossref","unstructured":"Wang, Z., Hu, S., & Sha, E. H.-M. (2003). Register aware scheduling for distributed cache clustered architecture. In Proc. IEEE\/ACM 2003 ASP design automation conference, Kitakyusyu, Japan.","DOI":"10.1145\/1119772.1119787"},{"issue":"6","key":"465_CR12","doi-asserted-by":"crossref","first-page":"604","DOI":"10.1109\/71.862210","volume":"11","author":"F Chen","year":"2000","unstructured":"Chen, F., O\u2019Neil, T. W., & Sha, E. H.-M. (2000). Optimizing overall loop schedules using prefetching and partitioning. IEEE Transactions on Parallel and Distributed Systems, 11(6), 604\u2013614.","journal-title":"IEEE Transactions on Parallel and Distributed Systems"},{"key":"465_CR13","volume-title":"High performance compilers for parallel computing","author":"M Wolfe","year":"1996","unstructured":"Wolfe, M. (1996). High performance compilers for parallel computing. Reading: Addison-Wesley."},{"key":"465_CR14","volume-title":"Optimizing compilers for modern architectures: A dependence-based approach","author":"R Allen","year":"2001","unstructured":"Allen, R., & Kennedy, K. (2001). Optimizing compilers for modern architectures: A dependence-based approach. San Francisco: Morgan Kaufmann."},{"key":"465_CR15","doi-asserted-by":"crossref","unstructured":"Kennedy, K., & Mckinley, K. S. (1990). Loop distribution with arbitrary control flow. In Proc. of the 1990 conference on supercomputing (pp. 407\u2013416).","DOI":"10.1109\/SUPERC.1990.130048"},{"key":"465_CR16","unstructured":"Kennedy, K., & Mckinley, K. S. (1993). Maximizing loop parallelism and improving data locality via loop fusion and distribution. In Languages and Compilers for Parallel Computing, Lecture Notes in Computer Science, 768, pp. 301\u2013320."},{"issue":"1","key":"465_CR17","first-page":"9","volume":"10","author":"EH-M Sha","year":"2003","unstructured":"Sha, E. H.-M., O\u2019Neil, T. W., & Passos, N. L. (2003). Efficient polynomial-time nested loop fusion with full parallelism. International Journal of Computers and Their Applications, 10(1), 9\u201324.","journal-title":"International Journal of Computers and Their Applications"},{"key":"465_CR18","unstructured":"Abdelrahman, T. S., Sawaya, R. (2000). Increasing perfect nests in scientific programs. In Proc. of international conference on parallel and distributed computing and systems (pp. 279\u2013285). Las Vegas, NV."},{"issue":"3","key":"465_CR19","doi-asserted-by":"crossref","first-page":"219","DOI":"10.1023\/B:SUPE.0000011386.69245.f5","volume":"27","author":"Q Yi","year":"2004","unstructured":"Yi, Q., Kennedy, K., & Adve, V. (2004). Transforming complex loop nests for locality. The Journal Of Supercomputing, 27(3), 219\u2013264.","journal-title":"The Journal Of Supercomputing"},{"issue":"2","key":"465_CR20","doi-asserted-by":"crossref","first-page":"237","DOI":"10.1177\/1094342004038956","volume":"18","author":"Q Yi","year":"2004","unstructured":"Yi, Q., & Kennedy, K. (2004). Improving memory hierarchy performance through combined loop interchange and multi-level fusion. International Journal of High Performance Computing Applications, 18(2), 237\u2013253.","journal-title":"International Journal of High Performance Computing Applications"},{"key":"465_CR21","doi-asserted-by":"crossref","unstructured":"Liu, M., Zhuge, Q., Shao, Z., & Sha, E. H.-M. (2004). General loop fusion technique for nested loops considering timing and code size. In Proc. ACM\/IEEE international conference on compilers, architectures, and synthesis for embedded systems (CASES 2004) (pp. 190\u2013201).","DOI":"10.1145\/1023833.1023860"},{"key":"465_CR22","doi-asserted-by":"crossref","unstructured":"Darte, A. (1999). On the complexity of loop fusion. In International conference on parallel architectures and compilation techniques (pp. 149\u2013157).","DOI":"10.1109\/PACT.1999.807510"},{"key":"465_CR23","doi-asserted-by":"crossref","unstructured":"Verdoolaege, S., Bruynooghe, M., & Catthoor, F. (2003). Multi-dimensional incremental loop fusion for data locality. In Proc. of the application-specific systems, architectures, and processors (pp. 14\u201324).","DOI":"10.1109\/ASAP.2003.1212826"},{"key":"465_CR24","volume-title":"Engineering a compiler","author":"KD Cooper","year":"2008","unstructured":"Cooper, K. D., Torczon, L. (2008). Engineering a compiler. San Francisco: Morgan Kaufmann."}],"container-title":["Journal of Signal Processing Systems"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-010-0465-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11265-010-0465-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11265-010-0465-x","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,6,1]],"date-time":"2019-06-01T08:19:33Z","timestamp":1559377173000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11265-010-0465-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2010,3,23]]},"references-count":24,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2011,3]]}},"alternative-id":["465"],"URL":"https:\/\/doi.org\/10.1007\/s11265-010-0465-x","relation":{},"ISSN":["1939-8018","1939-8115"],"issn-type":[{"value":"1939-8018","type":"print"},{"value":"1939-8115","type":"electronic"}],"subject":[],"published":{"date-parts":[[2010,3,23]]}}}