{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T07:44:37Z","timestamp":1740123877308,"version":"3.37.3"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"1","license":[{"start":{"date-parts":[[2021,8,6]],"date-time":"2021-08-06T00:00:00Z","timestamp":1628208000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,8,6]],"date-time":"2021-08-06T00:00:00Z","timestamp":1628208000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["CCF-1910488"],"award-info":[{"award-number":["CCF-1910488"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2022,2]]},"DOI":"10.1007\/s10766-021-00722-1","type":"journal-article","created":{"date-parts":[[2021,8,6]],"date-time":"2021-08-06T03:51:15Z","timestamp":1628221875000},"page":"65-88","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Enhancing the Effectiveness of Inlining in Automatic Parallelization"],"prefix":"10.1007","volume":"50","author":[{"given":"Jichi","family":"Guo","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qing","family":"Yi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3942-3823","authenticated-orcid":false,"given":"Kleanthis","family":"Psarris","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,8,6]]},"reference":[{"key":"722_CR1","volume-title":"Optimizing Compilers for Modern Architec- tures","author":"R Allen","year":"2001","unstructured":"Allen, R., Kennedy, K.: Optimizing Compilers for Modern Architec- tures. Morgan Kaufmann, San Francisco (2001)"},{"issue":"7","key":"722_CR2","doi-asserted-by":"publisher","first-page":"52","DOI":"10.1145\/351403.351416","volume":"35","author":"M Arnold","year":"2000","unstructured":"Arnold, M., Fink, S., Sarkar, V., Sweeney, P.F.: A comparative study of static and profile-based heuristics for inlining. SIGPLAN Not. 35(7), 52\u201364 (2000)","journal-title":"SIGPLAN Not."},{"key":"722_CR3","doi-asserted-by":"crossref","unstructured":"Ashley, J. M.: The effectiveness of flow analysis for inlining. In: In Proceedings of the 1997 ACM SIGPLAN International Conference on Functional Programming, pp 99\u2013111 (1997).","DOI":"10.1145\/258949.258959"},{"issue":"5","key":"722_CR4","doi-asserted-by":"publisher","first-page":"134","DOI":"10.1145\/258916.258928","volume":"32","author":"A Ayers","year":"1997","unstructured":"Ayers, A., Schooler, R., Gottlieb, R.: Aggressive inlining. SIGPLAN Not. 32(5), 134\u2013145 (1997)","journal-title":"SIGPLAN Not."},{"issue":"7","key":"722_CR5","doi-asserted-by":"publisher","first-page":"41","DOI":"10.1145\/74818.74822","volume":"24","author":"V Balasundaram","year":"1989","unstructured":"Balasundaram, V., Kennedy, K.: A technique for summarizing data access and its use in parallelism enhancing transformations. SIGPLAN Not. 24(7), 41\u201353 (1989)","journal-title":"SIGPLAN Not."},{"key":"722_CR6","doi-asserted-by":"publisher","first-page":"643","DOI":"10.1109\/71.180621","volume":"3","author":"W Blume","year":"1992","unstructured":"Blume, W., Eigenmann, R.: Performance analysis of parallelizing compilers on the perfect benchmarks programs. IEEE Trans. Parallel Distrib. Syst. 3, 643\u2013656 (1992)","journal-title":"IEEE Trans. Parallel Distrib. Syst."},{"key":"722_CR7","doi-asserted-by":"crossref","unstructured":"Blume, W., Eigenmann, R., Faigin, K., Grout, J., Hoeflinger, J., Padua, D., Petersen, P., Pottenger, W., Rauchwerger, L., Tu, P., Weatherford, S.: Polaris: Improving the effectiveness of parallelizing compilers. In: In Languages and Compilers for Parallel Computing, pp. 141\u2013154. Springer-Verlag (1994).","DOI":"10.1007\/BFb0025876"},{"key":"722_CR8","doi-asserted-by":"crossref","unstructured":"Bondhugula, U., Hartono, A., Ramanujam, J.: A practical automatic polyhedral parallelizer and locality optimizer. In: In PLDI 08: Proceedings of the ACM SIGPLAN 2008 Conference on Programming Language Design and Implementation (2008).","DOI":"10.1145\/1375581.1375595"},{"key":"722_CR9","doi-asserted-by":"crossref","unstructured":"Calder, B., Grunwald, D.: Reducing indirect function call overhead in c++ programs. In: POPL \u201994: Proceedings of the 21st ACM SIGPLAN-SIGACT Symposium on Principles of Programming Languages, pp. 397\u2013408, ACM, New York, NY, USA, (1994).","DOI":"10.1145\/174675.177973"},{"key":"722_CR10","unstructured":"Cavazos, J., OBoyle, M. F. P.: Automatic tuning of inlining heuristics. In: In ACM\/IEEE Conference on Supercomputing, p. 14 (2005)."},{"key":"722_CR11","doi-asserted-by":"crossref","unstructured":"Chaki, S., Clarke, E., Groce, A.: Modular verification of software components in c. Trans. Softw. Eng. 1(8), 388\u2013402 (2004).","DOI":"10.1109\/TSE.2004.22"},{"key":"722_CR12","doi-asserted-by":"crossref","unstructured":"Chang, Y.-S., Lee, H.-J., Park, D.-S., Lee, I.-Y.: Interprocedural transformations for extracting maximum parallelism. In: ADVIS \u201902: Proceedings of the Second International Conference on Advances in Information Systems, pp. 415\u2013424, Springer-Verlag, London, UK, (2002).","DOI":"10.1007\/3-540-36077-8_44"},{"issue":"1","key":"722_CR13","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1145\/130616.130619","volume":"1","author":"KD Cooper","year":"1992","unstructured":"Cooper, K.D., Hall, M.W., Torczon, L.: Unexpected side effects of inline substitution: a case study. ACM Lett. Program. Lang. Syst. 1(1), 22\u201332 (1992)","journal-title":"ACM Lett. Program. Lang. Syst."},{"key":"722_CR14","doi-asserted-by":"crossref","unstructured":"Dean, J., Chambers, C.: Towards better inlining decisions using inlining trials. In: In Proceedings of the 1994 ACM Conference on LISP and Functional Programming, pp. 273\u2013282 (1994).","DOI":"10.1145\/182590.182489"},{"key":"722_CR15","doi-asserted-by":"crossref","unstructured":"Flanagan, C., Leino, K. R. M., Lillibridge, M., Nelson, G., Saxe, J. B.,Stata, R.: Extended static checking for java. In: Proceedings of the ACM SIGPLAN 2002 Conference on Programming Language Design and Implementation, PLDI \u201902, pp. 234\u2013245, ACM, New York, NY, USA (2002).","DOI":"10.1145\/512529.512558"},{"issue":"1\u20132","key":"722_CR16","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1016\/S0304-3975(00)00051-7","volume":"248","author":"B Grant","year":"2000","unstructured":"Grant, B., Mock, M., Philipose, M., Chambers, C., Eggers, S.J.: Dyc: an expressive annotation-directed dynamic compiler for c. Theor. Comput. Sci. 248(1\u20132), 147\u2013199 (2000)","journal-title":"Theor. Comput. Sci."},{"key":"722_CR17","unstructured":"Grout, J. R.: Inline expansion for the polaris research compiler. Technical Report, (1995)."},{"issue":"2","key":"722_CR18","first-page":"342","volume":"93","author":"SZ Guyer","year":"2005","unstructured":"Guyer, S.Z., Lin, C.: Broadway: A compiler for exploiting the domain-specific semantics of software libraries. Proc. IEEE Special Issue Prog. Gener. Optim. Adapt. 93(2), 342\u2013357 (2005)","journal-title":"Proc. IEEE Special Issue Prog. Gener. Optim. Adapt."},{"key":"722_CR19","doi-asserted-by":"crossref","unstructured":"Hall, M., Kennedy, K., Kinley, K. S. M.: Interprocedural transformations for parallel code generation. In: In Proceedings of Supercomputing \u201991, pp. 424\u2013434 (1991).","DOI":"10.1145\/125826.126055"},{"key":"722_CR20","doi-asserted-by":"crossref","unstructured":"Hall, M. W., Mellor-Crummey, J. M., Carle, A., Rodrguez, R. G.: Fiat: A framework for interprocedural analysis and transformation. In: In Proceedings of the Sixth Workshop on Languages and Compilers for Parallel Computing, pp. 522\u2013545. Springer-Verlag (1995).","DOI":"10.1007\/3-540-57659-2_30"},{"key":"722_CR21","doi-asserted-by":"crossref","unstructured":"Hank, R. E., Mei, W., Hwu, W., Rau, B. R.: Region-based compilation: An introduction and motivation. In: In Proceedings of the 28th Annual International Symposium on Microarchitecture, pp. 158\u2013168 (1995).","DOI":"10.1109\/MICRO.1995.476823"},{"issue":"2","key":"722_CR22","doi-asserted-by":"publisher","first-page":"185","DOI":"10.1023\/A:1007685003043","volume":"29","author":"JP Hoeflinger","year":"2001","unstructured":"Hoeflinger, J.P., Paek, Y., Yi, K.: Unified interprocedural parallelism detection. Int. J. Parallel Program. 29(2), 185\u2013215 (2001)","journal-title":"Int. J. Parallel Program."},{"key":"722_CR23","doi-asserted-by":"crossref","unstructured":"Jagannathan, S., Wright, A.: Flow-directed inlining. In: In Proceedings of the ACM Conference on Programming Language Design and Implementation, pp. 193\u2013205 (1996).","DOI":"10.1145\/249069.231417"},{"key":"722_CR24","unstructured":"Liu, Y., Zhaoqing, Y. L., Qiao, R., Ching Ju, R. D.: A region- based compilation infrastructure. In: Proceefings of the 7th Workshop on Interaction between Compilers and Computer Architectures, pp. 75\u201384. IEEE Computer Society Press (2003)."},{"key":"722_CR25","doi-asserted-by":"crossref","unstructured":"Psarris, K., Kyriakopoulos, K.: The impact of data dependence analysis on compilation and program parallelization. In: ICS \u201903: Proceedings of the 17th Annual International Conference on Supercomputing, pp. 205\u2013214, ACM, New York, NY, USA (2003).","DOI":"10.1145\/782814.782843"},{"key":"722_CR26","doi-asserted-by":"publisher","unstructured":"Ramaswamy, S., Sapatnekar, S., Banerjee, P.: A framework for exploiting data and functional parallelism on distributed memory multicomputers. IEEE Trans. Parallel Distrib. Syst. 8(11), 1098\u20131116 (1997). https:\/\/doi.org\/10.1109\/71.642945","DOI":"10.1109\/71.642945"},{"key":"722_CR27","doi-asserted-by":"publisher","first-page":"344","DOI":"10.1145\/235543.235545","volume":"14","author":"RH Saavedra","year":"1996","unstructured":"Saavedra, R.H., Smith, A.J.: Analysis of benchmark characteristics and benchmark performance prediction. ACM Trans. Comput. Syst. 14, 344\u2013384 (1996)","journal-title":"ACM Trans. Comput. Syst."},{"issue":"4","key":"722_CR28","first-page":"375","volume":"7","author":"H Seidl","year":"2000","unstructured":"Seidl, H., Steffen, B.: Constraint-based inter-procedural analysis of parallel programs. Nordic J. of Computing 7(4), 375\u2013400 (2000)","journal-title":"Nordic J. of Computing"},{"issue":"3","key":"722_CR29","doi-asserted-by":"publisher","first-page":"135","DOI":"10.1016\/0096-0551(94)90001-9","volume":"20","author":"UN Shenoy","year":"1994","unstructured":"Shenoy, U.N., Srikant, Y.N., Bhatkar, V.P.: An automatic parallelization framework for multicomputers. Comput. Lang. 20(3), 135\u2013150 (1994)","journal-title":"Comput. Lang."},{"key":"722_CR30","unstructured":"Shirako, J., Nagasawa, K., Ishizaka, K., Obata, M., Kasahara, H.: Selective inline expansion for improvement of multi grain parallelism. In: Parallel and Distributed Computing and Networks, pp. 476\u2013482 (2004)."},{"key":"722_CR31","doi-asserted-by":"crossref","unstructured":"Triantafyllis, S., Bridges, M. J., Raman, E., Ottoni, G., August, D. I.: A framework for unrestricted whole-program optimization. In: In ACM SIGPLAN 2006 Conference on Programming Language Design and Implementation, pp. 61\u201371 (2006).","DOI":"10.1145\/1133255.1133989"},{"key":"722_CR32","doi-asserted-by":"crossref","unstructured":"Waddell, O., Dybig, R. K.: Fast and effective procedure inlining. In: SAS \u201997: Proceedings of the 4th International Symposium on Static Analysis, pp. 35\u201352, Springer-Verlag, London, UK (1997).","DOI":"10.1007\/BFb0032732"},{"key":"722_CR33","volume-title":"High Performance Compilers for Parallel Computing","author":"MJ Wolfe","year":"1995","unstructured":"Wolfe, M.J.: High Performance Compilers for Parallel Computing. Addison-Wesley Longman Publishing Co. Inc, Boston, MA, USA (1995)"},{"key":"722_CR34","unstructured":"Wu, P., Midkiff, S. P., Moreira, J. E., Gupta, M.: Improving Java performance through semantic inlining. In: Proceedings of the Ninth SIAM Conference on Parallel Processing for Scientific Computing (1999)."},{"key":"722_CR35","unstructured":"Wu, P., Moreira, J. E., Midkiff, S. P., Gupta, M., Padua, D. A.: Semantic inlining - the compiler support for java in technical computing. In: PPSC (1999)."},{"key":"722_CR36","doi-asserted-by":"crossref","unstructured":"Zhao, P., Amaral, J. N.: To inline or not to inline? enhanced inlining decisions. In: In Workshop on Languages and Compilers for Parallel Computing (LCPC), pp. 405\u2013419 (2003).","DOI":"10.1007\/978-3-540-24644-2_26"},{"issue":"5","key":"722_CR37","doi-asserted-by":"publisher","first-page":"465","DOI":"10.1002\/spe.774","volume":"37","author":"P Zhao","year":"2007","unstructured":"Zhao, P., Amaral, J.N.: Ablego: a function outlining and partial inlining framework. Softw. Pract. Exper. 37(5), 465\u2013491 (2007)","journal-title":"Softw. Pract. Exper."},{"key":"722_CR38","unstructured":"Aleksandar, P., et al.: An optimization-driven incremental inline substitution algorithm for just-in-time compilers.In: 2019 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO). IEEE (2019)."},{"issue":"4","key":"722_CR39","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1016\/j.cl.2013.04.002","volume":"39","author":"C H\u00e4ubl","year":"2013","unstructured":"H\u00e4ubl, C., Wimmer, C., M\u00f6ssenb\u00f6ck, H.: Context-sensitive Trace Inlining for Java. Comput. Lang. Syst. Struct. 39(4), 123\u2013141 (2013). https:\/\/doi.org\/10.1016\/j.cl.2013.04.002","journal-title":"Comput. Lang. Syst. Struct."},{"key":"722_CR40","doi-asserted-by":"publisher","unstructured":"Cammarota, R., Nicolau, A., Veidenbaum, A.V., Kejariwal, A., Donato, D., Madhugiri, M.: On the determination of inlining vectors for program optimization. In: Proceedings of the 22nd International Conference on Compiler Construction (CC\u201913). Springer-Verlag, Berlin, Heidelberg, pp. 164\u2013183. https:\/\/doi.org\/10.1007\/978-3-642-37051-9_9","DOI":"10.1007\/978-3-642-37051-9_9"},{"key":"722_CR41","doi-asserted-by":"publisher","unstructured":"Simon, D., Cavazos, J., Wimmer, C., Kulkarni, S.: Automatic construction of inlining heuristics using machine learning. In: Proceedings of the 2013 IEEE\/ACM International Symposium on Code Generation and Optimization (CGO \u201913). IEEE Computer Society, Washington, DC, pp. 1\u201312. https:\/\/doi.org\/10.1109\/CGO.2013.6495004","DOI":"10.1109\/CGO.2013.6495004"},{"key":"722_CR42","doi-asserted-by":"publisher","unstructured":"Stucki, N., Biboudis, A., Doeraene, S., Odersky, M.: Semantics-preserving inlining for metaprogramming. In: Proceedings of the 11th ACM SIGPLAN International Symposium on Scala (SCALA 2020). ACM, New York, NY, USA, pp. 14\u201324. https:\/\/doi.org\/10.1145\/3426426.3428486","DOI":"10.1145\/3426426.3428486"},{"key":"722_CR43","doi-asserted-by":"publisher","unstructured":"Norris, H. B., Sadayappan, P.: Annotation-based empirical performance tuning using Orio, In: 2009 IEEE International Symposium on Parallel & Distributed Processing, Rome, Italy, pp. 1-11 (2009). https:\/\/doi.org\/10.1109\/IPDPS.2009.5161004","DOI":"10.1109\/IPDPS.2009.5161004"},{"key":"722_CR44","doi-asserted-by":"publisher","unstructured":"Papadopoulos, I., Thomas, N., Fidel, A., Amato, N. M., Rauchwerger, L.: STAPL-RTS: An application driven runtime system. In: Proceedings of the 29th ACM on International Conference on Supercomputing (ICS '15). ACM, New York, NY, USA, pp. 425\u2013434. https:\/\/doi.org\/10.1145\/2751205.2751233","DOI":"10.1145\/2751205.2751233"},{"key":"722_CR45","doi-asserted-by":"publisher","unstructured":"Yi, Q., Wang, Q., Cui, H.: Specializing compiler optimizations through programmable composition for dense matrix computations. In: Proceedings of the 47th Annual IEEE\/ACM International Symposium on Microarchitecture (MICRO-47). IEEE Computer Society, USA, pp. 596\u2013608.  https:\/\/doi.org\/10.1109\/MICRO.2014.14","DOI":"10.1109\/MICRO.2014.14"},{"key":"722_CR46","unstructured":"Nguyen, T., Cicotti, P., Bylaska, E., Quinlan, D., Baden, S.B.: Bamboo: translating MPI applications to a latency-tolerant, data-driven form. In: Proceedings of the International Conference on High Performance Computing, Networking, Storage and Analysis (SC '12). IEEE Computer Society Press, Washington, DC, USA, Article 39, pp. 1\u201311."},{"key":"722_CR47","doi-asserted-by":"publisher","unstructured":"Ali, K., Lhot\u00e1k, O.: Application-only call graph construction. In: Proceedings of the European Conference on Object-Oriented Programming (ECOOP 2012). Lecture Notes in Computer Science, Vol 7313. Springer, Berlin, Heidelberg. https:\/\/doi.org\/10.1007\/978-3-642-31057-7_30","DOI":"10.1007\/978-3-642-31057-7_30"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-021-00722-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10766-021-00722-1\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-021-00722-1.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,1,29]],"date-time":"2022-01-29T14:09:13Z","timestamp":1643465353000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10766-021-00722-1"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,8,6]]},"references-count":47,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2022,2]]}},"alternative-id":["722"],"URL":"https:\/\/doi.org\/10.1007\/s10766-021-00722-1","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"type":"print","value":"0885-7458"},{"type":"electronic","value":"1573-7640"}],"subject":[],"published":{"date-parts":[[2021,8,6]]},"assertion":[{"value":"24 April 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 June 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 August 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}