{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,4,24]],"date-time":"2025-04-24T04:10:24Z","timestamp":1745467824655,"version":"3.40.4"},"reference-count":31,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2012,12,8]],"date-time":"2012-12-08T00:00:00Z","timestamp":1354924800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int J Parallel Prog"],"published-print":{"date-parts":[[2013,12]]},"DOI":"10.1007\/s10766-012-0236-3","type":"journal-article","created":{"date-parts":[[2012,12,7]],"date-time":"2012-12-07T12:20:38Z","timestamp":1354882838000},"page":"855-869","source":"Crossref","is-referenced-by-count":2,"title":["An Infrastructure for Tackling Input-Sensitivity of GPU Program Optimizations"],"prefix":"10.1007","volume":"41","author":[{"given":"Xipeng","family":"Shen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yixun","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Eddy Z.","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Poornima","family":"Bhamidipati","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2012,12,8]]},"reference":[{"key":"236_CR1","doi-asserted-by":"crossref","unstructured":"Arnold, M., Hind, M., Ryder, B.G.: Online feedback-directed optimization of Java. In: Proceedings of ACM Conference on Object-Oriented Programming Systems, Languages and Applications, pp. 111\u2013129 (2002)","DOI":"10.1145\/582431.582432"},{"key":"236_CR2","doi-asserted-by":"crossref","unstructured":"Bartolini, S., Prete, C.A.: A proposal for input-sensitivity analysis of profile-driven optimizations on embedded applications. In: Proceedings of the 2003 workshop on Memory Performance: DEaling with Applications, systems and, architecture (2003)","DOI":"10.1145\/1152923.1024305"},{"key":"236_CR3","doi-asserted-by":"crossref","unstructured":"Baskaran, M.M., Bondhugula, U., Krishnamoorthy, S., et al.: A compiler framework for optimization of affine loop nests for GPGPUs. In: ICS\u201908: Proceedings of the 22nd Annual International Conference on Supercomputing, pp. 225\u2013234 (2008)","DOI":"10.1145\/1375527.1375562"},{"key":"236_CR4","doi-asserted-by":"crossref","unstructured":"Bilmes, J., Asanovic, K., Chin, C.W., Demmel, J.: Optimizing matrix multiply using PHiPAC: A portable, high-performance, ANSI C coding methodology. In: Proceedings of the ACM International Conference on Supercomputing, pp. 340\u2013347 (1997)","DOI":"10.1145\/263580.263662"},{"key":"236_CR5","doi-asserted-by":"crossref","unstructured":"Cooper, K.D., Hall, M.W., Kennedy, K.: Procedure cloning. In: Computer Languages, pp. 96\u2013105 (1992)","DOI":"10.1016\/0096-0551(93)90005-L"},{"key":"236_CR6","doi-asserted-by":"crossref","unstructured":"Diniz, P., Rinard, M.: Dynamic feedback: An effective technique for adaptive computing. In: Proceedings of ACM SIGPLAN Conference on Programming Language Design and Implementation, pp. 71\u201384, Las Vegas, May (1997)","DOI":"10.1145\/258916.258923"},{"issue":"2","key":"236_CR7","doi-asserted-by":"crossref","first-page":"216","DOI":"10.1109\/JPROC.2004.840301","volume":"93","author":"M Frigo","year":"2005","unstructured":"Frigo, M., Johnson, S.G.: The design and implementation of FFTW3. Proc. IEEE 93(2), 216\u2013231 (2005)","journal-title":"Proc. IEEE"},{"key":"236_CR8","doi-asserted-by":"crossref","unstructured":"Fujimoto, N.: Faster matrix-vector multiplication on GeForce 8800GTX. In: Proceedings of the Workshop on Large-Scale Parallel Processing (co-located with IPDPS), pp. 1\u20138 (2008)","DOI":"10.1109\/IPDPS.2008.4536350"},{"key":"236_CR9","doi-asserted-by":"crossref","unstructured":"Fung, W., Sham, I., Yuan, G., Aamodt, T.: Dynamic warp formation and scheduling for efficient gpu control flow. In: MICRO \u201907: Proceedings of the 40th Annual IEEE\/ACM International Symposium on Microarchitecture, pp. 407\u2013420. IEEE Computer Society, Washington, DC, USA (2007)","DOI":"10.1109\/MICRO.2007.30"},{"key":"236_CR10","doi-asserted-by":"crossref","unstructured":"Rudy, G., Khan, M., Hall, M., Chen, C., Jacqueline, C.: A programming language interface to describe transformations and code generation. In: Proceedings of LCPC, Lecture Notes in Computer Science (2010)","DOI":"10.1007\/978-3-642-19595-2_10"},{"key":"236_CR11","doi-asserted-by":"crossref","DOI":"10.1007\/978-0-387-21606-5","volume-title":"The elements of statistical learning","author":"T Hastie","year":"2001","unstructured":"Hastie, T., Tibshirani, R., Friedman, J.: The elements of statistical learning. Springer, Berlin (2001)"},{"issue":"1","key":"236_CR12","doi-asserted-by":"crossref","first-page":"135","DOI":"10.1177\/1094342004041296","volume":"18","author":"EJ Im","year":"2004","unstructured":"Im, E.J., Yelick, Katherine, Vuduc, Richard: Sparsity: Optimization framework for sparse matrix kernels. Int. J. High Perform. Comput. Appl. 18(1), 135\u2013158 (2004)","journal-title":"Int. J. High Perform. Comput. Appl."},{"key":"236_CR13","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Zhang, E., Tian, K., Mao, F., Geathers, M., Shen, X., Gao, Y.: Exploiting statistical correlations for proactive prediction of program behaviors. In: Proceedings of the International Symposium on Code Generation and Optimization (CGO), pp. 248\u2013256 (2010)","DOI":"10.1145\/1772954.1772989"},{"issue":"2","key":"236_CR14","doi-asserted-by":"crossref","first-page":"387","DOI":"10.1109\/JPROC.2004.840447","volume":"93","author":"K Kennedy","year":"2005","unstructured":"Kennedy, K., Broom, B., Chauhan, A., et al.: Telescoping languages: A system for automatic generation of domain languages. Proc. IEEE 93(2), 387\u2013408 (2005)","journal-title":"Proc. IEEE"},{"key":"236_CR15","doi-asserted-by":"crossref","unstructured":"Lee, S., Johnson, T., Eigenmann, R.: Cetus\u2014An extensible compiler infrastructure for source-to-source transformation. In: Proceedings of the 16th Annual Workshop on Languages and Compilers for Parallel Computing (LCPC), pp. 539\u2013553 (2003)","DOI":"10.1007\/978-3-540-24644-2_35"},{"key":"236_CR16","doi-asserted-by":"crossref","unstructured":"Liu, Y., Zhang, E.Z., Shen, X.: A cross-input adaptive framework for gpu programs optimization. In: Proceedings of International Parallel and Distribute Processing Symposium (IPDPS), pp. 1\u201310 (2009)","DOI":"10.1109\/IPDPS.2009.5160988"},{"key":"236_CR17","doi-asserted-by":"crossref","unstructured":"Mao, F., Shen, X.: Cross-input learning and discriminative prediction in evolvable virtual machine. In: Proceedings of the International Symposium on Code Generation and Optimization (CGO), pp. 92\u2013101 (2009)","DOI":"10.1109\/CGO.2009.10"},{"key":"236_CR18","doi-asserted-by":"crossref","unstructured":"Marlet, R., Consel, C., Boinot, P.: Efficient incremental run-time specialization for free. In: Proceedings of ACM SIGPLAN Conference on Programming Language Design and Implementation, pp. 281\u2013292, Atlanta, GA, May (1999)","DOI":"10.1145\/301631.301681"},{"key":"236_CR19","doi-asserted-by":"crossref","unstructured":"Meng, J., Tarjan, D., Skadron, K.: Dynamic warp subdivision for integrated branch and memory divergence tolerance. In: ISCA (2010)","DOI":"10.1145\/1815961.1815992"},{"issue":"2","key":"236_CR20","doi-asserted-by":"crossref","first-page":"232","DOI":"10.1109\/JPROC.2004.840306","volume":"93","author":"M Puschel","year":"2005","unstructured":"Puschel, M., Moura, J.M.F., et al.: SPIRAL: Code generation for DSP transforms. Proc. IEEE 93(2), 232\u2013275 (2005)","journal-title":"Proc. IEEE"},{"key":"236_CR21","doi-asserted-by":"crossref","unstructured":"Ryoo, S., Rodrigues, C.I., Stone, S.S., Baghsorkhi, S.S., Ueng, S. Stratton, J.A., Hwu, W.W.: Program optimization space pruning for a multithreaded GPU. In CGO\u201908: Proceedings of the Sixth Annual IEEE\/ACM International Symposium on Code Generation and, Optimization, pp. 195\u2013204 (2008)","DOI":"10.1145\/1356058.1356084"},{"key":"236_CR22","doi-asserted-by":"crossref","unstructured":"Samadi, M., Hormati, A., Mehrara, M., Lee, J., Mahlke, S.: Adaptive input-aware compilation for graphics engines. In: Proceedings of ACM SIGPLAN Conference on Programming Languages Design and Implementation (2012)","DOI":"10.1145\/2254064.2254067"},{"key":"236_CR23","doi-asserted-by":"crossref","unstructured":"Schordan, M., Quinlan, D.: A source-to-source architecture for user-defined optimizations. In: Proceedings of the Joint Modular Languages Conference held in conjunction with EuroPar\u201903, (2003)","DOI":"10.1007\/978-3-540-45213-3_27"},{"key":"236_CR24","doi-asserted-by":"crossref","unstructured":"Tarjan, D., Meng, J., Skadron, K.: Increasing memory miss tolerance for simd cores. In: SC (2009)","DOI":"10.1145\/1654059.1654082"},{"key":"236_CR25","doi-asserted-by":"crossref","unstructured":"Thomas, N., Tanase, G., Tkachyshyn, O., Perdue, J., Amato, N.M., Rauchwerger, L.: A framework for adaptive algorithm selection in STAPL. In: Proceedings of the Tenth ACM SIGPLAN Symposium on Principles and Practice of Parallel Programming, pp. 277\u2013288 (2005)","DOI":"10.1145\/1065944.1065981"},{"key":"236_CR26","doi-asserted-by":"crossref","unstructured":"Tian, K., Jiang, Y., Zhang, E., Shen, X.: An input-centric paradigm for program dynamic optimizations. In: The Conference on Object-Oriented Programming, Systems, Languages, and Applications (OOPSLA) (2010)","DOI":"10.1145\/1869459.1869471"},{"key":"236_CR27","doi-asserted-by":"crossref","unstructured":"Tian, K., Zhang, E., Shen, X.: A step towards transparent integration of input-consciousness into dynamic program optimizations. In: The Conference on Object-Oriented Programming, Systems, Languages, and Applications (OOPSLA) (2011)","DOI":"10.1145\/2048066.2048103"},{"key":"236_CR28","doi-asserted-by":"crossref","unstructured":"Voss, M., Eigenmann, R.: High-level adaptive program optimization with ADAPT. In: Proceedings of ACM Symposium on Principles and Practice of Parallel Programming, pp. 93\u2013102. Snowbird, Utah, June (2001)","DOI":"10.1145\/379539.379583"},{"issue":"1\u20132","key":"236_CR29","doi-asserted-by":"crossref","first-page":"3","DOI":"10.1016\/S0167-8191(00)00087-9","volume":"27","author":"RC Whaley","year":"2001","unstructured":"Whaley, R.C., Petitet, A., Dongarra, J.: Automated empirical optimizations of software and the ATLAS project. Parallel Comput. 27(1\u20132), 3\u201335 (2001)","journal-title":"Parallel Comput."},{"key":"236_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, E., Jiang, Y., Guo, Z., Tian, K., Shen, X.: On-the-fly elimination of dynamic irregularities for gpu computing. In: Proceedings of the International Conference on Architectural Support for Programming Languages and Operating Systems (2011)","DOI":"10.1145\/1950365.1950408"},{"key":"236_CR31","doi-asserted-by":"crossref","unstructured":"Zhang, E.Z., Jiang, Y., Guo, Z., Shen, X.: Streamlining gpu applications on the fly. In: Proceedings of the ACM International Conference on Supercomputing (ICS), pp. 115\u2013125 (2010)","DOI":"10.1145\/1810085.1810104"}],"container-title":["International Journal of Parallel Programming"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-012-0236-3.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s10766-012-0236-3\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s10766-012-0236-3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,23]],"date-time":"2025-04-23T11:54:11Z","timestamp":1745409251000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s10766-012-0236-3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2012,12,8]]},"references-count":31,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2013,12]]}},"alternative-id":["236"],"URL":"https:\/\/doi.org\/10.1007\/s10766-012-0236-3","relation":{},"ISSN":["0885-7458","1573-7640"],"issn-type":[{"type":"print","value":"0885-7458"},{"type":"electronic","value":"1573-7640"}],"subject":[],"published":{"date-parts":[[2012,12,8]]}}}