Toward performance models of MPI implementations for understanding application scaling issues

Toward performance models of MPI implementations for understanding application scaling issues. Hoefler, T., Gropp, W., Thakur, R., & Träff, J., L. Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics), 6305 LNCS:21-30, 2010.

Website doi abstract bibtex

Designing and tuning parallel applications with MPI, particularly at large scale, requires understanding the performance implications of different choices of algorithms and implementation options. Which algorithm is better depends in part on the performance of the different possible communication approaches, which in turn can depend on both the system hardware and the MPI implementation. In the absence of detailed performance models for different MPI implementations, application developers often must select methods and tune codes without the means to realistically estimate the achievable performance and rationally defend their choices. In this paper, we advocate the construction of more useful performance models that take into account limitations on network-injection rates and effective bisection bandwidth. Since collective communication plays a crucial role in enabling scalability, we also provide analytical models for scalability of collective communication algorithms, such as broadcast, allreduce, and all-to-all. We apply these models to an IBM Blue Gene/P system and compare the analytical performance estimates with experimentally measured values. © 2010 Springer-Verlag.

@article{
 title = {Toward performance models of MPI implementations for understanding application scaling issues},
 type = {article},
 year = {2010},
 keywords = {Achievable performance; All-reduce; Analytical mod,Algorithms; Mathematical models; Scalability; Sup,Message passing},
 pages = {21-30},
 volume = {6305 LNCS},
 websites = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-78149250506&doi=10.1007%2F978-3-642-15646-5_3&partnerID=40&md5=19879e390e0e48653e21bdf68801935d},
 city = {Stuttgart},
 id = {9c07777f-7798-33b2-a08a-b1dfbd4915b8},
 created = {2018-01-09T20:30:38.468Z},
 file_attached = {false},
 profile_id = {42d295c0-0737-38d6-8b43-508cab6ea85d},
 last_modified = {2018-03-12T19:03:18.395Z},
 read = {false},
 starred = {false},
 authored = {true},
 confirmed = {true},
 hidden = {false},
 citation_key = {Hoefler201021},
 source_type = {article},
 notes = {cited By 9; Conference of 17th European MPI Users' Group Meeting, EuroMPI 2010 ; Conference Date: 12 September 2010 Through 15 September 2010; Conference Code:82267},
 folder_uuids = {2aba6c14-9027-4f47-8627-0902e1e2342b},
 private_publication = {false},
 abstract = {Designing and tuning parallel applications with MPI, particularly at large scale, requires understanding the performance implications of different choices of algorithms and implementation options. Which algorithm is better depends in part on the performance of the different possible communication approaches, which in turn can depend on both the system hardware and the MPI implementation. In the absence of detailed performance models for different MPI implementations, application developers often must select methods and tune codes without the means to realistically estimate the achievable performance and rationally defend their choices. In this paper, we advocate the construction of more useful performance models that take into account limitations on network-injection rates and effective bisection bandwidth. Since collective communication plays a crucial role in enabling scalability, we also provide analytical models for scalability of collective communication algorithms, such as broadcast, allreduce, and all-to-all. We apply these models to an IBM Blue Gene/P system and compare the analytical performance estimates with experimentally measured values. © 2010 Springer-Verlag.},
 bibtype = {article},
 author = {Hoefler, T and Gropp, W and Thakur, R and Träff, J L},
 doi = {10.1007/978-3-642-15646-5_3},
 journal = {Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)}
}

Downloads: 0

{"_id":"5K7nGfrAEzQpQyNNF","bibbaseid":"hoefler-gropp-thakur-trff-towardperformancemodelsofmpiimplementationsforunderstandingapplicationscalingissues-2010","downloads":0,"creationDate":"2018-03-12T19:10:28.101Z","title":"Toward performance models of MPI implementations for understanding application scaling issues","author_short":["Hoefler, T.","Gropp, W.","Thakur, R.","Träff, J., L."],"year":2010,"bibtype":"article","biburl":"https://bibbase.org/service/mendeley/42d295c0-0737-38d6-8b43-508cab6ea85d","bibdata":{"title":"Toward performance models of MPI implementations for understanding application scaling issues","type":"article","year":"2010","keywords":"Achievable performance; All-reduce; Analytical mod,Algorithms; Mathematical models; Scalability; Sup,Message passing","pages":"21-30","volume":"6305 LNCS","websites":"https://www.scopus.com/inward/record.uri?eid=2-s2.0-78149250506&doi=10.1007%2F978-3-642-15646-5_3&partnerID=40&md5=19879e390e0e48653e21bdf68801935d","city":"Stuttgart","id":"9c07777f-7798-33b2-a08a-b1dfbd4915b8","created":"2018-01-09T20:30:38.468Z","file_attached":false,"profile_id":"42d295c0-0737-38d6-8b43-508cab6ea85d","last_modified":"2018-03-12T19:03:18.395Z","read":false,"starred":false,"authored":"true","confirmed":"true","hidden":false,"citation_key":"Hoefler201021","source_type":"article","notes":"cited By 9; Conference of 17th European MPI Users' Group Meeting, EuroMPI 2010 ; Conference Date: 12 September 2010 Through 15 September 2010; Conference Code:82267","folder_uuids":"2aba6c14-9027-4f47-8627-0902e1e2342b","private_publication":false,"abstract":"Designing and tuning parallel applications with MPI, particularly at large scale, requires understanding the performance implications of different choices of algorithms and implementation options. Which algorithm is better depends in part on the performance of the different possible communication approaches, which in turn can depend on both the system hardware and the MPI implementation. In the absence of detailed performance models for different MPI implementations, application developers often must select methods and tune codes without the means to realistically estimate the achievable performance and rationally defend their choices. In this paper, we advocate the construction of more useful performance models that take into account limitations on network-injection rates and effective bisection bandwidth. Since collective communication plays a crucial role in enabling scalability, we also provide analytical models for scalability of collective communication algorithms, such as broadcast, allreduce, and all-to-all. We apply these models to an IBM Blue Gene/P system and compare the analytical performance estimates with experimentally measured values. © 2010 Springer-Verlag.","bibtype":"article","author":"Hoefler, T and Gropp, W and Thakur, R and Träff, J L","doi":"10.1007/978-3-642-15646-5_3","journal":"Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)","bibtex":"@article{\n title = {Toward performance models of MPI implementations for understanding application scaling issues},\n type = {article},\n year = {2010},\n keywords = {Achievable performance; All-reduce; Analytical mod,Algorithms; Mathematical models; Scalability; Sup,Message passing},\n pages = {21-30},\n volume = {6305 LNCS},\n websites = {https://www.scopus.com/inward/record.uri?eid=2-s2.0-78149250506&doi=10.1007%2F978-3-642-15646-5_3&partnerID=40&md5=19879e390e0e48653e21bdf68801935d},\n city = {Stuttgart},\n id = {9c07777f-7798-33b2-a08a-b1dfbd4915b8},\n created = {2018-01-09T20:30:38.468Z},\n file_attached = {false},\n profile_id = {42d295c0-0737-38d6-8b43-508cab6ea85d},\n last_modified = {2018-03-12T19:03:18.395Z},\n read = {false},\n starred = {false},\n authored = {true},\n confirmed = {true},\n hidden = {false},\n citation_key = {Hoefler201021},\n source_type = {article},\n notes = {cited By 9; Conference of 17th European MPI Users' Group Meeting, EuroMPI 2010 ; Conference Date: 12 September 2010 Through 15 September 2010; Conference Code:82267},\n folder_uuids = {2aba6c14-9027-4f47-8627-0902e1e2342b},\n private_publication = {false},\n abstract = {Designing and tuning parallel applications with MPI, particularly at large scale, requires understanding the performance implications of different choices of algorithms and implementation options. Which algorithm is better depends in part on the performance of the different possible communication approaches, which in turn can depend on both the system hardware and the MPI implementation. In the absence of detailed performance models for different MPI implementations, application developers often must select methods and tune codes without the means to realistically estimate the achievable performance and rationally defend their choices. In this paper, we advocate the construction of more useful performance models that take into account limitations on network-injection rates and effective bisection bandwidth. Since collective communication plays a crucial role in enabling scalability, we also provide analytical models for scalability of collective communication algorithms, such as broadcast, allreduce, and all-to-all. We apply these models to an IBM Blue Gene/P system and compare the analytical performance estimates with experimentally measured values. © 2010 Springer-Verlag.},\n bibtype = {article},\n author = {Hoefler, T and Gropp, W and Thakur, R and Träff, J L},\n doi = {10.1007/978-3-642-15646-5_3},\n journal = {Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)}\n}","author_short":["Hoefler, T.","Gropp, W.","Thakur, R.","Träff, J., L."],"urls":{"Website":"https://www.scopus.com/inward/record.uri?eid=2-s2.0-78149250506&doi=10.1007%2F978-3-642-15646-5_3&partnerID=40&md5=19879e390e0e48653e21bdf68801935d"},"biburl":"https://bibbase.org/service/mendeley/42d295c0-0737-38d6-8b43-508cab6ea85d","bibbaseid":"hoefler-gropp-thakur-trff-towardperformancemodelsofmpiimplementationsforunderstandingapplicationscalingissues-2010","role":"author","keyword":["Achievable performance; All-reduce; Analytical mod","Algorithms; Mathematical models; Scalability; Sup","Message passing"],"metadata":{"authorlinks":{}},"downloads":0},"search_terms":["toward","performance","models","mpi","implementations","understanding","application","scaling","issues","hoefler","gropp","thakur","träff"],"keywords":["achievable performance; all-reduce; analytical mod","algorithms; mathematical models; scalability; sup","message passing"],"authorIDs":[],"dataSources":["zgahneP4uAjKbudrQ","ya2CyA73rpZseyrZ8","2252seNhipfTmjEBQ","mZL7Ztbm8XZE2mt2K"]}