{"id":"https://openalex.org/W2801415324","doi":"https://doi.org/10.1109/tpds.2018.2827055","title":"A Novel Data-Partitioning Algorithm for Performance Optimization of Data-Parallel Applications on Heterogeneous HPC Platforms","display_name":"A Novel Data-Partitioning Algorithm for Performance Optimization of Data-Parallel Applications on Heterogeneous HPC Platforms","publication_year":2018,"publication_date":"2018-04-16","ids":{"openalex":"https://openalex.org/W2801415324","doi":"https://doi.org/10.1109/tpds.2018.2827055","mag":"2801415324"},"language":"en","primary_location":{"id":"doi:10.1109/tpds.2018.2827055","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tpds.2018.2827055","pdf_url":null,"source":{"id":"https://openalex.org/S97130795","display_name":"IEEE Transactions on Parallel and Distributed Systems","issn_l":"1045-9219","issn":["1045-9219","1558-2183","2161-9883"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Parallel and Distributed Systems","raw_type":"journal-article"},"type":"article","indexed_in":["crossref"],"open_access":{"is_oa":false,"oa_status":"closed","oa_url":null,"any_repository_has_fulltext":false},"authorships":[{"author_position":"first","author":{"id":"https://openalex.org/A5006106085","display_name":"Hamidreza Khaleghzadeh","orcid":"https://orcid.org/0000-0003-4070-7468"},"institutions":[{"id":"https://openalex.org/I100930933","display_name":"University College Dublin","ror":"https://ror.org/05m7pjf47","country_code":"IE","type":"education","lineage":["https://openalex.org/I100930933"]}],"countries":["IE"],"is_corresponding":false,"raw_author_name":"Hamidreza Khaleghzadeh","raw_affiliation_strings":["School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland"],"raw_orcid":"https://orcid.org/0000-0003-4070-7468","affiliations":[{"raw_affiliation_string":"School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland","institution_ids":["https://openalex.org/I100930933"]}]},{"author_position":"middle","author":{"id":"https://openalex.org/A5078230040","display_name":"Ravi Reddy Manumachu","orcid":"https://orcid.org/0000-0001-9181-3290"},"institutions":[{"id":"https://openalex.org/I100930933","display_name":"University College Dublin","ror":"https://ror.org/05m7pjf47","country_code":"IE","type":"education","lineage":["https://openalex.org/I100930933"]}],"countries":["IE"],"is_corresponding":false,"raw_author_name":"Ravindranath Reddy Manumachu","raw_affiliation_strings":["School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland"],"raw_orcid":"https://orcid.org/0000-0001-9181-3290","affiliations":[{"raw_affiliation_string":"School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland","institution_ids":["https://openalex.org/I100930933"]}]},{"author_position":"last","author":{"id":"https://openalex.org/A5084068586","display_name":"Alexey Lastovetsky","orcid":"https://orcid.org/0000-0001-9460-3897"},"institutions":[{"id":"https://openalex.org/I100930933","display_name":"University College Dublin","ror":"https://ror.org/05m7pjf47","country_code":"IE","type":"education","lineage":["https://openalex.org/I100930933"]}],"countries":["IE"],"is_corresponding":false,"raw_author_name":"Alexey Lastovetsky","raw_affiliation_strings":["School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland"],"raw_orcid":"https://orcid.org/0000-0001-9460-3897","affiliations":[{"raw_affiliation_string":"School of Computer Science, University College Dublin, Belfield, Dublin 4, Ireland","institution_ids":["https://openalex.org/I100930933"]}]}],"institutions":[],"countries_distinct_count":1,"institutions_distinct_count":1,"corresponding_author_ids":[],"corresponding_institution_ids":["https://openalex.org/I100930933"],"apc_list":null,"apc_paid":null,"fwci":6.177,"has_fulltext":false,"cited_by_count":47,"citation_normalized_percentile":{"value":0.97746852,"is_in_top_1_percent":false,"is_in_top_10_percent":true},"cited_by_percentile_year":{"min":96,"max":99},"biblio":{"volume":"29","issue":"10","first_page":"2176","last_page":"2190"},"is_retracted":false,"is_paratext":false,"is_xpac":false,"primary_topic":{"id":"https://openalex.org/T10054","display_name":"Parallel Computing and Optimization Techniques","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1708","display_name":"Hardware and Architecture"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},"topics":[{"id":"https://openalex.org/T10054","display_name":"Parallel Computing and Optimization Techniques","score":1.0,"subfield":{"id":"https://openalex.org/subfields/1708","display_name":"Hardware and Architecture"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T10715","display_name":"Distributed and Parallel Computing Systems","score":0.9997000098228455,"subfield":{"id":"https://openalex.org/subfields/1705","display_name":"Computer Networks and Communications"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}},{"id":"https://openalex.org/T11181","display_name":"Advanced Data Storage Technologies","score":0.9994999766349792,"subfield":{"id":"https://openalex.org/subfields/1705","display_name":"Computer Networks and Communications"},"field":{"id":"https://openalex.org/fields/17","display_name":"Computer Science"},"domain":{"id":"https://openalex.org/domains/3","display_name":"Physical Sciences"}}],"keywords":[{"id":"https://openalex.org/keywords/computer-science","display_name":"Computer science","score":0.9006576538085938},{"id":"https://openalex.org/keywords/parallel-computing","display_name":"Parallel computing","score":0.6797096133232117},{"id":"https://openalex.org/keywords/multi-core-processor","display_name":"Multi-core processor","score":0.5396125316619873},{"id":"https://openalex.org/keywords/xeon","display_name":"Xeon","score":0.5185216665267944},{"id":"https://openalex.org/keywords/supercomputer","display_name":"Supercomputer","score":0.47599661350250244},{"id":"https://openalex.org/keywords/pci-express","display_name":"PCI Express","score":0.43860718607902527},{"id":"https://openalex.org/keywords/memory-bandwidth","display_name":"Memory bandwidth","score":0.41132283210754395},{"id":"https://openalex.org/keywords/distributed-computing","display_name":"Distributed computing","score":0.39857491850852966},{"id":"https://openalex.org/keywords/embedded-system","display_name":"Embedded system","score":0.26025286316871643},{"id":"https://openalex.org/keywords/field-programmable-gate-array","display_name":"Field-programmable gate array","score":0.24438077211380005}],"concepts":[{"id":"https://openalex.org/C41008148","wikidata":"https://www.wikidata.org/wiki/Q21198","display_name":"Computer science","level":0,"score":0.9006576538085938},{"id":"https://openalex.org/C173608175","wikidata":"https://www.wikidata.org/wiki/Q232661","display_name":"Parallel computing","level":1,"score":0.6797096133232117},{"id":"https://openalex.org/C78766204","wikidata":"https://www.wikidata.org/wiki/Q555032","display_name":"Multi-core processor","level":2,"score":0.5396125316619873},{"id":"https://openalex.org/C145108525","wikidata":"https://www.wikidata.org/wiki/Q656154","display_name":"Xeon","level":2,"score":0.5185216665267944},{"id":"https://openalex.org/C83283714","wikidata":"https://www.wikidata.org/wiki/Q121117","display_name":"Supercomputer","level":2,"score":0.47599661350250244},{"id":"https://openalex.org/C64270927","wikidata":"https://www.wikidata.org/wiki/Q206924","display_name":"PCI Express","level":3,"score":0.43860718607902527},{"id":"https://openalex.org/C188045654","wikidata":"https://www.wikidata.org/wiki/Q17148339","display_name":"Memory bandwidth","level":2,"score":0.41132283210754395},{"id":"https://openalex.org/C120314980","wikidata":"https://www.wikidata.org/wiki/Q180634","display_name":"Distributed computing","level":1,"score":0.39857491850852966},{"id":"https://openalex.org/C149635348","wikidata":"https://www.wikidata.org/wiki/Q193040","display_name":"Embedded system","level":1,"score":0.26025286316871643},{"id":"https://openalex.org/C42935608","wikidata":"https://www.wikidata.org/wiki/Q190411","display_name":"Field-programmable gate array","level":2,"score":0.24438077211380005}],"mesh":[],"locations_count":1,"locations":[{"id":"doi:10.1109/tpds.2018.2827055","is_oa":false,"landing_page_url":"https://doi.org/10.1109/tpds.2018.2827055","pdf_url":null,"source":{"id":"https://openalex.org/S97130795","display_name":"IEEE Transactions on Parallel and Distributed Systems","issn_l":"1045-9219","issn":["1045-9219","1558-2183","2161-9883"],"is_oa":false,"is_in_doaj":false,"is_core":true,"host_organization":"https://openalex.org/P4310319808","host_organization_name":"Institute of Electrical and Electronics Engineers","host_organization_lineage":["https://openalex.org/P4310319808"],"host_organization_lineage_names":["Institute of Electrical and Electronics Engineers"],"type":"journal"},"license":null,"license_id":null,"version":"publishedVersion","is_accepted":true,"is_published":true,"raw_source_name":"IEEE Transactions on Parallel and Distributed Systems","raw_type":"journal-article"}],"best_oa_location":null,"sustainable_development_goals":[{"display_name":"Affordable and clean energy","score":0.800000011920929,"id":"https://metadata.un.org/sdg/7"}],"awards":[{"id":"https://openalex.org/G1766076957","display_name":"Meeting the Future Challenges of Heterogeneous and Extreme-Scale Parallel Computing","funder_award_id":"14/IA/2474","funder_id":"https://openalex.org/F4320320847","funder_display_name":"Science Foundation Ireland"}],"funders":[{"id":"https://openalex.org/F4320320847","display_name":"Science Foundation Ireland","ror":"https://ror.org/0271asj38"}],"has_content":{"pdf":false,"grobid_xml":false},"content_urls":null,"referenced_works_count":54,"referenced_works":["https://openalex.org/W1528000468","https://openalex.org/W1549662857","https://openalex.org/W2003094427","https://openalex.org/W2020177163","https://openalex.org/W2029856617","https://openalex.org/W2030129803","https://openalex.org/W2040015943","https://openalex.org/W2067052471","https://openalex.org/W2081422957","https://openalex.org/W2093326633","https://openalex.org/W2102624126","https://openalex.org/W2109426995","https://openalex.org/W2110165211","https://openalex.org/W2116367043","https://openalex.org/W2122570236","https://openalex.org/W2122619936","https://openalex.org/W2123031400","https://openalex.org/W2125980577","https://openalex.org/W2132745343","https://openalex.org/W2133684822","https://openalex.org/W2137789117","https://openalex.org/W2142421493","https://openalex.org/W2147152185","https://openalex.org/W2154750565","https://openalex.org/W2156114981","https://openalex.org/W2156371682","https://openalex.org/W2163674731","https://openalex.org/W2164675184","https://openalex.org/W2166417226","https://openalex.org/W2170611190","https://openalex.org/W2171473263","https://openalex.org/W2195884801","https://openalex.org/W2241264454","https://openalex.org/W2278129015","https://openalex.org/W2397644582","https://openalex.org/W2401335561","https://openalex.org/W2508492666","https://openalex.org/W2519627041","https://openalex.org/W2526539616","https://openalex.org/W2573700972","https://openalex.org/W2755294164","https://openalex.org/W2766505687","https://openalex.org/W4235597383","https://openalex.org/W4244789640","https://openalex.org/W4251173935","https://openalex.org/W6658133388","https://openalex.org/W6677237885","https://openalex.org/W6679759112","https://openalex.org/W6679787488","https://openalex.org/W6681419532","https://openalex.org/W6684368365","https://openalex.org/W6684857869","https://openalex.org/W6689915511","https://openalex.org/W6695135304"],"related_works":["https://openalex.org/W2791177606","https://openalex.org/W1615383022","https://openalex.org/W2340180648","https://openalex.org/W2103667109","https://openalex.org/W2022666014","https://openalex.org/W3210911584","https://openalex.org/W2031026393","https://openalex.org/W2995515486","https://openalex.org/W2079303253","https://openalex.org/W2085237598"],"abstract_inverted_index":{"Modern":[0],"HPC":[1,193],"platforms":[2,143,159,194],"have":[3,83,132],"become":[4],"highly":[5],"heterogeneous":[6,192,233,291],"owing":[7],"to":[8,30,41,87,100,108,136,147,174],"tight":[9],"integration":[10],"of":[11,35,138,153,185,187,211,217,244,251,263,276,293],"multicore":[12,102,301],"CPUs":[13],"and":[14,37,60,77,89,105,163,254,274,285,308],"accelerators":[15,94],"(such":[16],"as":[17,53,65,223],"Graphics":[18],"Processing":[19],"Units,":[20],"Intel":[21,300,310],"Xeon":[22,311],"Phis,":[23],"or":[24],"Field-Programmable":[25],"Gate":[26],"Arrays)":[27],"empowering":[28],"them":[29],"maximize":[31],"the":[32,93,101,124,150,167,183,208,214,218,242,249,252,261,264,272],"dominant":[33],"objectives":[34],"performance":[36,151],"energy":[38],"efficiency.":[39],"Due":[40,146],"this":[42,179],"inherent":[43],"characteristic,":[44],"processing":[45],"elements":[46],"contend":[47],"for":[48,118,144,195],"shared":[49,61],"on-chip":[50],"resources":[51,63],"such":[52,64],"Last":[54],"Level":[55],"Cache":[56],"(LLC),":[57],"interconnect,":[58],"etc.":[59,69],"nodal":[62],"DRAM,":[66],"PCI-E":[67,113],"links,":[68],"This":[70,220],"has":[71],"resulted":[72],"in":[73,213],"severe":[74],"resource":[75],"contention":[76],"Non-Uniform":[78],"Memory":[79],"Access":[80],"(NUMA)":[81],"that":[82,169],"posed":[84],"serious":[85],"challenges":[86,135],"model":[88],"algorithm":[90,221,253,278],"developers.":[91],"Moreover,":[92],"feature":[95],"limited":[96,111],"main":[97],"memory":[98],"compared":[99],"CPU":[103],"host":[104],"are":[106,160],"connected":[107],"it":[109],"via":[110],"bandwidth":[112],"links":[114],"thereby":[115],"requiring":[116],"support":[117],"efficient":[119],"out-of-card":[120],"execution.":[121],"To":[122],"summarize,":[123],"complexities":[125],"(resource":[126],"contention,":[127],"NUMA,":[128],"accelerator-specific":[129],"limitations,":[130],"etc.)":[131],"introduced":[133],"new":[134,201],"optimization":[137,186],"data-parallel":[139,154,188,281],"applications":[140,155,189],"on":[141,157,190,289],"these":[142,148,158,245],"performance.":[145,196],"complexities,":[149],"profiles":[152],"executing":[156],"not":[161,237],"smooth":[162],"deviate":[164],"significantly":[165],"from":[166],"shapes":[168,243],"allowed":[170],"state-of-the-art":[171],"load-balancing":[172],"algorithms":[173],"find":[175],"optimal":[176],"solutions.":[177],"In":[178],"paper,":[180],"we":[181],"formulate":[182],"problem":[184],"modern":[191],"We":[197,247,269],"then":[198],"propose":[199],"a":[200,225,290],"model-based":[202],"data":[203],"partitioning":[204],"algorithm,":[205],"which":[206],"minimizes":[207],"execution":[209,216],"time":[210],"computations":[212],"parallel":[215],"application.":[219],"takes":[222],"input":[224,265],"set":[226],"ofpdiscrete":[227],"speed":[228,267],"functions":[229],"corresponding":[230],"top":[231],"available":[232],"processors.":[234],"It":[235],"does":[236],"make":[238],"any":[239],"assumptions":[240],"about":[241],"functions.":[246,268],"prove":[248],"correctness":[250],"its":[255],"complexity":[256],"ofO(m3\u00d7p3),":[257],"where":[258,295],"m":[259],"is":[260],"cardinality":[262],"discrete":[266],"experimentally":[270],"demonstrate":[271],"optimality":[273],"efficiency":[275],"our":[277],"using":[279],"two":[280],"applications,":[282],"matrix":[283],"multiplication":[284],"fast":[286],"Fourier":[287],"transform,":[288],"cluster":[292],"nodes":[294],"each":[296],"node":[297],"contains":[298],"an":[299,304,309],"Haswell":[302],"CPU,":[303],"Nvidia":[305],"K40c":[306],"GPU,":[307],"Phi":[312],"co-processor.":[313]},"counts_by_year":[{"year":2025,"cited_by_count":5},{"year":2024,"cited_by_count":3},{"year":2023,"cited_by_count":4},{"year":2022,"cited_by_count":9},{"year":2021,"cited_by_count":6},{"year":2020,"cited_by_count":9},{"year":2019,"cited_by_count":7},{"year":2018,"cited_by_count":4}],"updated_date":"2026-07-29T09:40:50.615796","created_date":"2025-10-10T00:00:00"}
