@inproceedings{Castro-Pereira-WAMCA:2017,
abstract = {The stencil pattern is important in many scientific and engineering domains, spurring great interest from researchers and industry. In recent years, various optimizations have been proposed for parallel stencil applications running on GPUs. However, most of the runtime systems that execute those applications often fail to fully utilize the parallelism of modern heterogeneous systems. In this paper, we propose a mechanism based on machine learning that automatically partitions stencil computations across CPU and GPU. We implemented it into the PSkel framework and found that the mechanism can boost the performance of stencil applications on average by 17.9x compared to their sequential CPU-only counterparts, by 1.34x compared to a GPU-only version, and by 1.48x compared to a parallel CPU-only version.},
address = {Campinas},
author = {Pereira, Alyson Deives and Rocha, Rodrigo Caetano de Oliveira and Ramos, Luiz and Castro, M{\'{a}}rcio and G{\'{o}}es, Lu{\'{i}}s Fabr{\'{i}}cio Wanderley},
booktitle = {International Symposium on Computer Architecture and High Performance Computing Workshops (SBAC-PADW)},
doi = {10.1109/SBAC-PADW.2017.16},
keywords = {Decision Tree Learning,Feature extraction,Graphics processing units,Hardware,Performance evaluation,Programming,Runtime,Stencil,Training,Work Partitioning},
pages = {43--48},
publisher = {IEEE Computer Society},
title = {{Automatic Partitioning of Stencil Computations on Heterogeneous Systems}},
year = {2017}
}
