Powered by
2024 IEEE/ACM International Symposium on Code Generation and Optimization (CGO), March 02–06, 2024,
Edinburgh, United Kingdom
Frontmatter
Compilers for Machine Learning
A Tensor Algebra Compiler for Sparse Differentiation
Amir Shaikhha,
Mathieu Huot, and
Shideh Hashemian
(University of Edinburgh, United Kingdom; University of Oxford, United Kingdom)
@InProceedings{CGO24p1,
author = {Amir Shaikhha and Mathieu Huot and Shideh Hashemian},
title = {A Tensor Algebra Compiler for Sparse Differentiation},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {1-0},
doi = {},
year = {2024},
}
Article: cgo24main-p38-p doi:
Energy-Aware Tile Size Selection for Affine Programs on GPUs
Malith Jayaweera,
Martin Kong,
Yanzhi Wang, and
David Kaeli
(Northeastern University, USA; Ohio State University, USA)
@InProceedings{CGO24p17,
author = {Malith Jayaweera and Martin Kong and Yanzhi Wang and David Kaeli},
title = {Energy-Aware Tile Size Selection for Affine Programs on GPUs},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {17-16},
doi = {},
year = {2024},
}
Artifacts Functional
Article: cgo24main-p83-p doi:
PolyTOPS: Reconfigurable and Flexible Polyhedral Scheduler
Gianpietro Consolaro,
Zhen Zhang,
Harenome Razanajato,
Nelson Lossing,
Nassim Tchoulak,
Adilla Susungi,
Artur Cesar Araujo Alves,
Renwei Zhang,
Denis Barthou,
Corinne Ancourt, and
Cédric Bastoul
(Huawei Technologies, France; Mines Paris-PSL, France; Huawei Technologies, China)
@InProceedings{CGO24p33,
author = {Gianpietro Consolaro and Zhen Zhang and Harenome Razanajato and Nelson Lossing and Nassim Tchoulak and Adilla Susungi and Artur Cesar Araujo Alves and Renwei Zhang and Denis Barthou and Corinne Ancourt and Cédric Bastoul},
title = {PolyTOPS: Reconfigurable and Flexible Polyhedral Scheduler},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {33-32},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p143-p doi:
Machine-Learning Guided Optimizations
AskIt: Unified Programming Interface for Programming with Large Language Models
Katsumi Okuda and
Saman Amarasinghe
(Massachusetts Institute of Technology, USA; Mitsubishi Electric Corporation, Japan)
@InProceedings{CGO24p49,
author = {Katsumi Okuda and Saman Amarasinghe},
title = {AskIt: Unified Programming Interface for Programming with Large Language Models},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {49-48},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p229-p doi:
Revealing Compiler Heuristics through Automated Discovery and Optimization
Volker Seeker,
Chris Cummins,
Murray Cole,
Björn Franke,
Kim Hazelwood, and
Hugh Leather
(Meta AI Research, USA; University of Edinburgh, United Kingdom)
@InProceedings{CGO24p65,
author = {Volker Seeker and Chris Cummins and Murray Cole and Björn Franke and Kim Hazelwood and Hugh Leather},
title = {Revealing Compiler Heuristics through Automated Discovery and Optimization},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {65-64},
doi = {},
year = {2024},
}
Article: cgo24main-p73-p doi:
SLaDe: A Portable Small Language Model Decompiler for Optimized Assembly
Jordi Armengol-Estapé,
Jackson Woodruff,
Chris Cummins, and
Michael F. P. O'Boyle
(University of Edinburgh, United Kingdom; Meta AI Research, USA)
@InProceedings{CGO24p81,
author = {Jordi Armengol-Estapé and Jackson Woodruff and Chris Cummins and Michael F. P. O'Boyle},
title = {SLaDe: A Portable Small Language Model Decompiler for Optimized Assembly},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {81-80},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Article: cgo24main-p72-p doi:
TapeFlow: Streaming Gradient Tapes in Automatic Differentiation
Milad Hakimi and
Arrvindh Shriraman
(Simon Fraser University, Canada)
@InProceedings{CGO24p97,
author = {Milad Hakimi and Arrvindh Shriraman},
title = {TapeFlow: Streaming Gradient Tapes in Automatic Differentiation},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {97-96},
doi = {},
year = {2024},
}
Article: cgo24main-p55-p doi:
Compilers for GPUs
A Framework for Fine-Grained Synchronization of Dependent GPU Kernels
Abhinav Jangda,
Saeed Maleki,
Maryam Mehri Dehnavi,
Madan Musuvathi, and
Olli Saarikivi
(Microsoft Research, USA; University of Toronto, Canada)
@InProceedings{CGO24p113,
author = {Abhinav Jangda and Saeed Maleki and Maryam Mehri Dehnavi and Madan Musuvathi and Olli Saarikivi},
title = {A Framework for Fine-Grained Synchronization of Dependent GPU Kernels},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {113-112},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p62-p doi:
Enhancing Performance through Control-Flow Unmerging and Loop Unrolling on GPUs
Alnis Murtovi,
Giorgis Georgakoudis,
Konstantinos Parasyris,
Chunhua Liao,
Ignacio Laguna, and
Bernhard Steffen
(TU Dortmund, Germany; Lawrence Livermore National Laboratory, USA)
@InProceedings{CGO24p129,
author = {Alnis Murtovi and Giorgis Georgakoudis and Konstantinos Parasyris and Chunhua Liao and Ignacio Laguna and Bernhard Steffen},
title = {Enhancing Performance through Control-Flow Unmerging and Loop Unrolling on GPUs},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {129-128},
doi = {},
year = {2024},
}
Artifacts Reusable
Results Reproduced
Article: cgo24main-p103-p doi:
Retargeting and Respecializing GPU Workloads for Performance Portability
Ivan R. Ivanov,
Oleksandr Zinenko,
Jens Domke,
Toshio Endo, and
William S. Moses
(Tokyo Institute of Technology, Japan; RIKEN R-CCS, Japan; Google DeepMind, France; University of Illinois at Urbana-Champaign, USA; Google DeepMind, USA)
@InProceedings{CGO24p145,
author = {Ivan R. Ivanov and Oleksandr Zinenko and Jens Domke and Toshio Endo and William S. Moses},
title = {Retargeting and Respecializing GPU Workloads for Performance Portability},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {145-144},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p19-p doi:
Seer: Predictive Runtime Kernel Selection for Irregular Problems
Ryan Swann,
Muhammad Osama,
Karthik Sangaiah, and
Jalal Mahmud
(AMD, USA)
@InProceedings{CGO24p161,
author = {Ryan Swann and Muhammad Osama and Karthik Sangaiah and Jalal Mahmud},
title = {Seer: Predictive Runtime Kernel Selection for Irregular Problems},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {161-160},
doi = {},
year = {2024},
}
Artifacts Reusable
Results Reproduced
Article: cgo24main-p238-p doi:
Custom Processors
AXI4MLIR: User-Driven Automatic Host Code Generation for Custom AXI-Based Accelerators
Nicolas Bohm Agostini,
Jude Haris,
Perry Gibson,
Malith Jayaweera,
Norm Rubin,
Antonino Tumeo,
José L. Abellán,
José Cano, and
David Kaeli
(Northeastern University, USA; Pacific Northwest National Laboratory, USA; University of Glasgow, United Kingdom; University of Murcia, Spain)
@InProceedings{CGO24p177,
author = {Nicolas Bohm Agostini and Jude Haris and Perry Gibson and Malith Jayaweera and Norm Rubin and Antonino Tumeo and José L. Abellán and José Cano and David Kaeli},
title = {AXI4MLIR: User-Driven Automatic Host Code Generation for Custom AXI-Based Accelerators},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {177-176},
doi = {},
year = {2024},
}
Artifacts Reusable
Results Reproduced
Article: cgo24main-p7-p doi:
Ecmas: Efficient Circuit Mapping and Scheduling for Surface Code
Mingzheng Zhu,
Hao Fu,
Jun Wu,
Chi Zhang,
Wei Xie, and
Xiang-Yang Li
(University of Science and Technology of China, China)
@InProceedings{CGO24p193,
author = {Mingzheng Zhu and Hao Fu and Jun Wu and Chi Zhang and Wei Xie and Xiang-Yang Li},
title = {Ecmas: Efficient Circuit Mapping and Scheduling for Surface Code},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {193-192},
doi = {},
year = {2024},
}
Article: cgo24main-p100-p doi:
PresCount: Effective Register Allocation for Bank Conflict Reduction
Xiaofeng Guan,
Hao Zhou,
Guoqing Bao,
Handong Li,
Liang Zhu, and
Jianguo Yao
(Shanghai Jiao Tong University, China; Shanghai Enflame Technology, China)
@InProceedings{CGO24p209,
author = {Xiaofeng Guan and Hao Zhou and Guoqing Bao and Handong Li and Liang Zhu and Jianguo Yao},
title = {PresCount: Effective Register Allocation for Bank Conflict Reduction},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {209-208},
doi = {},
year = {2024},
}
Article: cgo24main-p16-p doi:
Tackling the Matrix Multiplication Micro-kernel Generation with Exo
Adrián Castelló,
Julian Bellavita,
Grace Dinh,
Yuka Ikarashi, and
Héctor Martínez
(Universitat Politècnica de València, Spain; Cornell University, USA; University of California at Berkeley, USA; Massachusetts Institute of Technology, USA; Universidad de Córdoba, Spain)
@InProceedings{CGO24p225,
author = {Adrián Castelló and Julian Bellavita and Grace Dinh and Yuka Ikarashi and Héctor Martínez},
title = {Tackling the Matrix Multiplication Micro-kernel Generation with Exo},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {225-224},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Article: cgo24main-p49-p doi:
Compiler Construction
One Automaton to Rule Them All: Beyond Multiple Regular Expressions Execution
Luisa Cicolini,
Filippo Carloni,
Marco D. Santambrogio, and
Davide Conficconi
(Politecnico di Milano, Italy)
@InProceedings{CGO24p241,
author = {Luisa Cicolini and Filippo Carloni and Marco D. Santambrogio and Davide Conficconi},
title = {One Automaton to Rule Them All: Beyond Multiple Regular Expressions Execution},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {241-240},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p368-p doi:
Enabling Fine-Grained Incremental Builds by Making Compiler Stateful
Ruobing Han,
Jisheng Zhao, and
Hyesoon Kim
(Georgia Institute of Technology, USA)
@InProceedings{CGO24p273,
author = {Ruobing Han and Jisheng Zhao and Hyesoon Kim},
title = {Enabling Fine-Grained Incremental Builds by Making Compiler Stateful},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {273-272},
doi = {},
year = {2024},
}
Article: cgo24main-p121-p doi:
Custom Environments
DrPy: Pinpointing Inefficient Memory Usage in Multi-Layer Python Applications
Jinku Cui,
Qidong Zhao,
Yueming Hao, and
Xu Liu
(North Carolina State University, USA)
@InProceedings{CGO24p305,
author = {Jinku Cui and Qidong Zhao and Yueming Hao and Xu Liu},
title = {DrPy: Pinpointing Inefficient Memory Usage in Multi-Layer Python Applications},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {305-304},
doi = {},
year = {2024},
}
Artifacts Functional
Article: cgo24main-p141-p doi:
SCHEMATIC: Compile-Time Checkpoint Placement and Memory Allocation for Intermittent Systems
Hugo Reymond,
Jean-Luc Béchennec,
Mikaël Briday,
Sébastien Faucou,
Isabelle Puaut, and
Erven Rohou
(Université de Rennes - Inria - CNRS - IRISA, France; Nantes Université - École Centrale Nantes - CNRS - LS2N - UMR 6004, France)
@InProceedings{CGO24p321,
author = {Hugo Reymond and Jean-Luc Béchennec and Mikaël Briday and Sébastien Faucou and Isabelle Puaut and Erven Rohou},
title = {SCHEMATIC: Compile-Time Checkpoint Placement and Memory Allocation for Intermittent Systems},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {321-320},
doi = {},
year = {2024},
}
Article: cgo24main-p24-p doi:
Static/Dynamic Analyses
Boosting the Performance of Multi-solver IFDS Algorithms with Flow-Sensitivity Optimizations
Haofeng Li,
Jie Lu,
Haining Meng,
Liqing Cao,
Lian Li, and
Lin Gao
(Institute of Computing Technology at Chinese Academy of Sciences, China; University of Chinese Academy of Sciences, China; Zhongguancun Laboratory, China; TianqiSoft, China)
@InProceedings{CGO24p369,
author = {Haofeng Li and Jie Lu and Haining Meng and Liqing Cao and Lian Li and Lin Gao},
title = {Boosting the Performance of Multi-solver IFDS Algorithms with Flow-Sensitivity Optimizations},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {369-368},
doi = {},
year = {2024},
}
Artifacts Reusable
Results Reproduced
Article: cgo24main-p193-p doi:
Representing Data Collections in an SSA Form
Tommy McMichen,
Nathan Greiner,
Peter Zhong,
Federico Sossai,
Atmn Patel, and
Simone Campanoni
(Northwestern University, USA)
@InProceedings{CGO24p385,
author = {Tommy McMichen and Nathan Greiner and Peter Zhong and Federico Sossai and Atmn Patel and Simone Campanoni},
title = {Representing Data Collections in an SSA Form},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {385-384},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p148-p doi:
Revamping Sampling-Based PGO with Context-Sensitivity and Pseudo-instrumentation
Wenlei He,
Hongtao Yu,
Lei Wang, and
Taewook Oh
(Meta, USA)
@InProceedings{CGO24p401,
author = {Wenlei He and Hongtao Yu and Lei Wang and Taewook Oh},
title = {Revamping Sampling-Based PGO with Context-Sensitivity and Pseudo-instrumentation},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {401-400},
doi = {},
year = {2024},
}
Article: cgo24main-p138-p doi:
Supporting Tools
Compiler Testing with Relaxed Memory Models
Luke Geeson and
Lee Smith
(University College London, United Kingdom; Arm, United Kingdom)
@InProceedings{CGO24p417,
author = {Luke Geeson and Lee Smith},
title = {Compiler Testing with Relaxed Memory Models},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {417-416},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p169-p doi:
High-Throughput, Formal-Methods-Assisted Fuzzing for LLVM
Yuyou Fan and
John Regehr
(University of Utah, USA)
@InProceedings{CGO24p433,
author = {Yuyou Fan and John Regehr},
title = {High-Throughput, Formal-Methods-Assisted Fuzzing for LLVM},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {433-432},
doi = {},
year = {2024},
}
Article: cgo24main-p241-p doi:
EasyTracker: A Python Library for Controlling and Inspecting Program Execution
Théo Barollet,
Christophe Guillon,
Manuel Selva,
François Broquedis,
Florent Bouchez-Tichadou, and
Fabrice Rastello
(University Grenoble Alpes - Inria - CNRS - Grenoble INP - LIG, France)
@InProceedings{CGO24p449,
author = {Théo Barollet and Christophe Guillon and Manuel Selva and François Broquedis and Florent Bouchez-Tichadou and Fabrice Rastello},
title = {EasyTracker: A Python Library for Controlling and Inspecting Program Execution},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {449-448},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p359-p doi:
OptiWISE: Combining Sampling and Instrumentation for Granular CPI Analysis
Yuxin Guo,
Alex W. Chadwick,
Márton Erdős,
Utpal Bora,
Ilias Vougioukas,
Giacomo Gabrielli, and
Timothy M. Jones
(University of Cambridge, United Kingdom; Arm, USA; Arm, United Kingdom)
@InProceedings{CGO24p465,
author = {Yuxin Guo and Alex W. Chadwick and Márton Erdős and Utpal Bora and Ilias Vougioukas and Giacomo Gabrielli and Timothy M. Jones},
title = {OptiWISE: Combining Sampling and Instrumentation for Granular CPI Analysis},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {465-464},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Reusable
Results Reproduced
Article: cgo24main-p78-p doi:
Practice and Experience
EasyView: Bringing Performance Profiles into Integrated Development Environments
Qidong Zhao,
Milind Chabbi, and
Xu Liu
(North Carolina State University, USA; Scalable Machines Research, USA)
@InProceedings{CGO24p481,
author = {Qidong Zhao and Milind Chabbi and Xu Liu},
title = {EasyView: Bringing Performance Profiles into Integrated Development Environments},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {481-480},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Artifacts Functional
Results Reproduced
Article: cgo24main-p107-p doi:
Experiences Building an MLIR-Based SYCL Compiler
Ettore Tiotto,
Víctor Pérez,
Whitney Tsang,
Lukas Sommer,
Julian Oppermann,
Victor Lomüller,
Mehdi Goli, and
James Brodman
(Intel Corporation, Canada; Codeplay Software, United Kingdom; Intel Corporation, USA)
@InProceedings{CGO24p497,
author = {Ettore Tiotto and Víctor Pérez and Whitney Tsang and Lukas Sommer and Julian Oppermann and Victor Lomüller and Mehdi Goli and James Brodman},
title = {Experiences Building an MLIR-Based SYCL Compiler},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {497-496},
doi = {},
year = {2024},
}
Published Artifact
Artifacts Available
Article: cgo24main-p70-p doi:
Unveiling and Vanquishing Goroutine Leaks in Enterprise Microservices: A Dynamic Analysis Approach
Georgian-Vlad Saioc,
Dmitriy Shirchenko, and
Milind Chabbi
(Aarhus University, Denmark; Uber Technologies, Denmark; Uber Technologies, USA)
@InProceedings{CGO24p513,
author = {Georgian-Vlad Saioc and Dmitriy Shirchenko and Milind Chabbi},
title = {Unveiling and Vanquishing Goroutine Leaks in Enterprise Microservices: A Dynamic Analysis Approach},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {513-512},
doi = {},
year = {2024},
}
Article: cgo24main-p237-p doi:
Acceleration Techniques
A System-Level Dynamic Binary Translator using Automatically-Learned Translation Rules
Jinhu Jiang,
Chaoyi Liang,
Rongchao Dong,
Zhaohui Yang,
Zhongjun Zhou,
Wenwen Wang,
Pen-Chung Yew, and
Weihua Zhang
(Fudan University, China; University of Georgia, USA; University of Minnesota at Twin Cities, USA)
@InProceedings{CGO24p529,
author = {Jinhu Jiang and Chaoyi Liang and Rongchao Dong and Zhaohui Yang and Zhongjun Zhou and Wenwen Wang and Pen-Chung Yew and Weihua Zhang},
title = {A System-Level Dynamic Binary Translator using Automatically-Learned Translation Rules},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {529-528},
doi = {},
year = {2024},
}
Article: cgo24main-p51-p doi:
Instruction Scheduling for the GPU on the GPU
Ghassan Shobaki,
Pınar Muyan-Özçelik,
Josh Hutton,
Bruce Linck,
Vladislav Malyshenko,
Austin Kerbow,
Ronaldo Ramirez-Ortega, and
Vahl Scott Gordon
(California State University, Sacramento, USA; Advanced Micro Devices, USA)
@InProceedings{CGO24p545,
author = {Ghassan Shobaki and Pınar Muyan-Özçelik and Josh Hutton and Bruce Linck and Vladislav Malyshenko and Austin Kerbow and Ronaldo Ramirez-Ortega and Vahl Scott Gordon},
title = {Instruction Scheduling for the GPU on the GPU},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {545-544},
doi = {},
year = {2024},
}
Artifacts Functional
Article: cgo24main-p69-p doi:
JITSPMM: Just-in-Time Instruction Generation for Accelerated Sparse Matrix-Matrix Multiplication
Qiang Fu,
Thomas B. Rolinger, and
H. Howie Huang
(Advanced Micro Devices, USA; NVIDIA, USA; George Washington University, USA)
@InProceedings{CGO24p561,
author = {Qiang Fu and Thomas B. Rolinger and H. Howie Huang},
title = {JITSPMM: Just-in-Time Instruction Generation for Accelerated Sparse Matrix-Matrix Multiplication},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {561-560},
doi = {},
year = {2024},
}
Article: cgo24main-p81-p doi:
oneDNN Graph Compiler: A Hybrid Approach for High-Performance Deep Learning Compilation
Jianhui Li,
Zhennan Qin,
Yijie Mei,
Jingze Cui,
Yunfei Song,
Ciyong Chen,
Yifei Zhang,
Longsheng Du,
Xianhang Cheng,
Baihui Jin,
Yan Zhang,
Jason Ye,
Eric Lin, and
Dan Lavery
(Intel, USA; Intel, China)
@InProceedings{CGO24p577,
author = {Jianhui Li and Zhennan Qin and Yijie Mei and Jingze Cui and Yunfei Song and Ciyong Chen and Yifei Zhang and Longsheng Du and Xianhang Cheng and Baihui Jin and Yan Zhang and Jason Ye and Eric Lin and Dan Lavery},
title = {oneDNN Graph Compiler: A Hybrid Approach for High-Performance Deep Learning Compilation},
booktitle = {Proc.\ CGO},
publisher = {IEEE},
pages = {577-576},
doi = {},
year = {2024},
}
Artifacts Reusable
Results Reproduced
Article: cgo24main-p120-p doi:
proc time: 0.07