Step-dLLM: Adaptive Step-aware Sparse Attention for Efficient Diffusion LLM Inference
Zhichen Zeng, Xichong Zhang, Junpan Wu, Yifei Zuo, Chi-Chih Chang, Jiayi Wang, Maohua Nie, Ji Liu, Ang Li, Banghua Zhu
The Fortieth Annual Conference on Neural Information Processing Systems, Dec 2026
MFSA: A Multi-Format Systolic Array with Native Micro Scaling Support
Jiayi Wang, Kearnan Bishop, Ang Li
The 44th IEEE International Conference on Computer Design, Nov 2026
CacheFlex: Direct Software-Managed Access to Higher-Level Cache for Scalable Vector Support
Jingqun Zhang, Maohua Nie, Jiayi Wang, Weihang Li, Rishi Sappidi, Shwet Chitnis, Ang Li
59th IEEE/ACM International Symposium on Microarchitecture, Oct 2026
DICE: Enabling Efficient General-Purpose SIMT Execution with Statically Scheduled Coarse-Grained Reconfigurable Arrays
Jiayi Wang, Darren Lu, Zhichen Zeng, Ang Li
The 53rd IEEE/ACM International Symposium on Computer Architecture, Jun 2026
@article{Wang2026DICE,
author = {Wang, Jiayi and Da Lu, Ang and Zeng, Zhichen and Li, Ang},
doi = {10.48550/ARXIV.2605.05496},
year = {2026},
publisher = {arXiv},
title = {DICE: Enabling {Efficient} {General}-{Purpose} {SIMT} {Execution} with {Statically} {Scheduled} {Coarse}-{Grained} {Reconfigurable} {Arrays}},
url = {https://arxiv.org/abs/2605.05496},
}
DORA: Open-Source Infrastructure for Prototyping Reconfigurable Fabrics
Shwet Chitnis, Jiayi Wang, Fergus Xu, Ayush Kulkarni, Rampranav Navendran, Juwon Jun, Jingqun Zhang, Ang Li
2026 Open-Source Computer Architecture Research, Jun 2026
All-Digital Bluetooth Low Energy (BLE) Backscatter ASIC using Standard I/O Pad Drivers in 180 nm CMOS
Ryan Lee, Kate Tseng, Te Min "Tyoma" Yu, Andrew Pan, Jiayi Wang, James Rosenthal, Kevin J Ho, Ang Li, Matthew Reynolds
2026 IEEE International Conference on RFID, Jun 2026
TransDot: An Area-efficient Reconfigurable Floating-Point Unit for Trans-Precision Dot-Product Accumulation for FPGA AI Engines
Jiayi Wang, Maohua Nie, Sin-Chen Lin, C. -J. Richard Shi, Ang Li
The 34th IEEE International Symposium on Field-Programmable Custom Computing Machines, May 2026
@inproceedings{Wang2026TransDot,
author = {Wang, Jiayi and Nie, Maohua and Lin, Sin-Chen and Shi, C.-J. Richard and Li, Ang},
booktitle = {2026 {IEEE} 34th {Annual} {International} {Symposium} on {Field}-{Programmable} {Custom} {Computing} {Machines} ({FCCM})},
doi = {10.1109/fccm68464.2026.00034},
year = {2026},
month = {may 13},
pages = {171--175},
organization = {IEEE},
title = {TransDot: An {Area}-efficient {Reconfigurable} {Floating}-{Point} {Unit} for {Trans}-{Precision} {Dot}-{Product} {Accumulation} for {FPGA} {AI} {Engines}},
url = {http://dx.doi.org/10.1109/FCCM68464.2026.00034},
}
@article{Wang2026TransDot,
author = {Wang, Jiayi and Nie, Maohua and Lin, Sin-Chen and Shi, C. -J. Richard and Li, Ang},
doi = {10.48550/ARXIV.2605.07245},
year = {2026},
publisher = {arXiv},
title = {TransDot: An {Area}-efficient {Reconfigurable} {Floating}-{Point} {Unit} for {Trans}-{Precision} {Dot}-{Product} {Accumulation} for {FPGA} {AI} {Engines}},
url = {https://arxiv.org/abs/2605.07245},
}
DisagMoE: Computation-Communication overlapped MoE Training via Disaggregated AF-Pipe Parallelism
Zhichen Zeng, Chi-Chih Chang, Jiayi Wang, Zezhou Wang, Ningxin Zheng, Zheng Zhong, Cesar A. Stuardo, Dongyang Wang, Mohamed S. Abdelfattah, Haibin Lin, Banghua Zhu, Ang Li, Ziheng Jiang
@article{Zeng2026DisagMoE,
author = {Zeng, Zhichen and Chang, Chi-Chih and Wang, Jiayi and Wang, Zezhou and Zheng, Ningxin and Zhong, Zheng and Stuardo, Cesar A. and Wang, Dongyang and Abdelfattah, Mohamed S. and Lin, Haibin and Zhu, Banghua and Li, Ang and Jiang, Ziheng},
doi = {10.48550/ARXIV.2605.11005},
year = {2026},
publisher = {arXiv},
title = {DisagMoE: Computation-{Communication} overlapped {MoE} {Training} via {Disaggregated} {AF}-{Pipe} {Parallelism}},
url = {https://arxiv.org/abs/2605.11005},
}
Tactic: Adaptive Sparse Attention with Clustering and Distribution Fitting for Long-Context LLMs
Kan Zhu, Tian Tang, Qinyu Xu, Yile Gu, Zhichen Zeng, Rohan Kadekodi, Liangyu Zhao, Ang Li, Arvind Krishnamurthy, Baris Kasikci
The Fourteenth International Conference on Learning Representations, Apr 2025
@article{Zhu2025Tactic,
author = {Zhu, Kan and Tang, Tian and Xu, Qinyu and Gu, Yile and Zeng, Zhichen and Kadekodi, Rohan and Zhao, Liangyu and Li, Ang and Krishnamurthy, Arvind and Kasikci, Baris},
doi = {10.48550/ARXIV.2502.12216},
year = {2025},
publisher = {arXiv},
title = {Tactic: Adaptive {Sparse} {Attention} with {Clustering} and {Distribution} {Fitting} for {Long}-{Context} {LLMs}},
url = {https://arxiv.org/abs/2502.12216},
}
Local Linear Attention: An Optimal Interpolation of Linear and Softmax Attention For Test-Time Regression
Yifei Zuo, Yutong Yin, Zhichen Zeng, Ang Li, Banghua Zhu, Zhaoran Wang
The Fourteenth International Conference on Learning Representations, Apr 2025
@article{Zuo2025Local,
author = {Zuo, Yifei and Yin, Yutong and Zeng, Zhichen and Li, Ang and Zhu, Banghua and Wang, Zhaoran},
doi = {10.48550/ARXIV.2510.01450},
year = {2025},
publisher = {arXiv},
title = {Local {Linear} {Attention}: An {Optimal} {Interpolation} of {Linear} and {Softmax} {Attention} {For} {Test}-{Time} {Regression}},
url = {https://arxiv.org/abs/2510.01450},
}
Exploring the Performance Improvement of Tensor Processing Engines through Transformation in the Bit-weight Dimension of MACs
Qizhe Wu, Huawen Liang, Yuchen Gui, Zhichen Zeng, Zerong He, Linfeng Tao, Xiaotian Wang, Letian Zhao, Zhaoxi Zeng, Wei Yuan, Wei Wu, Xi Jin
2025 IEEE International Symposium on High-Performance Computer Architecture, Mar 2025
@inproceedings{Wu2025Exploring,
author = {Wu, Qizhe and Liang, Huawen and Gui, Yuchen and Zeng, Zhichen and He, Zerong and Tao, Linfeng and Wang, Xiaotian and Zhao, Letian and Zeng, Zhaoxi and Yuan, Wei and Wu, Wei and Jin, Xi},
booktitle = {2025 {IEEE} {International} {Symposium} on {High} {Performance} {Computer} {Architecture} ({HPCA})},
doi = {10.1109/hpca61900.2025.00058},
year = {2025},
month = {mar 1},
pages = {685--700},
organization = {IEEE},
title = {Exploring the {Performance} {Improvement} of {Tensor} {Processing} {Engines} through {Transformation} in the {Bit}-weight {Dimension} of {MACs}},
url = {http://dx.doi.org/10.1109/HPCA61900.2025.00058},
}
@article{Wu2025Exploring,
author = {Wu, Qizhe and Liang, Huawen and Gui, Yuchen and Zeng, Zhichen and He, Zerong and Tao, Linfeng and Wang, Xiaotian and Zhao, Letian and Zeng, Zhaoxi and Yuan, Wei and Wu, Wei and Jin, Xi},
doi = {10.48550/ARXIV.2503.06342},
year = {2025},
publisher = {arXiv},
title = {Exploring the {Performance} {Improvement} of {Tensor} {Processing} {Engines} through {Transformation} in the {Bit}-weight {Dimension} of {MACs}},
url = {https://arxiv.org/abs/2503.06342},
}
PᴺCEL member
Equal contribution