@misc{indiciae5580db1628a3, title = {Pruning Attention Heads of Transformer Models Using A* Search: A Novel Approach to Compress Big NLP Architectures}, author = {Archit Parnami and Rahul Singh and Tarun Joshi}, year = {2021}, url = {https://arxiv.org/abs/2110.15225}, note = {Source identifier: 2110.15225} }