LPU: A Latency-Optimized and Highly Scalable Processor for Large Language Model Inference. Moon, S., Kim, J., Kim, J., Hong, S., Cha, J., Kim, M., Lim, S., Choi, G., Seo, D., Kim, J., Lee, H., Park, H., Ko, R., Choi, S., Park, J., Lee, J., & Kim, J. CoRR, 2024.
LPU: A Latency-Optimized and Highly Scalable Processor for Large Language Model Inference. [link]Link  LPU: A Latency-Optimized and Highly Scalable Processor for Large Language Model Inference. [link]Paper  bibtex   
@article{journals/corr/abs-2408-07326,
  added-at = {2025-01-14T00:00:00.000+0100},
  author = {Moon, Seungjae and Kim, Jung-Hoon and Kim, Junsoo and Hong, Seongmin and Cha, Junseo and Kim, Minsu and Lim, Sukbin and Choi, Gyubin and Seo, Dongjin and Kim, Jongho and Lee, Hunjong and Park, Hyunjun and Ko, Ryeowook and Choi, Soongyu and Park, Jongse and Lee, Jinwon and Kim, Joo-Young},
  biburl = {https://www.bibsonomy.org/bibtex/2bff5d992e2d644fa6b828a5468c4677b/dblp},
  ee = {https://doi.org/10.48550/arXiv.2408.07326},
  interhash = {65f572bb06d32ffdba0e265d53a82c1f},
  intrahash = {bff5d992e2d644fa6b828a5468c4677b},
  journal = {CoRR},
  keywords = {dblp},
  timestamp = {2025-01-20T07:09:58.000+0100},
  title = {LPU: A Latency-Optimized and Highly Scalable Processor for Large Language Model Inference.},
  url = {http://dblp.uni-trier.de/db/journals/corr/corr2408.html#abs-2408-07326},
  volume = {abs/2408.07326},
  year = 2024
}

Downloads: 0