@techreport{zhang-rtgwg-llmmoe-multicast-02, number = {draft-zhang-rtgwg-llmmoe-multicast-02}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-zhang-rtgwg-llmmoe-multicast/02/}, author = {Zheng Zhang and Wei Duan and Xiaohu Xu and Yisong Liu}, title = {{Multicast use case in LLM MoE}}, pagetotal = 7, year = 2026, month = apr, day = 13, abstract = {Large Language Models (LLMs) have been widely used in recent years. The Mixture of Experts (MoE) architecture is one of the features of LLMs that enables efficient inference and cost-effective training. With the MoE architecture, there are potential multicast use cases such as tokens dispatching. This draft attempts to analyze these use cases.}, }