@techreport{fu-nmop-tokenops-probelem-statement-00, number = {draft-fu-nmop-tokenops-probelem-statement-00}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-fu-nmop-tokenops-probelem-statement/00/}, author = {Yu Fu and Sun Qiong and Xin Song and Chongfeng Xie}, title = {{Token Operation Problem Statement}}, pagetotal = 10, year = 2026, month = jul, day = 6, abstract = {Distributed LLM inference relies heavily on high-performance networking to synchronize states across accelerators (e.g., GPUs) and nodes. Unlike traditional web services, inference workloads particularly those involving Mixture-of-Experts (MoE) models and long-context windows exhibit unique traffic patterns characterized by massive east-west traffic and strict latency constraints. Current network infrastructures and scheduling methods often treat compute resources and network paths independently, leading to suboptimal performance and degraded Quality of Experience (QoE). This document elaborates on these issues to guide potential protocol enhancements within the IETF.}, }