@techreport{li-tsvwg-inference-transport-00, number = {draft-li-tsvwg-inference-transport-00}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-li-tsvwg-inference-transport/00/}, author = {Zhiqiang Li and Zongpeng Du and Junjie Wang and Wei Cheng and Guoying Zhang and Xun Sun and Chunhao Zhao}, title = {{Transport Considerations for Large-Scale Distributed Inference Networks}}, pagetotal = 9, year = 2026, month = jul, day = 4, abstract = {Large-scale distributed inference systems generate traffic patterns that differ from both traditional data center workloads and distributed training workloads. Disaggregated prefill/decode serving transfers key-value cache state between server pools, and expert- parallel architectures generate all-to-all traffic among expert groups. These flows are typically carried over a small number of RDMA connections, producing low-entropy traffic that is prone to uneven link utilization under Equal-Cost Multipath (ECMP) forwarding. This document specifies transport considerations for such networks, covering path load awareness, path steering through ECMP entropy variation, ordering tolerance at the receiver, and differentiated reliability for data with different loss sensitivity. The discussion builds on existing IETF building blocks; this document does not define new protocol elements.}, }