%% You should probably cite draft-calabria-bmwg-ai-fabric-inference-bench-04 instead of this revision. @techreport{calabria-bmwg-ai-fabric-inference-bench-03, number = {draft-calabria-bmwg-ai-fabric-inference-bench-03}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-calabria-bmwg-ai-fabric-inference-bench/03/}, author = {Fernando Calabria and Carlos Pignataro and Qin Wu and Giuseppe Fioccola and Sowjanya Reddy}, title = {{Benchmarking Methodology for AI Inference Serving Network Fabrics}}, pagetotal = 46, year = , month = , day = , abstract = {This document defines benchmarking terminology, methodologies, and Key Performance Indicators (KPIs) for evaluating Ethernet-based AI inference serving network fabrics. As Large Language Model (LLM) inference deployments scale to disaggregated prefill/decode architectures spanning hundreds or thousands of accelerators (GPUs/ XPUs), the interconnect fabric determines Time to First Token (TTFT), Inter-Token Latency (ITL), and aggregate throughput in tokens per second (TPS). This document establishes vendor-independent, reproducible test procedures for benchmarking fabric-level performance under realistic AI inference workloads. Coverage includes RDMA-based KV cache transfer between disaggregated prefill and decode workers, Mixture-of-Experts (MoE) expert parallelism AllToAll communication, request routing and load balancing for inference serving, congestion management under bursty inference traffic patterns, and scale/soak testing. The methodology enables direct comparison across NIC transport stacks (RoCEv2 and UET) and fabric architectures. This document is a companion to the AI training fabric benchmarking methodology, which addresses training workloads.}, }