@techreport{irtf-nmrg-ai-deploy-03, number = {draft-irtf-nmrg-ai-deploy-03}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-irtf-nmrg-ai-deploy/03/}, author = {Yong-Geun Hong and Joo-Sang Youn and Seung-Woo Hong and Pedro Martinez-Julia and Qin Wu}, title = {{Considerations of network/system for AI services}}, pagetotal = 27, year = 2026, month = jul, day = 6, abstract = {As the development of AI technology has matured and AI technology has begun to be applied in various fields, the execution environment has evolved from dedicated high-performance servers to commodity servers and affordable, small-scale hardware, including microcontrollers, low-performance CPUs, and AI chipsets. This document outlines how to configure the network and system for an AI inference service, providing AI services in a distributed manner. It also outlines the factors to consider when a client connects to a cloud server and an edge device to request an AI service. It describes some use cases for deploying network-based AI services, such as self-driving vehicles and network digital twins.}, }