@techreport{contreras-bmwg-ai-agent-benchmarking-00, number = {draft-contreras-bmwg-ai-agent-benchmarking-00}, type = {Internet-Draft}, institution = {Internet Engineering Task Force}, publisher = {Internet Engineering Task Force}, note = {Work in Progress}, url = {https://datatracker.ietf.org/doc/draft-contreras-bmwg-ai-agent-benchmarking/00/}, author = {Luis M. Contreras}, title = {{Benchmarking Methodology for AI Agents in Network Operations}}, pagetotal = 8, year = 2026, month = jul, day = 6, abstract = {This document defines a benchmarking methodology for evaluating Artificial Intelligence (AI) agents performing network operations tasks such as configuration, troubleshooting, and optimization. This document focuses on task-oriented performance metrics, including task completion success, execution efficiency, and robustness across multi-step workflows, proposing benchmarking practices towards agent- based, closed-loop network operation scenarios. The proposed methodology aims to provide a reproducible and vendor- independent framework to compare AI agent effectiveness in controlled network environments.}, }