@inproceedings{26dddae50d5a4610bdaced3e9fdb1c89,
title = "τ-BENCH: A BENCHMARK FOR TOOL-AGENT-USER INTERACTION IN REAL-WORLD DOMAINS",
abstract = "Existing benchmarks do not test language agents on their interaction with human users or ability to follow domain-specific rules, both of which are vital for deploying them in real world applications. We propose τ-bench, a benchmark emulating dynamic conversations between a user (simulated by language models) and a language agent provided with domain-specific API tools and policy guidelines. We employ an efficient and faithful evaluation process that compares the database state at the end of a conversation with the annotated goal state. We also propose a new metric (pasŝk) to evaluate the reliability of agent behavior over multiple trials. Our experiments show that even state-of-the-art function calling agents (like gpt-4o) succeed on < 50\% of the tasks, and are quite inconsistent (pasŝ8 < 25\% in retail). Our findings point to the need for methods that can improve the ability of agents to act consistently and follow rules reliably.",
author = "Shunyu Yao and Noah Shinn and Pedram Razavi and Karthik Narasimhan",
note = "Publisher Copyright: {\textcopyright} 2025 13th International Conference on Learning Representations, ICLR 2025. All rights reserved.; 13th International Conference on Learning Representations, ICLR 2025 ; Conference date: 24-04-2025 Through 28-04-2025",
year = "2025",
language = "English (US)",
series = "13th International Conference on Learning Representations, ICLR 2025",
publisher = "International Conference on Learning Representations, ICLR",
pages = "74824--74876",
booktitle = "13th International Conference on Learning Representations, ICLR 2025",
}