{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YNDE57DAXA77KMEEM3QDZRNMKR","short_pith_number":"pith:YNDE57DA","schema_version":"1.0","canonical_sha256":"c3464efc60b83ff5308466e03cc5ac547b41995dbdf14c473d2856826d4e9237","source":{"kind":"arxiv","id":"2309.10691","version":3},"attestation_state":"computed","paper":{"title":"MINT: Evaluating LLMs in Multi-turn Interaction with Tools and Language Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hao Peng, Heng Ji, Jiateng Liu, Lifan Yuan, Xingyao Wang, Yangyi Chen, Zihan Wang","submitted_at":"2023-09-19T15:25:42Z","abstract_excerpt":"To solve complex tasks, large language models (LLMs) often require multiple rounds of interactions with the user, sometimes assisted by external tools. However, current evaluation protocols often emphasize benchmark performance with single-turn exchanges, neglecting the nuanced interactions among the user, LLMs, and external tools, while also underestimating the importance of natural language feedback from users. These oversights contribute to discrepancies between research benchmark evaluations and real-world use cases. We introduce MINT, a benchmark that evaluates LLMs' ability to solve task"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.10691","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-19T15:25:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"cf3f89404843380027e7ad4dd376f4cb4143bac52ed473efa7112f58e0f62aa4","abstract_canon_sha256":"96d0a49155bf6b4b61d6848710bb0274eba18da77a2359e2d550b724045b7e10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:13.298721Z","signature_b64":"zBKITMptKC+7sL7mmMy5bCfIvZkTwV7TkfyEr5MprxF6XdS1yGfg/8WFiK2Q1OYm3yhcadLu/clKRQvDm4KVCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3464efc60b83ff5308466e03cc5ac547b41995dbdf14c473d2856826d4e9237","last_reissued_at":"2026-07-05T07:55:13.298360Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:13.298360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MINT: Evaluating LLMs in Multi-turn Interaction with Tools and Language Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hao Peng, Heng Ji, Jiateng Liu, Lifan Yuan, Xingyao Wang, Yangyi Chen, Zihan Wang","submitted_at":"2023-09-19T15:25:42Z","abstract_excerpt":"To solve complex tasks, large language models (LLMs) often require multiple rounds of interactions with the user, sometimes assisted by external tools. However, current evaluation protocols often emphasize benchmark performance with single-turn exchanges, neglecting the nuanced interactions among the user, LLMs, and external tools, while also underestimating the importance of natural language feedback from users. These oversights contribute to discrepancies between research benchmark evaluations and real-world use cases. We introduce MINT, a benchmark that evaluates LLMs' ability to solve task"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.10691","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.10691/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.10691","created_at":"2026-07-05T07:55:13.298414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.10691v3","created_at":"2026-07-05T07:55:13.298414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.10691","created_at":"2026-07-05T07:55:13.298414+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNDE57DAXA77","created_at":"2026-07-05T07:55:13.298414+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNDE57DAXA77KMEE","created_at":"2026-07-05T07:55:13.298414+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNDE57DA","created_at":"2026-07-05T07:55:13.298414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19613","citing_title":"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30774","citing_title":"What Drives Interactive Improvement from Feedback?","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26788","citing_title":"SeDT: Sentence-Transformer Decision-Transformer Conditioning for Multi-Turn Conversation Reliability","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09999","citing_title":"ServeGen: Workload Characterization and Generation of Large Language Model Serving in Production","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16508","citing_title":"The Scaling Laws of Skills in LLM Agent Systems","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18630","citing_title":"SCICONVBENCH: Benchmarking LLMs on Multi-Turn Clarification for Task Formulation in Computational Science","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19185","citing_title":"An Empirical Study of Testing Practices in Open Source AI Agent Frameworks and Agentic Applications","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2511.02230","citing_title":"Continuum: Efficient and Robust Multi-Turn LLM Agent Scheduling with KV Cache Time-to-Live","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10516","citing_title":"Consistency as a Testable Property: Statistical Methods to Evaluate AI Agent Reliability","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23051","citing_title":"Evaluating Temporal Consistency in Multi-Turn Language Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01920","citing_title":"A Language for Describing Agentic LLM Contexts","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00737","citing_title":"To Call or Not to Call: A Framework to Assess and Optimize LLM Tool Calling","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00334","citing_title":"AgentFloor: How Far Up the tool use Ladder Can Small Open-Weight Models Go?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06392","citing_title":"Qualixar OS: A Universal Operating System for AI Agent Orchestration","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR","json":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR.json","graph_json":"https://pith.science/api/pith-number/YNDE57DAXA77KMEEM3QDZRNMKR/graph.json","events_json":"https://pith.science/api/pith-number/YNDE57DAXA77KMEEM3QDZRNMKR/events.json","paper":"https://pith.science/paper/YNDE57DA"},"agent_actions":{"view_html":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR","download_json":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR.json","view_paper":"https://pith.science/paper/YNDE57DA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.10691&json=true","fetch_graph":"https://pith.science/api/pith-number/YNDE57DAXA77KMEEM3QDZRNMKR/graph.json","fetch_events":"https://pith.science/api/pith-number/YNDE57DAXA77KMEEM3QDZRNMKR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR/action/storage_attestation","attest_author":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR/action/author_attestation","sign_citation":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR/action/citation_signature","submit_replication":"https://pith.science/pith/YNDE57DAXA77KMEEM3QDZRNMKR/action/replication_record"}},"created_at":"2026-07-05T07:55:13.298414+00:00","updated_at":"2026-07-05T07:55:13.298414+00:00"}