{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5VZKF5CJZGLO45ZSPPSBRBSA4Y","short_pith_number":"pith:5VZKF5CJ","schema_version":"1.0","canonical_sha256":"ed72a2f449c996ee77327be4188640e621ea816f963ce0b12e58e0314368d8e4","source":{"kind":"arxiv","id":"2503.20527","version":1},"attestation_state":"computed","paper":{"title":"StableToolBench-MirrorAPI: Modeling Tool Environments as Mirrors of 7,000+ Real-World APIs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hao Wang, Sicheng Zhou, Sijie Cheng, Wenbing Huang, Yang Liu, Yuchen Niu, Zhicheng Guo","submitted_at":"2025-03-26T13:13:03Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has spurred significant interest in tool learning, where LLMs are augmented with external tools to tackle complex tasks. However, existing tool environments face challenges in balancing stability, scalability, and realness, particularly for benchmarking purposes. To address this problem, we propose MirrorAPI, a novel framework that trains specialized LLMs to accurately simulate real API responses, effectively acting as \"mirrors\" to tool environments. Using a comprehensive dataset of request-response pairs from 7,000+ APIs, we employ supervi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.20527","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-26T13:13:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b7bfb696fabfa02a110577154311cfd98cdd3cf7861e3e87c7335c788f2216e","abstract_canon_sha256":"69daa5522dfb70023645846238747295b910b272ea5d104aa361eff32067a7be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:39:39.665297Z","signature_b64":"7N76NmekSdErNUDYDbi0BWkwYhorft6h78cgW4p6Q6WuG2CsQkt6StlQria6jykdn0vR19+g62QFoGI4tdhOAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed72a2f449c996ee77327be4188640e621ea816f963ce0b12e58e0314368d8e4","last_reissued_at":"2026-07-05T10:39:39.664776Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:39:39.664776Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StableToolBench-MirrorAPI: Modeling Tool Environments as Mirrors of 7,000+ Real-World APIs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hao Wang, Sicheng Zhou, Sijie Cheng, Wenbing Huang, Yang Liu, Yuchen Niu, Zhicheng Guo","submitted_at":"2025-03-26T13:13:03Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has spurred significant interest in tool learning, where LLMs are augmented with external tools to tackle complex tasks. However, existing tool environments face challenges in balancing stability, scalability, and realness, particularly for benchmarking purposes. To address this problem, we propose MirrorAPI, a novel framework that trains specialized LLMs to accurately simulate real API responses, effectively acting as \"mirrors\" to tool environments. Using a comprehensive dataset of request-response pairs from 7,000+ APIs, we employ supervi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.20527","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.20527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.20527","created_at":"2026-07-05T10:39:39.664836+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.20527v1","created_at":"2026-07-05T10:39:39.664836+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.20527","created_at":"2026-07-05T10:39:39.664836+00:00"},{"alias_kind":"pith_short_12","alias_value":"5VZKF5CJZGLO","created_at":"2026-07-05T10:39:39.664836+00:00"},{"alias_kind":"pith_short_16","alias_value":"5VZKF5CJZGLO45ZS","created_at":"2026-07-05T10:39:39.664836+00:00"},{"alias_kind":"pith_short_8","alias_value":"5VZKF5CJ","created_at":"2026-07-05T10:39:39.664836+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04098","citing_title":"TextAtari: 100K Frames Game Playing with Language Agents","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y","json":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y.json","graph_json":"https://pith.science/api/pith-number/5VZKF5CJZGLO45ZSPPSBRBSA4Y/graph.json","events_json":"https://pith.science/api/pith-number/5VZKF5CJZGLO45ZSPPSBRBSA4Y/events.json","paper":"https://pith.science/paper/5VZKF5CJ"},"agent_actions":{"view_html":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y","download_json":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y.json","view_paper":"https://pith.science/paper/5VZKF5CJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.20527&json=true","fetch_graph":"https://pith.science/api/pith-number/5VZKF5CJZGLO45ZSPPSBRBSA4Y/graph.json","fetch_events":"https://pith.science/api/pith-number/5VZKF5CJZGLO45ZSPPSBRBSA4Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y/action/storage_attestation","attest_author":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y/action/author_attestation","sign_citation":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y/action/citation_signature","submit_replication":"https://pith.science/pith/5VZKF5CJZGLO45ZSPPSBRBSA4Y/action/replication_record"}},"created_at":"2026-07-05T10:39:39.664836+00:00","updated_at":"2026-07-05T10:39:39.664836+00:00"}