{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N5BEKJMHTXP72R3UPIZIQNIABG","short_pith_number":"pith:N5BEKJMH","schema_version":"1.0","canonical_sha256":"6f424525879ddffd47747a3288350009ad43dac2a097a2d000f38fb06d9f6134","source":{"kind":"arxiv","id":"2508.13143","version":1},"attestation_state":"computed","paper":{"title":"Exploring Autonomous Agents: A Closer Look at Why They Fail When Completing Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.AI","authors_text":"Ruofan Lu, Yichen Li, Yintong Huo","submitted_at":"2025-08-18T17:55:22Z","abstract_excerpt":"Autonomous agent systems powered by Large Language Models (LLMs) have demonstrated promising capabilities in automating complex tasks. However, current evaluations largely rely on success rates without systematically analyzing the interactions, communication mechanisms, and failure causes within these systems. To bridge this gap, we present a benchmark of 34 representative programmable tasks designed to rigorously assess autonomous agents. Using this benchmark, we evaluate three popular open-source agent frameworks combined with two LLM backbones, observing a task completion rate of approximat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.13143","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-08-18T17:55:22Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"96adb3c3bd5b26bb84c74583b2a609f63f96de0eeabda42ba8615214bb199689","abstract_canon_sha256":"54df93382f8d91c8a4dea9751c4a6092c5de577a6c14c8977136095c590b3c34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:33.034402Z","signature_b64":"OFtoiMOxtr0CdhgLTtg08kZ/Fk9/m3iuuvr9IUjyg62NaUMmNPV7y+LMZHPifUd5j1j20Xo7Ya/q6Wt6TaGDCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f424525879ddffd47747a3288350009ad43dac2a097a2d000f38fb06d9f6134","last_reissued_at":"2026-07-05T11:55:33.033880Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:33.033880Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Autonomous Agents: A Closer Look at Why They Fail When Completing Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.AI","authors_text":"Ruofan Lu, Yichen Li, Yintong Huo","submitted_at":"2025-08-18T17:55:22Z","abstract_excerpt":"Autonomous agent systems powered by Large Language Models (LLMs) have demonstrated promising capabilities in automating complex tasks. However, current evaluations largely rely on success rates without systematically analyzing the interactions, communication mechanisms, and failure causes within these systems. To bridge this gap, we present a benchmark of 34 representative programmable tasks designed to rigorously assess autonomous agents. Using this benchmark, we evaluate three popular open-source agent frameworks combined with two LLM backbones, observing a task completion rate of approximat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.13143","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.13143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.13143","created_at":"2026-07-05T11:55:33.033945+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.13143v1","created_at":"2026-07-05T11:55:33.033945+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.13143","created_at":"2026-07-05T11:55:33.033945+00:00"},{"alias_kind":"pith_short_12","alias_value":"N5BEKJMHTXP7","created_at":"2026-07-05T11:55:33.033945+00:00"},{"alias_kind":"pith_short_16","alias_value":"N5BEKJMHTXP72R3U","created_at":"2026-07-05T11:55:33.033945+00:00"},{"alias_kind":"pith_short_8","alias_value":"N5BEKJMH","created_at":"2026-07-05T11:55:33.033945+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23414","citing_title":"When Planning Fails Despite Correct Execution: On Epistemic Calibration for LLM-Based Multi-Agent Systems","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06847","citing_title":"Characterizing Faults in Agentic AI: A Taxonomy of Types, Symptoms, and Root Causes","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04131","citing_title":"Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05701","citing_title":"Inference-Time Budget Control for LLM Search Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21255","citing_title":"When Agents Look the Same: Quantifying Distillation-Induced Similarity in Tool-Use Behaviors","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG","json":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG.json","graph_json":"https://pith.science/api/pith-number/N5BEKJMHTXP72R3UPIZIQNIABG/graph.json","events_json":"https://pith.science/api/pith-number/N5BEKJMHTXP72R3UPIZIQNIABG/events.json","paper":"https://pith.science/paper/N5BEKJMH"},"agent_actions":{"view_html":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG","download_json":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG.json","view_paper":"https://pith.science/paper/N5BEKJMH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.13143&json=true","fetch_graph":"https://pith.science/api/pith-number/N5BEKJMHTXP72R3UPIZIQNIABG/graph.json","fetch_events":"https://pith.science/api/pith-number/N5BEKJMHTXP72R3UPIZIQNIABG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG/action/storage_attestation","attest_author":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG/action/author_attestation","sign_citation":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG/action/citation_signature","submit_replication":"https://pith.science/pith/N5BEKJMHTXP72R3UPIZIQNIABG/action/replication_record"}},"created_at":"2026-07-05T11:55:33.033945+00:00","updated_at":"2026-07-05T11:55:33.033945+00:00"}