{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LWQZ4J66TNUDHM5EH3ZFOI3EOB","short_pith_number":"pith:LWQZ4J66","schema_version":"1.0","canonical_sha256":"5da19e27de9b6833b3a43ef2572364707403fc061482c2307e3878a789eeb2b2","source":{"kind":"arxiv","id":"2509.03312","version":2},"attestation_state":"computed","paper":{"title":"AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.CL","authors_text":"Guibin Zhang, Junhao Wang, Junjie Chen, Kun Wang, Shuicheng Yan, Wangchunshu Zhou","submitted_at":"2025-09-03T13:42:14Z","abstract_excerpt":"Large Language Model (LLM)-based agentic systems, often comprising multiple models, complex tool invocations, and orchestration protocols, substantially outperform monolithic agents. Yet this very sophistication amplifies their fragility, making them more prone to system failure. Pinpointing the specific agent or step responsible for an error within long execution traces defines the task of agentic system failure attribution. Current state-of-the-art reasoning LLMs, however, remain strikingly inadequate for this challenge, with accuracy generally below 10%. To address this gap, we propose Agen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.03312","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-09-03T13:42:14Z","cross_cats_sorted":["cs.MA"],"title_canon_sha256":"e0caf9ee39fd902bbf36ca670efaffad7a2da1484d7c5172c3642c36f0298fc7","abstract_canon_sha256":"d1577ba920c8c5dc00b137111082faa0101192f369fcda02a1a49f52f26e1591"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:04.272007Z","signature_b64":"JPrMVb+uwoibHPcoUqdUr1yC0390w+gkzw37ane7MJUU6FvCIkFoqEz+Jg+aGAunEjnUtEE11J/g6LT4WJRIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5da19e27de9b6833b3a43ef2572364707403fc061482c2307e3878a789eeb2b2","last_reissued_at":"2026-07-05T12:05:04.271342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:04.271342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AgenTracer: Who Is Inducing Failure in the LLM Agentic Systems?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MA"],"primary_cat":"cs.CL","authors_text":"Guibin Zhang, Junhao Wang, Junjie Chen, Kun Wang, Shuicheng Yan, Wangchunshu Zhou","submitted_at":"2025-09-03T13:42:14Z","abstract_excerpt":"Large Language Model (LLM)-based agentic systems, often comprising multiple models, complex tool invocations, and orchestration protocols, substantially outperform monolithic agents. Yet this very sophistication amplifies their fragility, making them more prone to system failure. Pinpointing the specific agent or step responsible for an error within long execution traces defines the task of agentic system failure attribution. Current state-of-the-art reasoning LLMs, however, remain strikingly inadequate for this challenge, with accuracy generally below 10%. To address this gap, we propose Agen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.03312","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.03312/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.03312","created_at":"2026-07-05T12:05:04.271422+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.03312v2","created_at":"2026-07-05T12:05:04.271422+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.03312","created_at":"2026-07-05T12:05:04.271422+00:00"},{"alias_kind":"pith_short_12","alias_value":"LWQZ4J66TNUD","created_at":"2026-07-05T12:05:04.271422+00:00"},{"alias_kind":"pith_short_16","alias_value":"LWQZ4J66TNUDHM5E","created_at":"2026-07-05T12:05:04.271422+00:00"},{"alias_kind":"pith_short_8","alias_value":"LWQZ4J66","created_at":"2026-07-05T12:05:04.271422+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07989","citing_title":"Who Broke the System? Failure Localization in LLM-Based Multi-Agent Systems","ref_index":97,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06184","citing_title":"What Resolve Rate Hides: Trajectory Structure Diagnostics for Coding Agents","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19659","citing_title":"SAGE-OPD: Selective Agent-Guided Intervention for Multi-Turn On-Policy Distillation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17838","citing_title":"Environment-Grounded Automated Prompt Optimization for LLM Game Agents","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08275","citing_title":"Causal Agent Replay: Counterfactual Attribution for LLM-Agent Failures","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07889","citing_title":"Strained Coherence: A Pre-Failure Signal in Coding Agent Execution Trajectories","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03467","citing_title":"StepFinder: A Temporal Semantic Framework for Failure Attribution in Multi-Agent Systems","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01581","citing_title":"Agent System Operations: Categorization, Challenges, and Future Directions","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09863","citing_title":"From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10913","citing_title":"Shepherd: Enabling Programmable Meta-Agents via Reversible Agentic Execution Traces","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24197","citing_title":"A Sober Look at Agentic Misalignment in Automated Workflows","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25376","citing_title":"KYA: A Framework-Agnostic Trust Layer for Autonomous Systems with Verifiable Provenance and Hierarchical Policy Composition","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26563","citing_title":"TrajAudit: Automated Failure Diagnosis for Agentic Coding Systems","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29790","citing_title":"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00765","citing_title":"FALAT: Tracing Failures in LLM Agent Trajectories via Dependency-Guided Search","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06708","citing_title":"Signal-Driven Observation for Long-Horizon Web Agents","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14133","citing_title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17467","citing_title":"VerifyMAS: Hypothesis Verification for Failure Attribution in LLM Multi-Agent Systems","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19240","citing_title":"CASPIAN: Online Detection and Attribution of Cascade Attacks in LLM Multi-Agent Systems via Cross-Channel Causal Monitoring","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14859","citing_title":"Do Coding Agents Understand Least-Privilege Authorization?","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08715","citing_title":"AgentForesight: Online Auditing for Early Failure Prediction in Multi-Agent Systems","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14133","citing_title":"ClawForge: Generating Executable Interactive Benchmarks for Command-Line Agents","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB","json":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB.json","graph_json":"https://pith.science/api/pith-number/LWQZ4J66TNUDHM5EH3ZFOI3EOB/graph.json","events_json":"https://pith.science/api/pith-number/LWQZ4J66TNUDHM5EH3ZFOI3EOB/events.json","paper":"https://pith.science/paper/LWQZ4J66"},"agent_actions":{"view_html":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB","download_json":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB.json","view_paper":"https://pith.science/paper/LWQZ4J66","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.03312&json=true","fetch_graph":"https://pith.science/api/pith-number/LWQZ4J66TNUDHM5EH3ZFOI3EOB/graph.json","fetch_events":"https://pith.science/api/pith-number/LWQZ4J66TNUDHM5EH3ZFOI3EOB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB/action/storage_attestation","attest_author":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB/action/author_attestation","sign_citation":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB/action/citation_signature","submit_replication":"https://pith.science/pith/LWQZ4J66TNUDHM5EH3ZFOI3EOB/action/replication_record"}},"created_at":"2026-07-05T12:05:04.271422+00:00","updated_at":"2026-07-05T12:05:04.271422+00:00"}