{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SSTT5BQ32E6ZWK73P5F5CUIF6Q","short_pith_number":"pith:SSTT5BQ3","schema_version":"1.0","canonical_sha256":"94a73e861bd13d9b2bfb7f4bd15105f4213e868969059873331adfd6978884b0","source":{"kind":"arxiv","id":"2505.08638","version":3},"attestation_state":"computed","paper":{"title":"TRAIL: Trace Reasoning and Agentic Issue Localization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anand Kannappan, Darshan Deshpande, Hersh Mehta, Jitin Krishnan, Rebecca Qian, Varun Gangal","submitted_at":"2025-05-13T14:55:31Z","abstract_excerpt":"The increasing adoption of agentic workflows across diverse domains brings a critical need to scalably and systematically evaluate the complex traces these systems generate. Current evaluation methods depend on manual, domain-specific human analysis of lengthy workflow traces - an approach that does not scale with the growing complexity and volume of agentic outputs. Error analysis in these settings is further complicated by the interplay of external tool outputs and language model reasoning, making it more challenging than traditional software debugging. In this work, we (1) articulate the ne"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.08638","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-13T14:55:31Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e5ae45d715fa9ed363238ef7c40ae23a56e39ac6e981e814cc024b0c1de2db39","abstract_canon_sha256":"2773a0979ae0f09e2222bb8f6afdb22bb179bbf4d5979d62b551341ebd8a14fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:04.312059Z","signature_b64":"x9OJY1tlfu2J/I+fFMDonwXAuGZMVX+TH1SSQ4ly1N9aWkHYYwe40aHFOYGa831jFcuVE0UbuhC/Xm8C7Ew7Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"94a73e861bd13d9b2bfb7f4bd15105f4213e868969059873331adfd6978884b0","last_reissued_at":"2026-07-05T11:26:04.311583Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:04.311583Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TRAIL: Trace Reasoning and Agentic Issue Localization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anand Kannappan, Darshan Deshpande, Hersh Mehta, Jitin Krishnan, Rebecca Qian, Varun Gangal","submitted_at":"2025-05-13T14:55:31Z","abstract_excerpt":"The increasing adoption of agentic workflows across diverse domains brings a critical need to scalably and systematically evaluate the complex traces these systems generate. Current evaluation methods depend on manual, domain-specific human analysis of lengthy workflow traces - an approach that does not scale with the growing complexity and volume of agentic outputs. Error analysis in these settings is further complicated by the interplay of external tool outputs and language model reasoning, making it more challenging than traditional software debugging. In this work, we (1) articulate the ne"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.08638","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.08638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.08638","created_at":"2026-07-05T11:26:04.311642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.08638v3","created_at":"2026-07-05T11:26:04.311642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.08638","created_at":"2026-07-05T11:26:04.311642+00:00"},{"alias_kind":"pith_short_12","alias_value":"SSTT5BQ32E6Z","created_at":"2026-07-05T11:26:04.311642+00:00"},{"alias_kind":"pith_short_16","alias_value":"SSTT5BQ32E6ZWK73","created_at":"2026-07-05T11:26:04.311642+00:00"},{"alias_kind":"pith_short_8","alias_value":"SSTT5BQ3","created_at":"2026-07-05T11:26:04.311642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06184","citing_title":"What Resolve Rate Hides: Trajectory Structure Diagnostics for Coding Agents","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24626","citing_title":"SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21627","citing_title":"Counsel: A Meta-Evaluation Dataset for Agentic Tasks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01760","citing_title":"Refploit: Facilitating Exploit Construction via Code-Agent Trajectory Repair","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07889","citing_title":"Strained Coherence: A Pre-Failure Signal in Coding Agent Execution Trajectories","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04990","citing_title":"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03467","citing_title":"StepFinder: A Temporal Semantic Framework for Failure Attribution in Multi-Agent Systems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01581","citing_title":"Agent System Operations: Categorization, Challenges, and Future Directions","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27409","citing_title":"Delayed Verification Destabilizes Multi-Agent LLM Belief: Instability Thresholds and Optimal Corrector Placement","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04990","citing_title":"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26563","citing_title":"TrajAudit: Automated Failure Diagnosis for Agentic Coding Systems","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22608","citing_title":"Agentic CLEAR: Automating Multi-Level Evaluation of LLM Agents","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21347","citing_title":"Insights Generator: Systematic Corpus-Level Trace Diagnostics for LLM Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14892","citing_title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02393","citing_title":"Process-Centric Analysis of Agentic Software Systems","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15232","citing_title":"When Agents Fail: A Comprehensive Study of Bugs in LLM Agents with Automated Labeling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14865","citing_title":"Holistic Evaluation and Failure Diagnosis of AI Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14892","citing_title":"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12925","citing_title":"AgentLens: Revealing The Lucky Pass Problem in SWE-Agent Evaluation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13334","citing_title":"A Survey of Context Engineering for Large Language Models","ref_index":220,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11225","citing_title":"PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23455","citing_title":"CUJBench: Benchmarking LLM-Agent on Cross-Modal Failure Diagnosis from Browser to Backend","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18240","citing_title":"AJ-Bench: Benchmarking Agent-as-a-Judge for Environment-Aware Evaluation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q","json":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q.json","graph_json":"https://pith.science/api/pith-number/SSTT5BQ32E6ZWK73P5F5CUIF6Q/graph.json","events_json":"https://pith.science/api/pith-number/SSTT5BQ32E6ZWK73P5F5CUIF6Q/events.json","paper":"https://pith.science/paper/SSTT5BQ3"},"agent_actions":{"view_html":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q","download_json":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q.json","view_paper":"https://pith.science/paper/SSTT5BQ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.08638&json=true","fetch_graph":"https://pith.science/api/pith-number/SSTT5BQ32E6ZWK73P5F5CUIF6Q/graph.json","fetch_events":"https://pith.science/api/pith-number/SSTT5BQ32E6ZWK73P5F5CUIF6Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q/action/storage_attestation","attest_author":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q/action/author_attestation","sign_citation":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q/action/citation_signature","submit_replication":"https://pith.science/pith/SSTT5BQ32E6ZWK73P5F5CUIF6Q/action/replication_record"}},"created_at":"2026-07-05T11:26:04.311642+00:00","updated_at":"2026-07-05T11:26:04.311642+00:00"}