{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W3VSJS6PV2HWYWWMJDMFR62EZL","short_pith_number":"pith:W3VSJS6P","schema_version":"1.0","canonical_sha256":"b6eb24cbcfae8f6c5acc48d858fb44caeaa055b43269ccfe14388b068b0963e8","source":{"kind":"arxiv","id":"2503.03056","version":1},"attestation_state":"computed","paper":{"title":"A2Perf: Real-World Autonomous Agents Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Austin Huang, Colton Bishop, Ebrahim M. Songhori, Ikechukwu Uchendu, Izzeddin Gur, Jason Jabbour, Jeffrey Ma, Jie Tan, Joel Runevic, Jordan K. Terry, Korneel Van den Berghe, Matthew Stewart, Paige Bailey, Sergio Guadarrama, Srivatsan Krishnan, Vijay Janapa Reddi, Wenjie Jiang","submitted_at":"2025-03-04T23:41:02Z","abstract_excerpt":"Autonomous agents and systems cover a number of application areas, from robotics and digital assistants to combinatorial optimization, all sharing common, unresolved research challenges. It is not sufficient for agents to merely solve a given task; they must generalize to out-of-distribution tasks, perform reliably, and use hardware resources efficiently during training and inference, among other requirements. Several methods, such as reinforcement learning and imitation learning, are commonly used to tackle these problems, each with different trade-offs. However, there is a lack of benchmarki"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.03056","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-04T23:41:02Z","cross_cats_sorted":[],"title_canon_sha256":"0c28589481f3f8984d731370ec158f5df44239050934ccfc10d304d49ca59612","abstract_canon_sha256":"59ef36c587833ae5a92bae8a5dc4195942bb4c177ac1d1541ca70e120e050adf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:37.852576Z","signature_b64":"w7XlohozV+Zb4qh0lJ86+FNCPw5Alhbdl2cKuQvvgkWIWCPucVDAXOoMGHD56nvcAJzE9dxx55Ndlkz88Z1fCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6eb24cbcfae8f6c5acc48d858fb44caeaa055b43269ccfe14388b068b0963e8","last_reissued_at":"2026-07-05T10:24:37.851477Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:37.851477Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A2Perf: Real-World Autonomous Agents Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aleksandra Faust, Austin Huang, Colton Bishop, Ebrahim M. Songhori, Ikechukwu Uchendu, Izzeddin Gur, Jason Jabbour, Jeffrey Ma, Jie Tan, Joel Runevic, Jordan K. Terry, Korneel Van den Berghe, Matthew Stewart, Paige Bailey, Sergio Guadarrama, Srivatsan Krishnan, Vijay Janapa Reddi, Wenjie Jiang","submitted_at":"2025-03-04T23:41:02Z","abstract_excerpt":"Autonomous agents and systems cover a number of application areas, from robotics and digital assistants to combinatorial optimization, all sharing common, unresolved research challenges. It is not sufficient for agents to merely solve a given task; they must generalize to out-of-distribution tasks, perform reliably, and use hardware resources efficiently during training and inference, among other requirements. Several methods, such as reinforcement learning and imitation learning, are commonly used to tackle these problems, each with different trade-offs. However, there is a lack of benchmarki"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.03056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.03056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.03056","created_at":"2026-07-05T10:24:37.851603+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.03056v1","created_at":"2026-07-05T10:24:37.851603+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.03056","created_at":"2026-07-05T10:24:37.851603+00:00"},{"alias_kind":"pith_short_12","alias_value":"W3VSJS6PV2HW","created_at":"2026-07-05T10:24:37.851603+00:00"},{"alias_kind":"pith_short_16","alias_value":"W3VSJS6PV2HWYWWM","created_at":"2026-07-05T10:24:37.851603+00:00"},{"alias_kind":"pith_short_8","alias_value":"W3VSJS6P","created_at":"2026-07-05T10:24:37.851603+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL","json":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL.json","graph_json":"https://pith.science/api/pith-number/W3VSJS6PV2HWYWWMJDMFR62EZL/graph.json","events_json":"https://pith.science/api/pith-number/W3VSJS6PV2HWYWWMJDMFR62EZL/events.json","paper":"https://pith.science/paper/W3VSJS6P"},"agent_actions":{"view_html":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL","download_json":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL.json","view_paper":"https://pith.science/paper/W3VSJS6P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.03056&json=true","fetch_graph":"https://pith.science/api/pith-number/W3VSJS6PV2HWYWWMJDMFR62EZL/graph.json","fetch_events":"https://pith.science/api/pith-number/W3VSJS6PV2HWYWWMJDMFR62EZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL/action/storage_attestation","attest_author":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL/action/author_attestation","sign_citation":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL/action/citation_signature","submit_replication":"https://pith.science/pith/W3VSJS6PV2HWYWWMJDMFR62EZL/action/replication_record"}},"created_at":"2026-07-05T10:24:37.851603+00:00","updated_at":"2026-07-05T10:24:37.851603+00:00"}