{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ANO54TSVLIWZVVDF7H55CNOQ2L","short_pith_number":"pith:ANO54TSV","schema_version":"1.0","canonical_sha256":"035dde4e555a2d9ad465f9fbd135d0d2c0b917ab9683c8f5c609152072653c4e","source":{"kind":"arxiv","id":"2503.08669","version":2},"attestation_state":"computed","paper":{"title":"SOPBench: Evaluating Language Agents at Following Standard Operating Procedures and Constraints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Antonis Antoniades, Chi Wang, Jiangtian Wang, Kaijie Zhu, Nathan Zhang, Shinda Huang, Sirui Zeng, Wenyue Hua, William Yang Wang, Xifeng Yan, Zekun Li","submitted_at":"2025-03-11T17:53:02Z","abstract_excerpt":"As language agents increasingly automate critical tasks, their ability to follow domain-specific standard operating procedures (SOPs), policies, and constraints when taking actions and making tool calls becomes essential yet remains underexplored. To address this gap, we develop an automated evaluation pipeline SOPBench with: (1) executable environments containing 167 tools/functions across seven customer service domains with service-specific SOPs and rule-based verifiers, (2) an automated test generation framework producing over 900 verified test cases, and (3) an automated evaluation framewo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.08669","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-11T17:53:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c036efa630d311699a0fe4d39bf623b6383ac84f5398d28928b7c0c1c5e01172","abstract_canon_sha256":"6709e00b06a5b5c5c20d85024eedcc73b4a97d44f72b3fe6d8b0b5755aa8dd29"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:09.424745Z","signature_b64":"kHDy/xdcCO0WZEVqADQU52dNIgqu4C/XqI+5B7elJLwXGGdolnK5JHX/4CwuMPIcsuqvbWsoHxajniAUiWC/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"035dde4e555a2d9ad465f9fbd135d0d2c0b917ab9683c8f5c609152072653c4e","last_reissued_at":"2026-07-05T11:23:09.424206Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:09.424206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SOPBench: Evaluating Language Agents at Following Standard Operating Procedures and Constraints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Antonis Antoniades, Chi Wang, Jiangtian Wang, Kaijie Zhu, Nathan Zhang, Shinda Huang, Sirui Zeng, Wenyue Hua, William Yang Wang, Xifeng Yan, Zekun Li","submitted_at":"2025-03-11T17:53:02Z","abstract_excerpt":"As language agents increasingly automate critical tasks, their ability to follow domain-specific standard operating procedures (SOPs), policies, and constraints when taking actions and making tool calls becomes essential yet remains underexplored. To address this gap, we develop an automated evaluation pipeline SOPBench with: (1) executable environments containing 167 tools/functions across seven customer service domains with service-specific SOPs and rule-based verifiers, (2) an automated test generation framework producing over 900 verified test cases, and (3) an automated evaluation framewo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.08669","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.08669/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.08669","created_at":"2026-07-05T11:23:09.424272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.08669v2","created_at":"2026-07-05T11:23:09.424272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.08669","created_at":"2026-07-05T11:23:09.424272+00:00"},{"alias_kind":"pith_short_12","alias_value":"ANO54TSVLIWZ","created_at":"2026-07-05T11:23:09.424272+00:00"},{"alias_kind":"pith_short_16","alias_value":"ANO54TSVLIWZVVDF","created_at":"2026-07-05T11:23:09.424272+00:00"},{"alias_kind":"pith_short_8","alias_value":"ANO54TSV","created_at":"2026-07-05T11:23:09.424272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08010","citing_title":"Tool-Making and Self-Evolving LLM Agents in Low-Latency Systems","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2605.19316","citing_title":"A Multi-Agent Framework for Feature-Constrained Difficulty Control in Reading Comprehension Item Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09285","citing_title":"SAGE: A Service Agent Graph-guided Evaluation Benchmark","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L","json":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L.json","graph_json":"https://pith.science/api/pith-number/ANO54TSVLIWZVVDF7H55CNOQ2L/graph.json","events_json":"https://pith.science/api/pith-number/ANO54TSVLIWZVVDF7H55CNOQ2L/events.json","paper":"https://pith.science/paper/ANO54TSV"},"agent_actions":{"view_html":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L","download_json":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L.json","view_paper":"https://pith.science/paper/ANO54TSV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.08669&json=true","fetch_graph":"https://pith.science/api/pith-number/ANO54TSVLIWZVVDF7H55CNOQ2L/graph.json","fetch_events":"https://pith.science/api/pith-number/ANO54TSVLIWZVVDF7H55CNOQ2L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L/action/storage_attestation","attest_author":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L/action/author_attestation","sign_citation":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L/action/citation_signature","submit_replication":"https://pith.science/pith/ANO54TSVLIWZVVDF7H55CNOQ2L/action/replication_record"}},"created_at":"2026-07-05T11:23:09.424272+00:00","updated_at":"2026-07-05T11:23:09.424272+00:00"}