{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:MDDP43SBJWKT4UUAY3SSHQ26U4","short_pith_number":"pith:MDDP43SB","canonical_record":{"source":{"id":"2607.23123","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-25T09:55:31Z","cross_cats_sorted":[],"title_canon_sha256":"105ba434fcb88546f2e0198a253741bbb47ae35292866763fe6b17d6c06cfd9f","abstract_canon_sha256":"1e672c2cbd3c20de4bb43d6039b7db0806f9b005777b24a1b93f9f9d77247294"},"schema_version":"1.0"},"canonical_sha256":"60c6fe6e414d953e5280c6e523c35ea72f4ecf168de2c3e85e7cf48df7c3a007","source":{"kind":"arxiv","id":"2607.23123","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.23123","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"arxiv_version","alias_value":"2607.23123v1","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.23123","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_12","alias_value":"MDDP43SBJWKT","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_16","alias_value":"MDDP43SBJWKT4UUA","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_8","alias_value":"MDDP43SB","created_at":"2026-07-28T01:22:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:MDDP43SBJWKT4UUAY3SSHQ26U4","target":"record","payload":{"canonical_record":{"source":{"id":"2607.23123","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-25T09:55:31Z","cross_cats_sorted":[],"title_canon_sha256":"105ba434fcb88546f2e0198a253741bbb47ae35292866763fe6b17d6c06cfd9f","abstract_canon_sha256":"1e672c2cbd3c20de4bb43d6039b7db0806f9b005777b24a1b93f9f9d77247294"},"schema_version":"1.0"},"canonical_sha256":"60c6fe6e414d953e5280c6e523c35ea72f4ecf168de2c3e85e7cf48df7c3a007","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:07.916279Z","signature_b64":"YTq3po3lHTRqI/J83FPiyLOlqYrYy0Hk266l+yl0PMpw41niigPZCXKlxlsaGv6rNMAlgDxkynnYYoBVs2fPAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"60c6fe6e414d953e5280c6e523c35ea72f4ecf168de2c3e85e7cf48df7c3a007","last_reissued_at":"2026-07-28T01:22:07.915518Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:07.915518Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.23123","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-28T01:22:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"rL0DkoVX3lacJwz7D2G+8HDEKM4mrlkftwEUFR5DAUtZtMxCgRgUtmDwm0QwpDxdoC3njakyc0VNx0ei2L6qAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:42:10.359540Z"},"content_sha256":"a3fb19725becdb79e0422833882db6ceb2af05cda7e61c254a34a37e8a223838","schema_version":"1.0","event_id":"sha256:a3fb19725becdb79e0422833882db6ceb2af05cda7e61c254a34a37e8a223838"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:MDDP43SBJWKT4UUAY3SSHQ26U4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"SQBench: A Benchmark for Evaluating Task Delivery by Language-Model Agents in Production-Oriented Workflows","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Summer Sun (Shaqiu Community)","submitted_at":"2026-07-25T09:55:31Z","abstract_excerpt":"Existing evaluations of large language models cover knowledge, reasoning, coding, and tool use, but they rarely treat a verifiable deliverable produced within a constrained workflow as the unit of evaluation. We introduce SQBench, a benchmark for evaluating production-oriented task delivery by language-model agents. SQBench v1.0 contains 220 standardized tasks organized into L1 atomic capabilities, L2 composite skills, and L3 business scenarios. Each task requires an agent to process input assets, use available tools, and produce an explicitly specified deliverable. The evaluation first comput"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.23123","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.23123/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-28T01:22:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CSoYQCXKoE1fR0e6HUnYSlUeexZtpuBDS10Kgp8FVuSmSaWIiJYMfno0Qz8d0RWaEOn9Nbu/YW0SEO5dMfS3AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:42:10.359919Z"},"content_sha256":"c406a9b9465a65a3eda289358f38a5e6653bd4ff2abfcee3c565436cdd273e32","schema_version":"1.0","event_id":"sha256:c406a9b9465a65a3eda289358f38a5e6653bd4ff2abfcee3c565436cdd273e32"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/bundle.json","state_url":"https://pith.science/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:42:10Z","links":{"resolver":"https://pith.science/pith/MDDP43SBJWKT4UUAY3SSHQ26U4","bundle":"https://pith.science/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/bundle.json","state":"https://pith.science/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MDDP43SBJWKT4UUAY3SSHQ26U4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:MDDP43SBJWKT4UUAY3SSHQ26U4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1e672c2cbd3c20de4bb43d6039b7db0806f9b005777b24a1b93f9f9d77247294","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-25T09:55:31Z","title_canon_sha256":"105ba434fcb88546f2e0198a253741bbb47ae35292866763fe6b17d6c06cfd9f"},"schema_version":"1.0","source":{"id":"2607.23123","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.23123","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"arxiv_version","alias_value":"2607.23123v1","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.23123","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_12","alias_value":"MDDP43SBJWKT","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_16","alias_value":"MDDP43SBJWKT4UUA","created_at":"2026-07-28T01:22:07Z"},{"alias_kind":"pith_short_8","alias_value":"MDDP43SB","created_at":"2026-07-28T01:22:07Z"}],"graph_snapshots":[{"event_id":"sha256:c406a9b9465a65a3eda289358f38a5e6653bd4ff2abfcee3c565436cdd273e32","target":"graph","created_at":"2026-07-28T01:22:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.23123/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Existing evaluations of large language models cover knowledge, reasoning, coding, and tool use, but they rarely treat a verifiable deliverable produced within a constrained workflow as the unit of evaluation. We introduce SQBench, a benchmark for evaluating production-oriented task delivery by language-model agents. SQBench v1.0 contains 220 standardized tasks organized into L1 atomic capabilities, L2 composite skills, and L3 business scenarios. Each task requires an agent to process input assets, use available tools, and produce an explicitly specified deliverable. The evaluation first comput","authors_text":"Summer Sun (Shaqiu Community)","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-25T09:55:31Z","title":"SQBench: A Benchmark for Evaluating Task Delivery by Language-Model Agents in Production-Oriented Workflows"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.23123","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a3fb19725becdb79e0422833882db6ceb2af05cda7e61c254a34a37e8a223838","target":"record","created_at":"2026-07-28T01:22:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1e672c2cbd3c20de4bb43d6039b7db0806f9b005777b24a1b93f9f9d77247294","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-25T09:55:31Z","title_canon_sha256":"105ba434fcb88546f2e0198a253741bbb47ae35292866763fe6b17d6c06cfd9f"},"schema_version":"1.0","source":{"id":"2607.23123","kind":"arxiv","version":1}},"canonical_sha256":"60c6fe6e414d953e5280c6e523c35ea72f4ecf168de2c3e85e7cf48df7c3a007","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"60c6fe6e414d953e5280c6e523c35ea72f4ecf168de2c3e85e7cf48df7c3a007","first_computed_at":"2026-07-28T01:22:07.915518Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-28T01:22:07.915518Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"YTq3po3lHTRqI/J83FPiyLOlqYrYy0Hk266l+yl0PMpw41niigPZCXKlxlsaGv6rNMAlgDxkynnYYoBVs2fPAg==","signature_status":"signed_v1","signed_at":"2026-07-28T01:22:07.916279Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.23123","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a3fb19725becdb79e0422833882db6ceb2af05cda7e61c254a34a37e8a223838","sha256:c406a9b9465a65a3eda289358f38a5e6653bd4ff2abfcee3c565436cdd273e32"],"state_sha256":"79372455be373f91b2cb696ac432f165aa6df812fb29b6e700734e5535b78931"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Hr67PH0qbLkAZ1lsG3weY+OndpxBWacXigJ3yMBBsPsRYa2EoyiHXdyj7yPrF5nXn2uPCckcQI75MGot/fItCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:42:10.362274Z","bundle_sha256":"f5ad715ab3d3b673ecce7551e929803068e051b0d4f51da336e491da5ca42108"}}