{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","short_pith_number":"pith:MX7CKMOA","canonical_record":{"source":{"id":"2605.13625","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-13T14:52:40Z","cross_cats_sorted":[],"title_canon_sha256":"24f750ded8e0ac1b824ce62503a6fc18b9627aa001265b439c94738eae709819","abstract_canon_sha256":"6c023ff3ab7fcba1a82b2fb3cd5c1b7b20d8f34a480d8e4a8a1e563d64253f0d"},"schema_version":"1.0"},"canonical_sha256":"65fe2531c0fdeb42f3ff6a19a404f3121e84d0c863f372fb86336848a651dc2f","source":{"kind":"arxiv","id":"2605.13625","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.13625","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"arxiv_version","alias_value":"2605.13625v1","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.13625","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"pith_short_12","alias_value":"MX7CKMOA7XVU","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_16","alias_value":"MX7CKMOA7XVUF477","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_8","alias_value":"MX7CKMOA","created_at":"2026-05-18T12:33:37Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","target":"record","payload":{"canonical_record":{"source":{"id":"2605.13625","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-13T14:52:40Z","cross_cats_sorted":[],"title_canon_sha256":"24f750ded8e0ac1b824ce62503a6fc18b9627aa001265b439c94738eae709819","abstract_canon_sha256":"6c023ff3ab7fcba1a82b2fb3cd5c1b7b20d8f34a480d8e4a8a1e563d64253f0d"},"schema_version":"1.0"},"canonical_sha256":"65fe2531c0fdeb42f3ff6a19a404f3121e84d0c863f372fb86336848a651dc2f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T02:44:17.828370Z","signature_b64":"IgsJIKthF22JhioSsxzs5Q6nYS3w3IK51ZB+2GyM61u0QYO13HfzP6g9nVb0wVUeHhKP41dysvBpWMSky91LCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65fe2531c0fdeb42f3ff6a19a404f3121e84d0c863f372fb86336848a651dc2f","last_reissued_at":"2026-05-18T02:44:17.827927Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T02:44:17.827927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.13625","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T02:44:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KpZkstDKQshHoHuyvKue/7GOL1CEwNBlOZrDPBXkAo9GhcVt0tLOPEe4cJD/ZRpIa4CImUcrkz/fCrQ1JJE3Ag==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:00:00.402696Z"},"content_sha256":"a559206b772bbdb7205fe1ba142b5f7bcb95747fa192b6ed5609c92001e0e7ca","schema_version":"1.0","event_id":"sha256:a559206b772bbdb7205fe1ba142b5f7bcb95747fa192b6ed5609c92001e0e7ca"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"How to Interpret Agent Behavior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Daniel Khashabi, Heyuan Huang, Jen-tse Huang, Jie Gao, Kaiser Sun, Katherine Van Koevering, Mark Dredze, Sijie Ji, Weiyan Shi, Zhuoran Lu, Ziang Xiao","submitted_at":"2026-05-13T14:52:40Z","abstract_excerpt":"Autonomous agents such as Claude Code and Codex now operate for hours or even days. Understanding their runtime behavior has become critical for downstream tasks such as diagnosing inefficiencies, fixing bugs, and ensuring better oversight. A primary way to gain this understanding is analyzing the reasoning trajectories and execution traces these agents generate. Yet such data remains in unstructured natural-language form, making it difficult for humans to interpret at scale. We introduce ACT*ONOMY (a combination of Action and Taxonomy), a taxonomy for describing and analyzing agent behavior a"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our experiments show that ACTONOMY can compare behavioral profiles across agents and characterize a single agent's behavior across diverse trajectories, surfacing patterns indicative of failure modes.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That a taxonomy developed via Grounded Theory on a limited set of agent traces will remain comprehensive and unbiased when applied to new agents, tasks, and longer trajectories without substantial revision.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"ACT*ONOMY is a Grounded-Theory-derived hierarchical taxonomy and open repository that enables systematic comparison and characterization of autonomous agent behavior across trajectories.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"406de0a2f67dd711febcdd99034686130e4584150a168a40b9f9476aeefa08dd"},"source":{"id":"2605.13625","kind":"arxiv","version":1},"verdict":{"id":"d2162ca1-04fd-46c8-a57a-eaa122cb4f38","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-14T18:19:50.443478Z","strongest_claim":"Our experiments show that ACTONOMY can compare behavioral profiles across agents and characterize a single agent's behavior across diverse trajectories, surfacing patterns indicative of failure modes.","one_line_summary":"ACT*ONOMY is a Grounded-Theory-derived hierarchical taxonomy and open repository that enables systematic comparison and characterization of autonomous agent behavior across trajectories.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That a taxonomy developed via Grounded Theory on a limited set of agent traces will remain comprehensive and unbiased when applied to new agents, tasks, and longer trajectories without substantial revision.","pith_extraction_headline":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation."},"references":{"count":64,"sample":[{"doi":"","year":2003,"title":"J. R. Anderson and C. Lebiere. The newell test for a theory of cognition.Behavioral and brain Sciences, 26(5):587–601, 2003","work_id":"10176ab3-3c6a-4cb3-968c-da1174150e98","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.1162/coli","year":2022,"title":"Computational Linguistics , volume =","work_id":"816c1e36-4060-4ea8-8f0b-b8c0efaf2db9","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"10.5281/zenodo.18842011","year":2026,"title":"V . P. Bhardwaj. Agentassay: Token-efficient regression testing for non-deterministic ai agent workflows, 2026. URLhttps://zenodo.org/doi/10.5281/zenodo.18842011","work_id":"5c7b704d-3c23-40d1-b377-386a95b3e68f","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2022,"title":"Measuring Progress on Scalable Oversight for Large Language Models","work_id":"e7f92eb1-2050-4e60-bc27-82c94d7694c5","ref_index":4,"cited_arxiv_id":"2211.03540","is_internal_anchor":true},{"doi":"","year":2006,"title":"V . Braun and V . Clarke. Using thematic analysis in psychology.Qualitative research in psy- chology, 3(2):77–101, 2006","work_id":"2446f94f-3bcd-49d5-8fa5-0934610493d7","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":64,"snapshot_sha256":"5d4c0289669aa201a2d040e011966dcd5b9765be347d3e4d17adf7e5179ddffc","internal_anchors":13},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"d2162ca1-04fd-46c8-a57a-eaa122cb4f38"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T02:44:17Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"owg7DNpugmCjrv0KkuIvLBsvOkBWcKr8w4AlpRvI9mDz7UtkyivV8AhG7yaPfmRghPVM2x6/jNzCZYLjiEb0CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:00:00.403789Z"},"content_sha256":"7e9b22f78ff4a740a7fd32cdaba6300ce1da69dab8711f4788f8f45d94982106","schema_version":"1.0","event_id":"sha256:7e9b22f78ff4a740a7fd32cdaba6300ce1da69dab8711f4788f8f45d94982106"},{"event_type":"integrity_finding","subject_pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","target":"integrity","payload":{"note":"DOI in the printed bibliography is fragmented by whitespace or line breaks. A longer candidate (10.1145/3797084.URLhttps://arxiv.org/abs/2509.23586) was visible in the surrounding text but could not be confirmed against doi.org as printed.","snippet":"Y .-A. Xiao, P. Gao, C. Peng, and Y . Xiong. Reducing cost of llm agents with trajectory reduc- tion. 2026. doi: https://doi.org/10.1145/3797084. URLhttps://arxiv.org/abs/2509.2 3586","arxiv_id":"2605.13625","detector":"doi_compliance","evidence":{"ref_index":58,"verdict_class":"incontrovertible","resolved_title":null,"printed_excerpt":"Y .-A. Xiao, P. Gao, C. Peng, and Y . Xiong. Reducing cost of llm agents with trajectory reduc- tion. 2026. doi: https://doi.org/10.1145/3797084. URLhttps://arxiv.org/abs/2509.2 3586","reconstructed_doi":"10.1145/3797084.URLhttps://arxiv.org/abs/2509.23586"},"severity":"advisory","ref_index":58,"audited_at":"2026-05-19T06:20:49.956246Z","event_type":"pith.integrity.v1","detected_doi":"10.1145/3797084.URLhttps://arxiv.org/abs/2509.23586","detector_url":"https://pith.science/pith-integrity-protocol#doi_compliance","external_url":null,"finding_type":"recoverable_identifier","evidence_hash":"b38fbc965e035fc97b0811622d0f2b2b33799be0074e02b0f5981b85fc8c2784","paper_version":1,"verdict_class":"incontrovertible","resolved_title":null,"detector_version":"1.0.0","detected_arxiv_id":null,"integrity_event_id":123,"payload_sha256":"ed3733cc801fe51dffb5b25856c6ec3e4c60d19ea44b17d8bed05092dff21328","signature_b64":"zF24L1pdycn3JSSHZ+PEKans812+SJtoQOBrCWCpVALUQLl5/sJEkL926q+iGQgq35Rgp4uqsyokrSBJ8yokDQ==","signing_key_id":"pith-v1-2026-05"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-19T06:21:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CrWDdeYXY7lpLn0ql9lknNyaA3Zx9jCP2QQd0fqshZBRIAzgjrPV4t4oW6r4owzFTsTEWOOHXnfowSgTb1bnAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:00:00.404954Z"},"content_sha256":"8966ab28e6460e847b3c13b4e5deed4a1bd69e4590ef9df361d42a0e4681b1dc","schema_version":"1.0","event_id":"sha256:8966ab28e6460e847b3c13b4e5deed4a1bd69e4590ef9df361d42a0e4681b1dc"},{"event_type":"integrity_finding","subject_pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","target":"integrity","payload":{"note":"DOI in the printed bibliography is fragmented by whitespace or line breaks. A longer candidate (10.1162/colia) was visible in the surrounding text but could not be confirmed against doi.org as printed.","snippet":"Y . Belinkov. Probing classifiers: Promises, shortcomings, and advances.Computational Linguistics, 48(1):207–219, Mar. 2022. doi: 10.1162/coli a 00422. URLhttps: //aclanthology.org/2022.cl-1.7/","arxiv_id":"2605.13625","detector":"doi_compliance","evidence":{"ref_index":2,"verdict_class":"incontrovertible","resolved_title":null,"printed_excerpt":"Y . Belinkov. Probing classifiers: Promises, shortcomings, and advances.Computational Linguistics, 48(1):207–219, Mar. 2022. doi: 10.1162/coli a 00422. URLhttps: //aclanthology.org/2022.cl-1.7/","reconstructed_doi":"10.1162/colia"},"severity":"advisory","ref_index":2,"audited_at":"2026-05-19T06:20:49.956246Z","event_type":"pith.integrity.v1","detected_doi":"10.1162/colia","detector_url":"https://pith.science/pith-integrity-protocol#doi_compliance","external_url":null,"finding_type":"recoverable_identifier","evidence_hash":"f5789cba0286308ea2c1194e6069cd372282dedc0954fd6e4b85eeaaee5afbae","paper_version":1,"verdict_class":"incontrovertible","resolved_title":null,"detector_version":"1.0.0","detected_arxiv_id":null,"integrity_event_id":122,"payload_sha256":"f43404a4f7bb58be1b16c6f09930f00668c87dc0c546332b32c8c275cd1994b4","signature_b64":"Dwot+S2GFZujfLlk2y40Elp6x8oRPGcv9g10wmsZ2nyCn6T/D0K4QdyylMaoUahDnwqX2OANfydD2DaIAgeeCQ==","signing_key_id":"pith-v1-2026-05"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-19T06:21:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"H8R6UHDpkDapHBqSv/RluQePE4PZgRLLNSJ9pqqwQKalQigYe+BVoXtP+4sYdYAz4E5yWFZ/OlmkrZtGpFbyAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-25T04:00:00.405358Z"},"content_sha256":"0b42e10d7187c6ef7d0cc54fa3177b83fac7aee9c5c48a1f00babec544d03196","schema_version":"1.0","event_id":"sha256:0b42e10d7187c6ef7d0cc54fa3177b83fac7aee9c5c48a1f00babec544d03196"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MX7CKMOA7XVUF477NIM2IBHTCI/bundle.json","state_url":"https://pith.science/pith/MX7CKMOA7XVUF477NIM2IBHTCI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MX7CKMOA7XVUF477NIM2IBHTCI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-25T04:00:00Z","links":{"resolver":"https://pith.science/pith/MX7CKMOA7XVUF477NIM2IBHTCI","bundle":"https://pith.science/pith/MX7CKMOA7XVUF477NIM2IBHTCI/bundle.json","state":"https://pith.science/pith/MX7CKMOA7XVUF477NIM2IBHTCI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MX7CKMOA7XVUF477NIM2IBHTCI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:MX7CKMOA7XVUF477NIM2IBHTCI","merge_version":"pith-open-graph-merge-v1","event_count":4,"valid_event_count":4,"invalid_event_count":0,"equivocation_count":1,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6c023ff3ab7fcba1a82b2fb3cd5c1b7b20d8f34a480d8e4a8a1e563d64253f0d","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-13T14:52:40Z","title_canon_sha256":"24f750ded8e0ac1b824ce62503a6fc18b9627aa001265b439c94738eae709819"},"schema_version":"1.0","source":{"id":"2605.13625","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.13625","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"arxiv_version","alias_value":"2605.13625v1","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.13625","created_at":"2026-05-18T02:44:17Z"},{"alias_kind":"pith_short_12","alias_value":"MX7CKMOA7XVU","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_16","alias_value":"MX7CKMOA7XVUF477","created_at":"2026-05-18T12:33:37Z"},{"alias_kind":"pith_short_8","alias_value":"MX7CKMOA","created_at":"2026-05-18T12:33:37Z"}],"graph_snapshots":[{"event_id":"sha256:7e9b22f78ff4a740a7fd32cdaba6300ce1da69dab8711f4788f8f45d94982106","target":"graph","created_at":"2026-05-18T02:44:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Our experiments show that ACTONOMY can compare behavioral profiles across agents and characterize a single agent's behavior across diverse trajectories, surfacing patterns indicative of failure modes."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That a taxonomy developed via Grounded Theory on a limited set of agent traces will remain comprehensive and unbiased when applied to new agents, tasks, and longer trajectories without substantial revision."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"ACT*ONOMY is a Grounded-Theory-derived hierarchical taxonomy and open repository that enables systematic comparison and characterization of autonomous agent behavior across trajectories."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation."}],"snapshot_sha256":"406de0a2f67dd711febcdd99034686130e4584150a168a40b9f9476aeefa08dd"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Autonomous agents such as Claude Code and Codex now operate for hours or even days. Understanding their runtime behavior has become critical for downstream tasks such as diagnosing inefficiencies, fixing bugs, and ensuring better oversight. A primary way to gain this understanding is analyzing the reasoning trajectories and execution traces these agents generate. Yet such data remains in unstructured natural-language form, making it difficult for humans to interpret at scale. We introduce ACT*ONOMY (a combination of Action and Taxonomy), a taxonomy for describing and analyzing agent behavior a","authors_text":"Daniel Khashabi, Heyuan Huang, Jen-tse Huang, Jie Gao, Kaiser Sun, Katherine Van Koevering, Mark Dredze, Sijie Ji, Weiyan Shi, Zhuoran Lu, Ziang Xiao","cross_cats":[],"headline":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-13T14:52:40Z","title":"How to Interpret Agent Behavior"},"references":{"count":64,"internal_anchors":13,"resolved_work":64,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"J. R. Anderson and C. Lebiere. The newell test for a theory of cognition.Behavioral and brain Sciences, 26(5):587–601, 2003","work_id":"10176ab3-3c6a-4cb3-968c-da1174150e98","year":2003},{"cited_arxiv_id":"","doi":"10.1162/coli","is_internal_anchor":false,"ref_index":2,"title":"Computational Linguistics , volume =","work_id":"816c1e36-4060-4ea8-8f0b-b8c0efaf2db9","year":2022},{"cited_arxiv_id":"","doi":"10.5281/zenodo.18842011","is_internal_anchor":false,"ref_index":3,"title":"V . P. Bhardwaj. Agentassay: Token-efficient regression testing for non-deterministic ai agent workflows, 2026. URLhttps://zenodo.org/doi/10.5281/zenodo.18842011","work_id":"5c7b704d-3c23-40d1-b377-386a95b3e68f","year":2026},{"cited_arxiv_id":"2211.03540","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Measuring Progress on Scalable Oversight for Large Language Models","work_id":"e7f92eb1-2050-4e60-bc27-82c94d7694c5","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"V . Braun and V . Clarke. Using thematic analysis in psychology.Qualitative research in psy- chology, 3(2):77–101, 2006","work_id":"2446f94f-3bcd-49d5-8fa5-0934610493d7","year":2006}],"snapshot_sha256":"5d4c0289669aa201a2d040e011966dcd5b9765be347d3e4d17adf7e5179ddffc"},"source":{"id":"2605.13625","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-14T18:19:50.443478Z","id":"d2162ca1-04fd-46c8-a57a-eaa122cb4f38","model_set":{"reader":"grok-4.3"},"one_line_summary":"ACT*ONOMY is a Grounded-Theory-derived hierarchical taxonomy and open repository that enables systematic comparison and characterization of autonomous agent behavior across trajectories.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"ACTONOMY taxonomy structures agent behavior into 10 actions and 120 categories for consistent interpretation.","strongest_claim":"Our experiments show that ACTONOMY can compare behavioral profiles across agents and characterize a single agent's behavior across diverse trajectories, surfacing patterns indicative of failure modes.","weakest_assumption":"That a taxonomy developed via Grounded Theory on a limited set of agent traces will remain comprehensive and unbiased when applied to new agents, tasks, and longer trajectories without substantial revision."}},"verdict_id":"d2162ca1-04fd-46c8-a57a-eaa122cb4f38"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a559206b772bbdb7205fe1ba142b5f7bcb95747fa192b6ed5609c92001e0e7ca","target":"record","created_at":"2026-05-18T02:44:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6c023ff3ab7fcba1a82b2fb3cd5c1b7b20d8f34a480d8e4a8a1e563d64253f0d","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-13T14:52:40Z","title_canon_sha256":"24f750ded8e0ac1b824ce62503a6fc18b9627aa001265b439c94738eae709819"},"schema_version":"1.0","source":{"id":"2605.13625","kind":"arxiv","version":1}},"canonical_sha256":"65fe2531c0fdeb42f3ff6a19a404f3121e84d0c863f372fb86336848a651dc2f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"65fe2531c0fdeb42f3ff6a19a404f3121e84d0c863f372fb86336848a651dc2f","first_computed_at":"2026-05-18T02:44:17.827927Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T02:44:17.827927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IgsJIKthF22JhioSsxzs5Q6nYS3w3IK51ZB+2GyM61u0QYO13HfzP6g9nVb0wVUeHhKP41dysvBpWMSky91LCg==","signature_status":"signed_v1","signed_at":"2026-05-18T02:44:17.828370Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.13625","source_kind":"arxiv","source_version":1}}},"equivocations":[{"signer_id":"pith.science","event_type":"integrity_finding","target":"integrity","event_ids":["sha256:0b42e10d7187c6ef7d0cc54fa3177b83fac7aee9c5c48a1f00babec544d03196","sha256:8966ab28e6460e847b3c13b4e5deed4a1bd69e4590ef9df361d42a0e4681b1dc"]}],"invalid_events":[],"applied_event_ids":["sha256:a559206b772bbdb7205fe1ba142b5f7bcb95747fa192b6ed5609c92001e0e7ca","sha256:7e9b22f78ff4a740a7fd32cdaba6300ce1da69dab8711f4788f8f45d94982106"],"state_sha256":"e89d57b19912b364e5c607b51ab5d5b35b5d7e51f569c139b954c702040f4ecf"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wD01GWt3W4moOyaGfCpmqF061d4P8ZVCNi9XTmrWRN1TMj3Irm0opEI3d9ANsClxEV9WenaJqsZzJw1ysgFIBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-25T04:00:00.409231Z","bundle_sha256":"618542b132b73b4911bb78ef626bf0b296e3cd8a4983690ab7b90cdefb7269b6"}}