{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DJPA7YNB4KI4FUNY6SXRBPZ5ES","short_pith_number":"pith:DJPA7YNB","schema_version":"1.0","canonical_sha256":"1a5e0fe1a1e291c2d1b8f4af10bf3d24b72cb944f6df2115217ac181db8e0d20","source":{"kind":"arxiv","id":"2410.04683","version":2},"attestation_state":"computed","paper":{"title":"Towards Measuring Goal-Directedness in AI Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dylan Xu, Juan-Pablo Rivera","submitted_at":"2024-10-07T01:34:42Z","abstract_excerpt":"Recent advances in deep learning have brought attention to the possibility of creating advanced, general AI systems that outperform humans across many tasks. However, if these systems pursue unintended goals, there could be catastrophic consequences. A key prerequisite for AI systems pursuing unintended goals is whether they will behave in a coherent and goal-directed manner in the first place, optimizing for some unknown goal; there exists significant research trying to evaluate systems for said behaviors. However, the most rigorous definitions of goal-directedness we currently have are diffi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04683","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-07T01:34:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7ca96afcf645aace1ef7f40d1dcf86bc28907dbdc51a0154d1c7c08d768325c4","abstract_canon_sha256":"3a5b26eba4f3fc2b6d352adf1c964f5899ff58b9d22f119d3efefae9c186006f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:50.229388Z","signature_b64":"34YFKrkDU0b6m6h2A2jWv0Tpiy20Zv3DqmvHIN7gArWcJ73/XZsatBIWRml/Te+oUxJn7IYFzbzVws1XqMPdDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a5e0fe1a1e291c2d1b8f4af10bf3d24b72cb944f6df2115217ac181db8e0d20","last_reissued_at":"2026-07-05T09:38:50.228895Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:50.228895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Measuring Goal-Directedness in AI Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dylan Xu, Juan-Pablo Rivera","submitted_at":"2024-10-07T01:34:42Z","abstract_excerpt":"Recent advances in deep learning have brought attention to the possibility of creating advanced, general AI systems that outperform humans across many tasks. However, if these systems pursue unintended goals, there could be catastrophic consequences. A key prerequisite for AI systems pursuing unintended goals is whether they will behave in a coherent and goal-directed manner in the first place, optimizing for some unknown goal; there exists significant research trying to evaluate systems for said behaviors. However, the most rigorous definitions of goal-directedness we currently have are diffi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04683","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04683","created_at":"2026-07-05T09:38:50.228953+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04683v2","created_at":"2026-07-05T09:38:50.228953+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04683","created_at":"2026-07-05T09:38:50.228953+00:00"},{"alias_kind":"pith_short_12","alias_value":"DJPA7YNB4KI4","created_at":"2026-07-05T09:38:50.228953+00:00"},{"alias_kind":"pith_short_16","alias_value":"DJPA7YNB4KI4FUNY","created_at":"2026-07-05T09:38:50.228953+00:00"},{"alias_kind":"pith_short_8","alias_value":"DJPA7YNB","created_at":"2026-07-05T09:38:50.228953+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00248","citing_title":"Causal Foundations of Collective Agency","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES","json":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES.json","graph_json":"https://pith.science/api/pith-number/DJPA7YNB4KI4FUNY6SXRBPZ5ES/graph.json","events_json":"https://pith.science/api/pith-number/DJPA7YNB4KI4FUNY6SXRBPZ5ES/events.json","paper":"https://pith.science/paper/DJPA7YNB"},"agent_actions":{"view_html":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES","download_json":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES.json","view_paper":"https://pith.science/paper/DJPA7YNB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04683&json=true","fetch_graph":"https://pith.science/api/pith-number/DJPA7YNB4KI4FUNY6SXRBPZ5ES/graph.json","fetch_events":"https://pith.science/api/pith-number/DJPA7YNB4KI4FUNY6SXRBPZ5ES/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES/action/storage_attestation","attest_author":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES/action/author_attestation","sign_citation":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES/action/citation_signature","submit_replication":"https://pith.science/pith/DJPA7YNB4KI4FUNY6SXRBPZ5ES/action/replication_record"}},"created_at":"2026-07-05T09:38:50.228953+00:00","updated_at":"2026-07-05T09:38:50.228953+00:00"}