{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:JBM4CAXLWIRCW7MAG6OM5WUH6A","short_pith_number":"pith:JBM4CAXL","canonical_record":{"source":{"id":"2607.10601","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-12T06:38:55Z","cross_cats_sorted":[],"title_canon_sha256":"9ebcb1e0244f453c5ef279f3c5789d6508baaf0ef306e12f2efd60cac259f465","abstract_canon_sha256":"1654d5c60ff5573a38944ffc69c0611346988cffc4c939d0cde8585ab25aac5b"},"schema_version":"1.0"},"canonical_sha256":"4859c102ebb2222b7d80379cceda87f0149c6190cb3028fa056a6836d133c64c","source":{"kind":"arxiv","id":"2607.10601","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.10601","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"arxiv_version","alias_value":"2607.10601v1","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.10601","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_12","alias_value":"JBM4CAXLWIRC","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_16","alias_value":"JBM4CAXLWIRCW7MA","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_8","alias_value":"JBM4CAXL","created_at":"2026-07-14T01:21:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:JBM4CAXLWIRCW7MAG6OM5WUH6A","target":"record","payload":{"canonical_record":{"source":{"id":"2607.10601","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-12T06:38:55Z","cross_cats_sorted":[],"title_canon_sha256":"9ebcb1e0244f453c5ef279f3c5789d6508baaf0ef306e12f2efd60cac259f465","abstract_canon_sha256":"1654d5c60ff5573a38944ffc69c0611346988cffc4c939d0cde8585ab25aac5b"},"schema_version":"1.0"},"canonical_sha256":"4859c102ebb2222b7d80379cceda87f0149c6190cb3028fa056a6836d133c64c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:21:23.363208Z","signature_b64":"7+P5LY8QbYRyAQukiI/EU5t6yYmUvRg+okh1s8KIFHz9EPkD0hcZRvy3pqNtq/YSoFGHq3C9ShC5aCj6V6x1DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4859c102ebb2222b7d80379cceda87f0149c6190cb3028fa056a6836d133c64c","last_reissued_at":"2026-07-14T01:21:23.362416Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:21:23.362416Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.10601","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T01:21:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9KCjSGz2K1pIH7S67xI2fN+YDMjN46Hj3WyPQQs/LsA0Auk+jMqDJRnBbK3rgWvvT4KzfyjUbZwLmTYSz7lDDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T08:21:50.530766Z"},"content_sha256":"82c62b1ce6994f58f45e58c4226c5d662e13caaabcb3db2a6e2aebd8a1156ff6","schema_version":"1.0","event_id":"sha256:82c62b1ce6994f58f45e58c4226c5d662e13caaabcb3db2a6e2aebd8a1156ff6"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:JBM4CAXLWIRCW7MAG6OM5WUH6A","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Agentic-DPO: From Imitation to Agentic Policy Optimization on Expert Trajectories","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alan Yuille, Yixiong Chen","submitted_at":"2026-07-12T06:38:55Z","abstract_excerpt":"Large Language Model (LLM) agents are commonly trained from expert trajectories using supervised fine-tuning (SFT), which treats multi-turn agent behavior as ordinary text imitation. This recipe is simple and low-cost, but it only learns to imitate the sequence of expert actions, rather than training the agent to choose the right action against plausible mistakes at each state. Existing methods to mitigate this problem include preference learning or reinforcement learning, but they usually need high-cost environment rollouts and reward models. We propose Agentic-DPO, a lightweight offline agen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.10601","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.10601/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-14T01:21:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vRXtVs93ffbbeLu069yZAaOwPbqsT6KHUGoDbVSsDMlfyDIpphX8BRpB2JmBfqyQvMenVEZW7yZirPuI6CuxCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T08:21:50.531428Z"},"content_sha256":"5aa7a54c33f5d7e021a9679434b5843d27680b94204d28c41665b898255ecec5","schema_version":"1.0","event_id":"sha256:5aa7a54c33f5d7e021a9679434b5843d27680b94204d28c41665b898255ecec5"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/bundle.json","state_url":"https://pith.science/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T08:21:50Z","links":{"resolver":"https://pith.science/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A","bundle":"https://pith.science/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/bundle.json","state":"https://pith.science/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JBM4CAXLWIRCW7MAG6OM5WUH6A/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:JBM4CAXLWIRCW7MAG6OM5WUH6A","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1654d5c60ff5573a38944ffc69c0611346988cffc4c939d0cde8585ab25aac5b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-12T06:38:55Z","title_canon_sha256":"9ebcb1e0244f453c5ef279f3c5789d6508baaf0ef306e12f2efd60cac259f465"},"schema_version":"1.0","source":{"id":"2607.10601","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.10601","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"arxiv_version","alias_value":"2607.10601v1","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.10601","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_12","alias_value":"JBM4CAXLWIRC","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_16","alias_value":"JBM4CAXLWIRCW7MA","created_at":"2026-07-14T01:21:23Z"},{"alias_kind":"pith_short_8","alias_value":"JBM4CAXL","created_at":"2026-07-14T01:21:23Z"}],"graph_snapshots":[{"event_id":"sha256:5aa7a54c33f5d7e021a9679434b5843d27680b94204d28c41665b898255ecec5","target":"graph","created_at":"2026-07-14T01:21:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.10601/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Language Model (LLM) agents are commonly trained from expert trajectories using supervised fine-tuning (SFT), which treats multi-turn agent behavior as ordinary text imitation. This recipe is simple and low-cost, but it only learns to imitate the sequence of expert actions, rather than training the agent to choose the right action against plausible mistakes at each state. Existing methods to mitigate this problem include preference learning or reinforcement learning, but they usually need high-cost environment rollouts and reward models. We propose Agentic-DPO, a lightweight offline agen","authors_text":"Alan Yuille, Yixiong Chen","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-12T06:38:55Z","title":"Agentic-DPO: From Imitation to Agentic Policy Optimization on Expert Trajectories"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.10601","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:82c62b1ce6994f58f45e58c4226c5d662e13caaabcb3db2a6e2aebd8a1156ff6","target":"record","created_at":"2026-07-14T01:21:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1654d5c60ff5573a38944ffc69c0611346988cffc4c939d0cde8585ab25aac5b","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2026-07-12T06:38:55Z","title_canon_sha256":"9ebcb1e0244f453c5ef279f3c5789d6508baaf0ef306e12f2efd60cac259f465"},"schema_version":"1.0","source":{"id":"2607.10601","kind":"arxiv","version":1}},"canonical_sha256":"4859c102ebb2222b7d80379cceda87f0149c6190cb3028fa056a6836d133c64c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4859c102ebb2222b7d80379cceda87f0149c6190cb3028fa056a6836d133c64c","first_computed_at":"2026-07-14T01:21:23.362416Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-14T01:21:23.362416Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7+P5LY8QbYRyAQukiI/EU5t6yYmUvRg+okh1s8KIFHz9EPkD0hcZRvy3pqNtq/YSoFGHq3C9ShC5aCj6V6x1DA==","signature_status":"signed_v1","signed_at":"2026-07-14T01:21:23.363208Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.10601","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:82c62b1ce6994f58f45e58c4226c5d662e13caaabcb3db2a6e2aebd8a1156ff6","sha256:5aa7a54c33f5d7e021a9679434b5843d27680b94204d28c41665b898255ecec5"],"state_sha256":"bd17ed13272471d321ac22dc948185244bb0f279d2024f3266fc939643a0c54a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cTB24wRefPMveZXxPxglsP3o8tOCl0M64UkPQIfy4R0I8kfHvmwGB7wiOv1mvBMl7JsIpy8O842tu0RbhtWGDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T08:21:50.536384Z","bundle_sha256":"6cb1623d63c498064c2844467ce67a5e9d3dcca96dbc30b06de83ff18c10ec60"}}