{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:SICBTWMLTA3QGFV54ED5N3OBOL","short_pith_number":"pith:SICBTWML","canonical_record":{"source":{"id":"2607.08647","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:18:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7a695d94dc8003fa2d5814dad4cf9166647753acafef0056f4174b6169c3e1c9","abstract_canon_sha256":"cd151a29b16ed3cc45642e05014eaedfe2e050feb8c52b5f40a0aea6124c4b47"},"schema_version":"1.0"},"canonical_sha256":"920419d98b98370316bde107d6edc172d36cf0ab921f38d0f1bccbc993030d51","source":{"kind":"arxiv","id":"2607.08647","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.08647","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"arxiv_version","alias_value":"2607.08647v1","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.08647","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_12","alias_value":"SICBTWMLTA3Q","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_16","alias_value":"SICBTWMLTA3QGFV5","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_8","alias_value":"SICBTWML","created_at":"2026-07-10T01:19:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:SICBTWMLTA3QGFV54ED5N3OBOL","target":"record","payload":{"canonical_record":{"source":{"id":"2607.08647","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:18:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7a695d94dc8003fa2d5814dad4cf9166647753acafef0056f4174b6169c3e1c9","abstract_canon_sha256":"cd151a29b16ed3cc45642e05014eaedfe2e050feb8c52b5f40a0aea6124c4b47"},"schema_version":"1.0"},"canonical_sha256":"920419d98b98370316bde107d6edc172d36cf0ab921f38d0f1bccbc993030d51","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-10T01:19:56.507555Z","signature_b64":"qDEtZEediGLIFo31usAcRc6GxT4LBDLQE+afX9T4m2GpCaNkv6JRP6dCPYnCFcsrY4uK8rF5P9VJ10z29g3+Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"920419d98b98370316bde107d6edc172d36cf0ab921f38d0f1bccbc993030d51","last_reissued_at":"2026-07-10T01:19:56.507145Z","signature_status":"signed_v1","first_computed_at":"2026-07-10T01:19:56.507145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.08647","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-10T01:19:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eo3YZmgh7YFatl17m8Ec63pEbDS/46eK+22kUxAcE/YtzzYyfRtCAL61oBES2SslxHQ1ZFWhKd3A96nq3GYUBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T04:24:09.924320Z"},"content_sha256":"956a443bb3dbad4ed2d5e3a28ce2eda2f7095d43fe04f7afa8d4afc534037381","schema_version":"1.0","event_id":"sha256:956a443bb3dbad4ed2d5e3a28ce2eda2f7095d43fe04f7afa8d4afc534037381"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:SICBTWMLTA3QGFV54ED5N3OBOL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Multi-Modal, Multi-Environment Machine Teaching for Robust Reward Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ali Larian, Chang Zong Wu, Daniel S. Brown, Qian Lin","submitted_at":"2026-07-09T16:18:16Z","abstract_excerpt":"As autonomous agents are increasingly deployed across diverse operational contexts, aligning their behavior with human intent demands reward functions that remain robust to such changes rather than overfitting to any single environment. Inverse reinforcement learning (IRL) provides a principled way to infer such objectives from human feedback. However, existing analyses of optimal teaching approaches for IRL focus on single-environment, demonstration-only settings, leaving underexplored how heterogeneous feedback modalities and environment dynamics jointly constrain reward functions that gener"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.08647","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.08647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-10T01:19:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tbFfMwVFVKm6v5msgMyf4jVZsf0GYygiWFFd5wUSRFp66Do4R/6vlo/lmf+kevrRilxIoxbwzeXZ3jHS8+dhAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T04:24:09.924897Z"},"content_sha256":"be240c9fbcff047e6fed52e1ba7af9bd509a1db5cda0ebb975bab0247f0aea6d","schema_version":"1.0","event_id":"sha256:be240c9fbcff047e6fed52e1ba7af9bd509a1db5cda0ebb975bab0247f0aea6d"},{"event_type":"integrity_finding","subject_pith_number":"pith:2026:SICBTWMLTA3QGFV54ED5N3OBOL","target":"integrity","payload":{"note":"URL 'https://ojs.aaai.org/index' returned status 404 (Not Found) at last check.","snippet":null,"arxiv_id":"2607.08647","detector":"external_links","evidence":{"url":"https://ojs.aaai.org/index","final_url":"https://ojs.aaai.org/index","host_kind":"website","status_code":404,"status_text":"Not Found","verdict_class":"incontrovertible","checked_at_unix":1783683266.8312519},"severity":"advisory","ref_index":null,"audited_at":"2026-07-10T11:34:27.597005Z","event_type":"pith.integrity.v1","detected_doi":null,"detector_url":"https://pith.science/pith-integrity-protocol#external_links","external_url":"https://ojs.aaai.org/index","finding_type":"dead_url","evidence_hash":"f7b313e7da2ad3c926e0540c2db12bb50297cbd7920f13dba7a1f5f9b59de357","paper_version":1,"verdict_class":"incontrovertible","resolved_title":null,"detector_version":"1.0.0","detected_arxiv_id":null,"integrity_event_id":11941,"payload_sha256":"e1c4be93fd815aca20a638824d4c30191c6adb476db49f474961550f25198bca","signature_b64":"iNMQWIc8ydjgesgJeyc+RqT/xhzzJyg3P4TErQOvU6ZE3ZdWFBLbk1exxLpJ3iFlC9Nxur3fWNwTiNFrYgDeBw==","signing_key_id":"pith-v1-2026-05"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-10T11:39:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Wu5OB9wDP/7/hFe4OzLGE42VyqrvCRotrtbrnr8hfg/1lA28uP6jCeNW2ukJqDvcKniTb/bB7EPF+VdTUpIlDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T04:24:09.927491Z"},"content_sha256":"d37deacab2be33ea6a41ca9544d7397977028a13d2cc8a7cc4280bcb9e6db698","schema_version":"1.0","event_id":"sha256:d37deacab2be33ea6a41ca9544d7397977028a13d2cc8a7cc4280bcb9e6db698"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SICBTWMLTA3QGFV54ED5N3OBOL/bundle.json","state_url":"https://pith.science/pith/SICBTWMLTA3QGFV54ED5N3OBOL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SICBTWMLTA3QGFV54ED5N3OBOL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T04:24:09Z","links":{"resolver":"https://pith.science/pith/SICBTWMLTA3QGFV54ED5N3OBOL","bundle":"https://pith.science/pith/SICBTWMLTA3QGFV54ED5N3OBOL/bundle.json","state":"https://pith.science/pith/SICBTWMLTA3QGFV54ED5N3OBOL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SICBTWMLTA3QGFV54ED5N3OBOL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:SICBTWMLTA3QGFV54ED5N3OBOL","merge_version":"pith-open-graph-merge-v1","event_count":3,"valid_event_count":3,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"cd151a29b16ed3cc45642e05014eaedfe2e050feb8c52b5f40a0aea6124c4b47","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:18:16Z","title_canon_sha256":"7a695d94dc8003fa2d5814dad4cf9166647753acafef0056f4174b6169c3e1c9"},"schema_version":"1.0","source":{"id":"2607.08647","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.08647","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"arxiv_version","alias_value":"2607.08647v1","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.08647","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_12","alias_value":"SICBTWMLTA3Q","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_16","alias_value":"SICBTWMLTA3QGFV5","created_at":"2026-07-10T01:19:56Z"},{"alias_kind":"pith_short_8","alias_value":"SICBTWML","created_at":"2026-07-10T01:19:56Z"}],"graph_snapshots":[{"event_id":"sha256:be240c9fbcff047e6fed52e1ba7af9bd509a1db5cda0ebb975bab0247f0aea6d","target":"graph","created_at":"2026-07-10T01:19:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.08647/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As autonomous agents are increasingly deployed across diverse operational contexts, aligning their behavior with human intent demands reward functions that remain robust to such changes rather than overfitting to any single environment. Inverse reinforcement learning (IRL) provides a principled way to infer such objectives from human feedback. However, existing analyses of optimal teaching approaches for IRL focus on single-environment, demonstration-only settings, leaving underexplored how heterogeneous feedback modalities and environment dynamics jointly constrain reward functions that gener","authors_text":"Ali Larian, Chang Zong Wu, Daniel S. Brown, Qian Lin","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:18:16Z","title":"Multi-Modal, Multi-Environment Machine Teaching for Robust Reward Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.08647","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:956a443bb3dbad4ed2d5e3a28ce2eda2f7095d43fe04f7afa8d4afc534037381","target":"record","created_at":"2026-07-10T01:19:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"cd151a29b16ed3cc45642e05014eaedfe2e050feb8c52b5f40a0aea6124c4b47","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:18:16Z","title_canon_sha256":"7a695d94dc8003fa2d5814dad4cf9166647753acafef0056f4174b6169c3e1c9"},"schema_version":"1.0","source":{"id":"2607.08647","kind":"arxiv","version":1}},"canonical_sha256":"920419d98b98370316bde107d6edc172d36cf0ab921f38d0f1bccbc993030d51","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"920419d98b98370316bde107d6edc172d36cf0ab921f38d0f1bccbc993030d51","first_computed_at":"2026-07-10T01:19:56.507145Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-10T01:19:56.507145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qDEtZEediGLIFo31usAcRc6GxT4LBDLQE+afX9T4m2GpCaNkv6JRP6dCPYnCFcsrY4uK8rF5P9VJ10z29g3+Dw==","signature_status":"signed_v1","signed_at":"2026-07-10T01:19:56.507555Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.08647","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:956a443bb3dbad4ed2d5e3a28ce2eda2f7095d43fe04f7afa8d4afc534037381","sha256:be240c9fbcff047e6fed52e1ba7af9bd509a1db5cda0ebb975bab0247f0aea6d","sha256:d37deacab2be33ea6a41ca9544d7397977028a13d2cc8a7cc4280bcb9e6db698"],"state_sha256":"3d45bd71726e8f70824a3c6918e0a62138ad0c8d80fa5f5de5bda80dc947fd5a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"W+6iVPxpcD0DG2HMq0CfF0002J5Zx8tFm8LipfebJMNGWeA35Pz0Xekvym2UL6tPXW5U59OzG0OzIZhB9uO/CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T04:24:09.929778Z","bundle_sha256":"2fcaf892b3f63c3aa573d9865f4f74866b2b43a082239cf59c964ea57e1ff109"}}