{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:PUN5QHNXHBQVO3L7EVLUWMS26H","short_pith_number":"pith:PUN5QHNX","canonical_record":{"source":{"id":"2508.21553","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-29T12:10:05Z","cross_cats_sorted":[],"title_canon_sha256":"f0abc56c0d62876c340073b45a869e10d28e96674d3e0037189b14c767c7bb42","abstract_canon_sha256":"6d0f190649365f222c20794b3d536ea44379a0d5c29a2ad4b25ed6844040db5a"},"schema_version":"1.0"},"canonical_sha256":"7d1bd81db73861576d7f25574b325af1cf910679e4e3db1e8b0b7ab01bd55d49","source":{"kind":"arxiv","id":"2508.21553","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.21553","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"arxiv_version","alias_value":"2508.21553v1","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21553","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_12","alias_value":"PUN5QHNXHBQV","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_16","alias_value":"PUN5QHNXHBQVO3L7","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_8","alias_value":"PUN5QHNX","created_at":"2026-07-05T12:01:36Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:PUN5QHNXHBQVO3L7EVLUWMS26H","target":"record","payload":{"canonical_record":{"source":{"id":"2508.21553","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-29T12:10:05Z","cross_cats_sorted":[],"title_canon_sha256":"f0abc56c0d62876c340073b45a869e10d28e96674d3e0037189b14c767c7bb42","abstract_canon_sha256":"6d0f190649365f222c20794b3d536ea44379a0d5c29a2ad4b25ed6844040db5a"},"schema_version":"1.0"},"canonical_sha256":"7d1bd81db73861576d7f25574b325af1cf910679e4e3db1e8b0b7ab01bd55d49","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:36.623472Z","signature_b64":"Qh6SwrKUYdwkdYL1p+/ofMdvYsSs+kN1RDfzQNeHaPWcMcXXdn7gkTHGq1Mdbhd9C9bndZvvBGo+i02KGwwYBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d1bd81db73861576d7f25574b325af1cf910679e4e3db1e8b0b7ab01bd55d49","last_reissued_at":"2026-07-05T12:01:36.622891Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:36.622891Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.21553","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"p3J0xz10GQYa3y1GtzuC67UanuJGit4yNvDPF6dzsNJzmg5pB0LZSDRryxsIQS1sp01VfAnmjylYK+fRCmr8AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T23:58:10.393207Z"},"content_sha256":"4e70bc8087ae769a79f92ba12ce3547275a01e634f34348ee5dead66a16120d3","schema_version":"1.0","event_id":"sha256:4e70bc8087ae769a79f92ba12ce3547275a01e634f34348ee5dead66a16120d3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:PUN5QHNXHBQVO3L7EVLUWMS26H","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reusable Test Suites for Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Dennis Gross, Helge Spieker, J{\\o}rn Eirik Betten, Pedro Lind, Quentin Mazouni","submitted_at":"2025-08-29T12:10:05Z","abstract_excerpt":"Reinforcement learning (RL) agents show great promise in solving sequential decision-making tasks. However, validating the reliability and performance of the agent policies' behavior for deployment remains challenging. Most reinforcement learning policy testing methods produce test suites tailored to the agent policy being tested, and their relevance to other policies is unclear. This work presents Multi-Policy Test Case Selection (MPTCS), a novel automated test suite selection method for RL environments, designed to extract test cases generated by any policy testing framework based on their s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21553","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.21553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:36Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DomwNxQIepuw7GLPCJX8XXZwWKmGIsSrKYVFqeyCc/ApSj6kZyyFgOO5/bVrI3Vf7rA+ei+i4PMfUj6EmArBAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T23:58:10.393781Z"},"content_sha256":"5b6fc6b9b9603ef16185e2884d2b4233a4cdbb33b207f90513b4739c362f6152","schema_version":"1.0","event_id":"sha256:5b6fc6b9b9603ef16185e2884d2b4233a4cdbb33b207f90513b4739c362f6152"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/bundle.json","state_url":"https://pith.science/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T23:58:10Z","links":{"resolver":"https://pith.science/pith/PUN5QHNXHBQVO3L7EVLUWMS26H","bundle":"https://pith.science/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/bundle.json","state":"https://pith.science/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PUN5QHNXHBQVO3L7EVLUWMS26H/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:PUN5QHNXHBQVO3L7EVLUWMS26H","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6d0f190649365f222c20794b3d536ea44379a0d5c29a2ad4b25ed6844040db5a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-29T12:10:05Z","title_canon_sha256":"f0abc56c0d62876c340073b45a869e10d28e96674d3e0037189b14c767c7bb42"},"schema_version":"1.0","source":{"id":"2508.21553","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.21553","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"arxiv_version","alias_value":"2508.21553v1","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.21553","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_12","alias_value":"PUN5QHNXHBQV","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_16","alias_value":"PUN5QHNXHBQVO3L7","created_at":"2026-07-05T12:01:36Z"},{"alias_kind":"pith_short_8","alias_value":"PUN5QHNX","created_at":"2026-07-05T12:01:36Z"}],"graph_snapshots":[{"event_id":"sha256:5b6fc6b9b9603ef16185e2884d2b4233a4cdbb33b207f90513b4739c362f6152","target":"graph","created_at":"2026-07-05T12:01:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.21553/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) agents show great promise in solving sequential decision-making tasks. However, validating the reliability and performance of the agent policies' behavior for deployment remains challenging. Most reinforcement learning policy testing methods produce test suites tailored to the agent policy being tested, and their relevance to other policies is unclear. This work presents Multi-Policy Test Case Selection (MPTCS), a novel automated test suite selection method for RL environments, designed to extract test cases generated by any policy testing framework based on their s","authors_text":"Dennis Gross, Helge Spieker, J{\\o}rn Eirik Betten, Pedro Lind, Quentin Mazouni","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-29T12:10:05Z","title":"Reusable Test Suites for Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.21553","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4e70bc8087ae769a79f92ba12ce3547275a01e634f34348ee5dead66a16120d3","target":"record","created_at":"2026-07-05T12:01:36Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6d0f190649365f222c20794b3d536ea44379a0d5c29a2ad4b25ed6844040db5a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-29T12:10:05Z","title_canon_sha256":"f0abc56c0d62876c340073b45a869e10d28e96674d3e0037189b14c767c7bb42"},"schema_version":"1.0","source":{"id":"2508.21553","kind":"arxiv","version":1}},"canonical_sha256":"7d1bd81db73861576d7f25574b325af1cf910679e4e3db1e8b0b7ab01bd55d49","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7d1bd81db73861576d7f25574b325af1cf910679e4e3db1e8b0b7ab01bd55d49","first_computed_at":"2026-07-05T12:01:36.622891Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:01:36.622891Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Qh6SwrKUYdwkdYL1p+/ofMdvYsSs+kN1RDfzQNeHaPWcMcXXdn7gkTHGq1Mdbhd9C9bndZvvBGo+i02KGwwYBw==","signature_status":"signed_v1","signed_at":"2026-07-05T12:01:36.623472Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.21553","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4e70bc8087ae769a79f92ba12ce3547275a01e634f34348ee5dead66a16120d3","sha256:5b6fc6b9b9603ef16185e2884d2b4233a4cdbb33b207f90513b4739c362f6152"],"state_sha256":"aef334cdf384537a23caef80996c437778612c0236ec78f696989e2ecb047913"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VkORJQj0jPAZVorzE00Y4ZE5v0k7wYsGoBJR4SIVt2IBFr3Tf+L+zKPol3j1AyNEkDv7sE809NmT0Tu5WW6RDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T23:58:10.399409Z","bundle_sha256":"9e077379eb576b1a56beb578484581b1e381a45e9effbbd2832153beb07c2b20"}}