{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JY4L4O4UI52FSWVW6Y6WYTB5X5","short_pith_number":"pith:JY4L4O4U","schema_version":"1.0","canonical_sha256":"4e38be3b944774595ab6f63d6c4c3dbf571dcf9ae93dcff66b507cc1526093ba","source":{"kind":"arxiv","id":"2205.04887","version":2},"attestation_state":"computed","paper":{"title":"Search-Based Testing of Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.LG","authors_text":"Bernhard K. Aichernig, Bettina K\\\"onighofer, Filip Cano C\\'ordoba, Martin Tappler","submitted_at":"2022-05-07T12:40:45Z","abstract_excerpt":"Evaluation of deep reinforcement learning (RL) is inherently challenging. Especially the opaqueness of learned policies and the stochastic nature of both agents and environments make testing the behavior of deep RL agents difficult. We present a search-based testing framework that enables a wide range of novel analysis capabilities for evaluating the safety and performance of deep RL agents. For safety testing, our framework utilizes a search algorithm that searches for a reference trace that solves the RL task. The backtracking states of the search, called boundary states, pose safety-critica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.04887","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-07T12:40:45Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"4714a57205acd9d494415152b8ce22cc2ca31eeed219f1f629edd2929d60997c","abstract_canon_sha256":"2da5b1d089016e5d4713b5af4340862420c24a12900c11b4bf07f031e7c95fcc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:08.874914Z","signature_b64":"U6Z8voRblhTTbr4PvPmHjCvRBc48UwSfM/m9l9LzaCa20xDPHMpnGnx9kPdgT1IRmTCs+YYFeBEcT0e2agvjDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4e38be3b944774595ab6f63d6c4c3dbf571dcf9ae93dcff66b507cc1526093ba","last_reissued_at":"2026-07-05T04:23:08.874525Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:08.874525Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Search-Based Testing of Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.LG","authors_text":"Bernhard K. Aichernig, Bettina K\\\"onighofer, Filip Cano C\\'ordoba, Martin Tappler","submitted_at":"2022-05-07T12:40:45Z","abstract_excerpt":"Evaluation of deep reinforcement learning (RL) is inherently challenging. Especially the opaqueness of learned policies and the stochastic nature of both agents and environments make testing the behavior of deep RL agents difficult. We present a search-based testing framework that enables a wide range of novel analysis capabilities for evaluating the safety and performance of deep RL agents. For safety testing, our framework utilizes a search algorithm that searches for a reference trace that solves the RL task. The backtracking states of the search, called boundary states, pose safety-critica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.04887","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.04887/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.04887","created_at":"2026-07-05T04:23:08.874582+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.04887v2","created_at":"2026-07-05T04:23:08.874582+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.04887","created_at":"2026-07-05T04:23:08.874582+00:00"},{"alias_kind":"pith_short_12","alias_value":"JY4L4O4UI52F","created_at":"2026-07-05T04:23:08.874582+00:00"},{"alias_kind":"pith_short_16","alias_value":"JY4L4O4UI52FSWVW","created_at":"2026-07-05T04:23:08.874582+00:00"},{"alias_kind":"pith_short_8","alias_value":"JY4L4O4U","created_at":"2026-07-05T04:23:08.874582+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.21553","citing_title":"Reusable Test Suites for Reinforcement Learning","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5","json":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5.json","graph_json":"https://pith.science/api/pith-number/JY4L4O4UI52FSWVW6Y6WYTB5X5/graph.json","events_json":"https://pith.science/api/pith-number/JY4L4O4UI52FSWVW6Y6WYTB5X5/events.json","paper":"https://pith.science/paper/JY4L4O4U"},"agent_actions":{"view_html":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5","download_json":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5.json","view_paper":"https://pith.science/paper/JY4L4O4U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.04887&json=true","fetch_graph":"https://pith.science/api/pith-number/JY4L4O4UI52FSWVW6Y6WYTB5X5/graph.json","fetch_events":"https://pith.science/api/pith-number/JY4L4O4UI52FSWVW6Y6WYTB5X5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5/action/storage_attestation","attest_author":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5/action/author_attestation","sign_citation":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5/action/citation_signature","submit_replication":"https://pith.science/pith/JY4L4O4UI52FSWVW6Y6WYTB5X5/action/replication_record"}},"created_at":"2026-07-05T04:23:08.874582+00:00","updated_at":"2026-07-05T04:23:08.874582+00:00"}