{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WRFCPNIKAAP4WY6ZQ3KV7ED3NE","short_pith_number":"pith:WRFCPNIK","schema_version":"1.0","canonical_sha256":"b44a27b50a001fcb63d986d55f907b691ff5662a6d0031edac5a475b007d66b7","source":{"kind":"arxiv","id":"2410.04640","version":2},"attestation_state":"computed","paper":{"title":"Unpacking Failure Modes of Generative Policies: Runtime Monitoring of Consistency and Progress","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Christopher Agia, Jeannette Bohg, Jingyun Yang, Marco Pavone, Rika Antonova, Rohan Sinha, Zi-Ang Cao","submitted_at":"2024-10-06T22:13:30Z","abstract_excerpt":"Robot behavior policies trained via imitation learning are prone to failure under conditions that deviate from their training data. Thus, algorithms that monitor learned policies at test time and provide early warnings of failure are necessary to facilitate scalable deployment. We propose Sentinel, a runtime monitoring framework that splits the detection of failures into two complementary categories: 1) Erratic failures, which we detect using statistical measures of temporal action consistency, and 2) task progression failures, where we use Vision Language Models (VLMs) to detect when the poli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04640","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-06T22:13:30Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"96f397411a29aa3377663528a0819d33a9c1949369e368f55c483f5639622c24","abstract_canon_sha256":"d300df931e47748e11868e39731adb12fec784b034e10937591caa2a40904602"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:15.860441Z","signature_b64":"mITxO9uUTPrH9bjp12cpOcOgAdrZeNcmfIVwS9bPsj+FWDO2L81qpuVTCdQy+8GoO597y12leOq2v9L7Jk+gAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b44a27b50a001fcb63d986d55f907b691ff5662a6d0031edac5a475b007d66b7","last_reissued_at":"2026-07-05T09:29:15.859885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:15.859885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unpacking Failure Modes of Generative Policies: Runtime Monitoring of Consistency and Progress","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Christopher Agia, Jeannette Bohg, Jingyun Yang, Marco Pavone, Rika Antonova, Rohan Sinha, Zi-Ang Cao","submitted_at":"2024-10-06T22:13:30Z","abstract_excerpt":"Robot behavior policies trained via imitation learning are prone to failure under conditions that deviate from their training data. Thus, algorithms that monitor learned policies at test time and provide early warnings of failure are necessary to facilitate scalable deployment. We propose Sentinel, a runtime monitoring framework that splits the detection of failures into two complementary categories: 1) Erratic failures, which we detect using statistical measures of temporal action consistency, and 2) task progression failures, where we use Vision Language Models (VLMs) to detect when the poli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04640","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04640/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04640","created_at":"2026-07-05T09:29:15.859947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04640v2","created_at":"2026-07-05T09:29:15.859947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04640","created_at":"2026-07-05T09:29:15.859947+00:00"},{"alias_kind":"pith_short_12","alias_value":"WRFCPNIKAAP4","created_at":"2026-07-05T09:29:15.859947+00:00"},{"alias_kind":"pith_short_16","alias_value":"WRFCPNIKAAP4WY6Z","created_at":"2026-07-05T09:29:15.859947+00:00"},{"alias_kind":"pith_short_8","alias_value":"WRFCPNIK","created_at":"2026-07-05T09:29:15.859947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":89,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19998","citing_title":"Tri-Info: Generalizable, Interpretable Failure Prediction for VLA Models via Information Theory","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20754","citing_title":"Perturbation-Based Uncertainty for Failure Detection in Vision-Language-Action Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06660","citing_title":"AEGIS: A Backup Reflex for Physical AI","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03385","citing_title":"Grasp-Then-Plan with Failure Attribution: A Closed Two-Stage Framework for Precise and Generalizable Robotic Manipulation","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29605","citing_title":"VLAConf: Calibrated Task-Success Confidence for Vision-Language-Action Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28726","citing_title":"How VLAs Fail Differently: Black-Box Action Monitoring Reveals Architecture-Specific Failure Signatures","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2412.02818","citing_title":"RoboMD: Uncovering Robot Vulnerabilities through Semantic Potential Fields","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE","json":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE.json","graph_json":"https://pith.science/api/pith-number/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/graph.json","events_json":"https://pith.science/api/pith-number/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/events.json","paper":"https://pith.science/paper/WRFCPNIK"},"agent_actions":{"view_html":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE","download_json":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE.json","view_paper":"https://pith.science/paper/WRFCPNIK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04640&json=true","fetch_graph":"https://pith.science/api/pith-number/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/graph.json","fetch_events":"https://pith.science/api/pith-number/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/action/storage_attestation","attest_author":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/action/author_attestation","sign_citation":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/action/citation_signature","submit_replication":"https://pith.science/pith/WRFCPNIKAAP4WY6ZQ3KV7ED3NE/action/replication_record"}},"created_at":"2026-07-05T09:29:15.859947+00:00","updated_at":"2026-07-05T09:29:15.859947+00:00"}