{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NQWE6GJS4JMQM2IYPDURLEYV5P","short_pith_number":"pith:NQWE6GJS","schema_version":"1.0","canonical_sha256":"6c2c4f1932e25906691878e9159315ebf277b72dd1639269f3edd39ca1466680","source":{"kind":"arxiv","id":"2305.12951","version":1},"attestation_state":"computed","paper":{"title":"Cross-functional Analysis of Generalisation in Behavioural Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Roth, Pedro Henrique Luz de Araujo","submitted_at":"2023-05-22T11:54:19Z","abstract_excerpt":"In behavioural testing, system functionalities underrepresented in the standard evaluation setting (with a held-out test set) are validated through controlled input-output pairs. Optimising performance on the behavioural tests during training (behavioural learning) would improve coverage of phenomena not sufficiently represented in the i.i.d. data and could lead to seemingly more robust models. However, there is the risk that the model narrowly captures spurious correlations from the behavioural test suite, leading to overestimation and misrepresentation of model performance -- one of the orig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.12951","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-22T11:54:19Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7d7f1a4d219ecd950cc79de42e9dd88d9968824103a46e3a5d0495b41299fc90","abstract_canon_sha256":"515e3fa11e754c334ad8d36494a2df7604aa798935ff76bd39cc01487f35637e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:44:31.582742Z","signature_b64":"G+yJjaxt19PPLcfCrBRDFh1f7elilAYbD7tvg/zQuw1WpHafQig0nW6mIX/xrSN9e5jtwvqrgWCEhTZmOthRAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c2c4f1932e25906691878e9159315ebf277b72dd1639269f3edd39ca1466680","last_reissued_at":"2026-07-05T06:44:31.582327Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:44:31.582327Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cross-functional Analysis of Generalisation in Behavioural Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Roth, Pedro Henrique Luz de Araujo","submitted_at":"2023-05-22T11:54:19Z","abstract_excerpt":"In behavioural testing, system functionalities underrepresented in the standard evaluation setting (with a held-out test set) are validated through controlled input-output pairs. Optimising performance on the behavioural tests during training (behavioural learning) would improve coverage of phenomena not sufficiently represented in the i.i.d. data and could lead to seemingly more robust models. However, there is the risk that the model narrowly captures spurious correlations from the behavioural test suite, leading to overestimation and misrepresentation of model performance -- one of the orig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.12951","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.12951/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.12951","created_at":"2026-07-05T06:44:31.582388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.12951v1","created_at":"2026-07-05T06:44:31.582388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.12951","created_at":"2026-07-05T06:44:31.582388+00:00"},{"alias_kind":"pith_short_12","alias_value":"NQWE6GJS4JMQ","created_at":"2026-07-05T06:44:31.582388+00:00"},{"alias_kind":"pith_short_16","alias_value":"NQWE6GJS4JMQM2IY","created_at":"2026-07-05T06:44:31.582388+00:00"},{"alias_kind":"pith_short_8","alias_value":"NQWE6GJS","created_at":"2026-07-05T06:44:31.582388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P","json":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P.json","graph_json":"https://pith.science/api/pith-number/NQWE6GJS4JMQM2IYPDURLEYV5P/graph.json","events_json":"https://pith.science/api/pith-number/NQWE6GJS4JMQM2IYPDURLEYV5P/events.json","paper":"https://pith.science/paper/NQWE6GJS"},"agent_actions":{"view_html":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P","download_json":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P.json","view_paper":"https://pith.science/paper/NQWE6GJS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.12951&json=true","fetch_graph":"https://pith.science/api/pith-number/NQWE6GJS4JMQM2IYPDURLEYV5P/graph.json","fetch_events":"https://pith.science/api/pith-number/NQWE6GJS4JMQM2IYPDURLEYV5P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P/action/storage_attestation","attest_author":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P/action/author_attestation","sign_citation":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P/action/citation_signature","submit_replication":"https://pith.science/pith/NQWE6GJS4JMQM2IYPDURLEYV5P/action/replication_record"}},"created_at":"2026-07-05T06:44:31.582388+00:00","updated_at":"2026-07-05T06:44:31.582388+00:00"}