{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:TARGY2IFQN5YBI5P73MQANDELV","short_pith_number":"pith:TARGY2IF","schema_version":"1.0","canonical_sha256":"98226c6905837b80a3affed90034645d77d3c806f24996fe5684eb3596522ed7","source":{"kind":"arxiv","id":"2112.05125","version":2},"attestation_state":"computed","paper":{"title":"Rethinking the Authorship Verification Experimental Setups","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrei Manolache, Antonio Barbalau, Elena Burceanu, Florin Brad, Marius Popescu, Radu Ionescu","submitted_at":"2021-12-09T18:57:29Z","abstract_excerpt":"One of the main drivers of the recent advances in authorship verification is the PAN large-scale authorship dataset. Despite generating significant progress in the field, inconsistent performance differences between the closed and open test sets have been reported. To this end, we improve the experimental setup by proposing five new public splits over the PAN dataset, specifically designed to isolate and identify biases related to the text topic and to the author's writing style. We evaluate several BERT-like baselines on these splits, showing that such models are competitive with authorship v"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.05125","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-12-09T18:57:29Z","cross_cats_sorted":[],"title_canon_sha256":"042a123a95916e335999887430dbba06de2c67b6788d1ed460c3e40aa9868222","abstract_canon_sha256":"4bf601c51b682636d5830bbd637547550f657c6e4bfbd83abddb69690874c15c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:12:09.925405Z","signature_b64":"+4s+o8AXXCPsGU2EQ448+mFT70kmecPWtd0GYkG7uNF2qDvewpyo91VQHGXTnzul5nHcgVm9ja4sD/zbrqNZBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"98226c6905837b80a3affed90034645d77d3c806f24996fe5684eb3596522ed7","last_reissued_at":"2026-07-05T05:12:09.924952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:12:09.924952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking the Authorship Verification Experimental Setups","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Andrei Manolache, Antonio Barbalau, Elena Burceanu, Florin Brad, Marius Popescu, Radu Ionescu","submitted_at":"2021-12-09T18:57:29Z","abstract_excerpt":"One of the main drivers of the recent advances in authorship verification is the PAN large-scale authorship dataset. Despite generating significant progress in the field, inconsistent performance differences between the closed and open test sets have been reported. To this end, we improve the experimental setup by proposing five new public splits over the PAN dataset, specifically designed to isolate and identify biases related to the text topic and to the author's writing style. We evaluate several BERT-like baselines on these splits, showing that such models are competitive with authorship v"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.05125","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.05125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.05125","created_at":"2026-07-05T05:12:09.925008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.05125v2","created_at":"2026-07-05T05:12:09.925008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.05125","created_at":"2026-07-05T05:12:09.925008+00:00"},{"alias_kind":"pith_short_12","alias_value":"TARGY2IFQN5Y","created_at":"2026-07-05T05:12:09.925008+00:00"},{"alias_kind":"pith_short_16","alias_value":"TARGY2IFQN5YBI5P","created_at":"2026-07-05T05:12:09.925008+00:00"},{"alias_kind":"pith_short_8","alias_value":"TARGY2IF","created_at":"2026-07-05T05:12:09.925008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.08560","citing_title":"Uncertainty Estimation for the Open-Set Text Classification systems","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV","json":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV.json","graph_json":"https://pith.science/api/pith-number/TARGY2IFQN5YBI5P73MQANDELV/graph.json","events_json":"https://pith.science/api/pith-number/TARGY2IFQN5YBI5P73MQANDELV/events.json","paper":"https://pith.science/paper/TARGY2IF"},"agent_actions":{"view_html":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV","download_json":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV.json","view_paper":"https://pith.science/paper/TARGY2IF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.05125&json=true","fetch_graph":"https://pith.science/api/pith-number/TARGY2IFQN5YBI5P73MQANDELV/graph.json","fetch_events":"https://pith.science/api/pith-number/TARGY2IFQN5YBI5P73MQANDELV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV/action/storage_attestation","attest_author":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV/action/author_attestation","sign_citation":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV/action/citation_signature","submit_replication":"https://pith.science/pith/TARGY2IFQN5YBI5P73MQANDELV/action/replication_record"}},"created_at":"2026-07-05T05:12:09.925008+00:00","updated_at":"2026-07-05T05:12:09.925008+00:00"}