{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QPIQVLXHFQKNRLCCYHXOMI375N","short_pith_number":"pith:QPIQVLXH","schema_version":"1.0","canonical_sha256":"83d10aaee72c14d8ac42c1eee6237feb7111c2a0dde01bd3cff4f4fc20400255","source":{"kind":"arxiv","id":"2310.17121","version":1},"attestation_state":"computed","paper":{"title":"Test-time Augmentation for Factual Probing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin Heinzerling, Go kamoda, Keisuke Sakaguchi, Kentaro Inui","submitted_at":"2023-10-26T03:41:32Z","abstract_excerpt":"Factual probing is a method that uses prompts to test if a language model \"knows\" certain world knowledge facts. A problem in factual probing is that small changes to the prompt can lead to large changes in model output. Previous work aimed to alleviate this problem by optimizing prompts via text mining or fine-tuning. However, such approaches are relation-specific and do not generalize to unseen relation types. Here, we propose to use test-time augmentation (TTA) as a relation-agnostic method for reducing sensitivity to prompt variations by automatically augmenting and ensembling prompts at t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17121","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-26T03:41:32Z","cross_cats_sorted":[],"title_canon_sha256":"8e24deb283776317ff9fd95735a773305b837c7061930cc6ec22f073d4d0fca6","abstract_canon_sha256":"0655157d0cf3c0d14a3982a0acf5ef64de959e203f516a0d3d07e3c4f48e4e45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:19.953574Z","signature_b64":"H63Cafs7J6hCipWYjLQFWu+mX3SAvYNqoCaKRJVKY5PvUtN61H5IVahNg1KPkUa+GvVfaj8GBXXwMKRhaXwFDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83d10aaee72c14d8ac42c1eee6237feb7111c2a0dde01bd3cff4f4fc20400255","last_reissued_at":"2026-07-05T07:05:19.953169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:19.953169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Test-time Augmentation for Factual Probing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin Heinzerling, Go kamoda, Keisuke Sakaguchi, Kentaro Inui","submitted_at":"2023-10-26T03:41:32Z","abstract_excerpt":"Factual probing is a method that uses prompts to test if a language model \"knows\" certain world knowledge facts. A problem in factual probing is that small changes to the prompt can lead to large changes in model output. Previous work aimed to alleviate this problem by optimizing prompts via text mining or fine-tuning. However, such approaches are relation-specific and do not generalize to unseen relation types. Here, we propose to use test-time augmentation (TTA) as a relation-agnostic method for reducing sensitivity to prompt variations by automatically augmenting and ensembling prompts at t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17121","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17121","created_at":"2026-07-05T07:05:19.953227+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17121v1","created_at":"2026-07-05T07:05:19.953227+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17121","created_at":"2026-07-05T07:05:19.953227+00:00"},{"alias_kind":"pith_short_12","alias_value":"QPIQVLXHFQKN","created_at":"2026-07-05T07:05:19.953227+00:00"},{"alias_kind":"pith_short_16","alias_value":"QPIQVLXHFQKNRLCC","created_at":"2026-07-05T07:05:19.953227+00:00"},{"alias_kind":"pith_short_8","alias_value":"QPIQVLXH","created_at":"2026-07-05T07:05:19.953227+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.09351","citing_title":"Test-Time Augmentation for LLMs: When Input Diversity Beats Output Diversity at Matched Compute","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N","json":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N.json","graph_json":"https://pith.science/api/pith-number/QPIQVLXHFQKNRLCCYHXOMI375N/graph.json","events_json":"https://pith.science/api/pith-number/QPIQVLXHFQKNRLCCYHXOMI375N/events.json","paper":"https://pith.science/paper/QPIQVLXH"},"agent_actions":{"view_html":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N","download_json":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N.json","view_paper":"https://pith.science/paper/QPIQVLXH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17121&json=true","fetch_graph":"https://pith.science/api/pith-number/QPIQVLXHFQKNRLCCYHXOMI375N/graph.json","fetch_events":"https://pith.science/api/pith-number/QPIQVLXHFQKNRLCCYHXOMI375N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N/action/storage_attestation","attest_author":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N/action/author_attestation","sign_citation":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N/action/citation_signature","submit_replication":"https://pith.science/pith/QPIQVLXHFQKNRLCCYHXOMI375N/action/replication_record"}},"created_at":"2026-07-05T07:05:19.953227+00:00","updated_at":"2026-07-05T07:05:19.953227+00:00"}