{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:4FMB54MC46H4CUOVV45ZTBTN53","short_pith_number":"pith:4FMB54MC","schema_version":"1.0","canonical_sha256":"e1581ef182e78fc151d5af3b99866deecc8d44fc566d65d47b11f539b0c030ac","source":{"kind":"arxiv","id":"2001.08435","version":1},"attestation_state":"computed","paper":{"title":"The Pushshift Reddit Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.SI","authors_text":"Brian Keegan, Jason Baumgartner, Jeremy Blackburn, Megan Squire, Savvas Zannettou","submitted_at":"2020-01-23T10:31:29Z","abstract_excerpt":"Social media data has become crucial to the advancement of scientific understanding. However, even though it has become ubiquitous, just collecting large-scale social media data involves a high degree of engineering skill set and computational resources. In fact, research is often times gated by data engineering problems that must be overcome before analysis can proceed. This has resulted recognition of datasets as meaningful research contributions in and of themselves. Reddit, the so called \"front page of the Internet,\" in particular has been the subject of numerous scientific studies. Althou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2001.08435","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SI","submitted_at":"2020-01-23T10:31:29Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"cc5cfb27e4c35f1f7882f2c8e3475b9ffa0f722a72cf07c996d7085b3ceeaf31","abstract_canon_sha256":"d67c43704ac60cb4be51f8c9e5bb266161c29c2ff25247c4729cd003634c69e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:35:20.857557Z","signature_b64":"lqs1J26Utzk5HjcHZHS0RECB2RlipGEbYa+xtb2KkJplq2jhh9SvqD4nqsPPINX/NiN7q94VN8us8lG8Rr/3BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1581ef182e78fc151d5af3b99866deecc8d44fc566d65d47b11f539b0c030ac","last_reissued_at":"2026-07-05T00:35:20.857126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:35:20.857126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Pushshift Reddit Dataset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.SI","authors_text":"Brian Keegan, Jason Baumgartner, Jeremy Blackburn, Megan Squire, Savvas Zannettou","submitted_at":"2020-01-23T10:31:29Z","abstract_excerpt":"Social media data has become crucial to the advancement of scientific understanding. However, even though it has become ubiquitous, just collecting large-scale social media data involves a high degree of engineering skill set and computational resources. In fact, research is often times gated by data engineering problems that must be overcome before analysis can proceed. This has resulted recognition of datasets as meaningful research contributions in and of themselves. Reddit, the so called \"front page of the Internet,\" in particular has been the subject of numerous scientific studies. Althou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.08435","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.08435/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2001.08435","created_at":"2026-07-05T00:35:20.857185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2001.08435v1","created_at":"2026-07-05T00:35:20.857185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.08435","created_at":"2026-07-05T00:35:20.857185+00:00"},{"alias_kind":"pith_short_12","alias_value":"4FMB54MC46H4","created_at":"2026-07-05T00:35:20.857185+00:00"},{"alias_kind":"pith_short_16","alias_value":"4FMB54MC46H4CUOV","created_at":"2026-07-05T00:35:20.857185+00:00"},{"alias_kind":"pith_short_8","alias_value":"4FMB54MC","created_at":"2026-07-05T00:35:20.857185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26738","citing_title":"KARMA: Karma-Aligned Reward Model Adaptation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17187","citing_title":"PluRule: A Benchmark for Moderating Pluralistic Communities on Social Media","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21300","citing_title":"Explainable Disentangled Representation Learning for Generalizable Authorship Attribution in the Era of Generative AI","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2205.01068","citing_title":"OPT: Open Pre-trained Transformer Language Models","ref_index":283,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53","json":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53.json","graph_json":"https://pith.science/api/pith-number/4FMB54MC46H4CUOVV45ZTBTN53/graph.json","events_json":"https://pith.science/api/pith-number/4FMB54MC46H4CUOVV45ZTBTN53/events.json","paper":"https://pith.science/paper/4FMB54MC"},"agent_actions":{"view_html":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53","download_json":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53.json","view_paper":"https://pith.science/paper/4FMB54MC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2001.08435&json=true","fetch_graph":"https://pith.science/api/pith-number/4FMB54MC46H4CUOVV45ZTBTN53/graph.json","fetch_events":"https://pith.science/api/pith-number/4FMB54MC46H4CUOVV45ZTBTN53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53/action/storage_attestation","attest_author":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53/action/author_attestation","sign_citation":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53/action/citation_signature","submit_replication":"https://pith.science/pith/4FMB54MC46H4CUOVV45ZTBTN53/action/replication_record"}},"created_at":"2026-07-05T00:35:20.857185+00:00","updated_at":"2026-07-05T00:35:20.857185+00:00"}