{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NOVUSAVOZ3GWNS3YPVZKUQMC26","short_pith_number":"pith:NOVUSAVO","schema_version":"1.0","canonical_sha256":"6bab4902aececd66cb787d72aa4182d7a3185e655c66c5928c825a9726476faf","source":{"kind":"arxiv","id":"2411.17863","version":1},"attestation_state":"computed","paper":{"title":"LongKey: Keyphrase Extraction for Long Documents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cinthia Obladen de Almendra Freitas, Jean Paul Barddal, Jeovane Honorio Alves, Radu State","submitted_at":"2024-11-26T20:26:47Z","abstract_excerpt":"In an era of information overload, manually annotating the vast and growing corpus of documents and scholarly papers is increasingly impractical. Automated keyphrase extraction addresses this challenge by identifying representative terms within texts. However, most existing methods focus on short documents (up to 512 tokens), leaving a gap in processing long-context documents. In this paper, we introduce LongKey, a novel framework for extracting keyphrases from lengthy documents, which uses an encoder-based language model to capture extended text intricacies. LongKey uses a max-pooling embedde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17863","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-26T20:26:47Z","cross_cats_sorted":["cs.AI","cs.IR","cs.LG"],"title_canon_sha256":"42c008877542280884c0b656e4e0bbd74c8da3ff78c0d8e419f35eab2d680d21","abstract_canon_sha256":"5e71a3b0bbe0fd89c909f45e3ebb758290823312cad35df71d47b41a7c06ab05"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:55.352736Z","signature_b64":"r34ASWpj+iAicGxsd090i1HIjb/pdRUdUe+bzZyMzBHlnDuF8hlNwwQQqd5ajkfumJv6aiTY+0PxZowGBta3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6bab4902aececd66cb787d72aa4182d7a3185e655c66c5928c825a9726476faf","last_reissued_at":"2026-07-05T10:02:55.352285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:55.352285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongKey: Keyphrase Extraction for Long Documents","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cinthia Obladen de Almendra Freitas, Jean Paul Barddal, Jeovane Honorio Alves, Radu State","submitted_at":"2024-11-26T20:26:47Z","abstract_excerpt":"In an era of information overload, manually annotating the vast and growing corpus of documents and scholarly papers is increasingly impractical. Automated keyphrase extraction addresses this challenge by identifying representative terms within texts. However, most existing methods focus on short documents (up to 512 tokens), leaving a gap in processing long-context documents. In this paper, we introduce LongKey, a novel framework for extracting keyphrases from lengthy documents, which uses an encoder-based language model to capture extended text intricacies. LongKey uses a max-pooling embedde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17863","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17863/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17863","created_at":"2026-07-05T10:02:55.352353+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17863v1","created_at":"2026-07-05T10:02:55.352353+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17863","created_at":"2026-07-05T10:02:55.352353+00:00"},{"alias_kind":"pith_short_12","alias_value":"NOVUSAVOZ3GW","created_at":"2026-07-05T10:02:55.352353+00:00"},{"alias_kind":"pith_short_16","alias_value":"NOVUSAVOZ3GWNS3Y","created_at":"2026-07-05T10:02:55.352353+00:00"},{"alias_kind":"pith_short_8","alias_value":"NOVUSAVO","created_at":"2026-07-05T10:02:55.352353+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10716","citing_title":"Attention Expansion: Enhancing Keyphrase Extraction from Long Documents with Attention-Augmented Contextualized Embeddings","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26","json":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26.json","graph_json":"https://pith.science/api/pith-number/NOVUSAVOZ3GWNS3YPVZKUQMC26/graph.json","events_json":"https://pith.science/api/pith-number/NOVUSAVOZ3GWNS3YPVZKUQMC26/events.json","paper":"https://pith.science/paper/NOVUSAVO"},"agent_actions":{"view_html":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26","download_json":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26.json","view_paper":"https://pith.science/paper/NOVUSAVO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17863&json=true","fetch_graph":"https://pith.science/api/pith-number/NOVUSAVOZ3GWNS3YPVZKUQMC26/graph.json","fetch_events":"https://pith.science/api/pith-number/NOVUSAVOZ3GWNS3YPVZKUQMC26/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26/action/storage_attestation","attest_author":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26/action/author_attestation","sign_citation":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26/action/citation_signature","submit_replication":"https://pith.science/pith/NOVUSAVOZ3GWNS3YPVZKUQMC26/action/replication_record"}},"created_at":"2026-07-05T10:02:55.352353+00:00","updated_at":"2026-07-05T10:02:55.352353+00:00"}