{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2KAFBVSE6NCPT5AR3ZAWWOIFPL","short_pith_number":"pith:2KAFBVSE","schema_version":"1.0","canonical_sha256":"d28050d644f344f9f411de416b39057afaf0451e179e25c2e03de8017cbf6f52","source":{"kind":"arxiv","id":"2502.19413","version":2},"attestation_state":"computed","paper":{"title":"Project Alexandria: Towards Freeing Scientific Knowledge from Copyright Burdens via LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ameya Prabhu, Andreas Hochlehnert, Christoph Schuhmann, Gollam Rabby, Huu Nguyen, Jenia Jitsev, Ludwig Schmidt, Matthias Bethge, Nick Akinci, Robert Kaczmarczyk, S\\\"oren Auer, Tawsif Ahmed","submitted_at":"2025-02-26T18:56:52Z","abstract_excerpt":"Paywalls, licenses and copyright rules often restrict the broad dissemination and reuse of scientific knowledge. We take the position that it is both legally and technically feasible to extract the scientific knowledge in scholarly texts. Current methods, like text embeddings, fail to reliably preserve factual content, and simple paraphrasing may not be legally sound. We propose a new idea for the community to adopt: convert scholarly documents into knowledge preserving, but style agnostic representations we term Knowledge Units using LLMs. These units use structured data capturing entities, a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.19413","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-26T18:56:52Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7fd5ac3d6446bab7b9f5a74aaf9be6c2573d2161d070d596faa745dd4dd351d6","abstract_canon_sha256":"7056d71a45bced5e3c513c1f35dac3065981271610b1c50b376cd43f9bce8513"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:47.543925Z","signature_b64":"vtnjA71SIt04zMTZMikAwiLZB4Qv/lp/rN4kbXNmYnaE9ZU1AFVenMJxMDBLaiE8j6sNQZqAUYmMYqS3J1myDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d28050d644f344f9f411de416b39057afaf0451e179e25c2e03de8017cbf6f52","last_reissued_at":"2026-07-05T10:50:47.543461Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:47.543461Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Project Alexandria: Towards Freeing Scientific Knowledge from Copyright Burdens via LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ameya Prabhu, Andreas Hochlehnert, Christoph Schuhmann, Gollam Rabby, Huu Nguyen, Jenia Jitsev, Ludwig Schmidt, Matthias Bethge, Nick Akinci, Robert Kaczmarczyk, S\\\"oren Auer, Tawsif Ahmed","submitted_at":"2025-02-26T18:56:52Z","abstract_excerpt":"Paywalls, licenses and copyright rules often restrict the broad dissemination and reuse of scientific knowledge. We take the position that it is both legally and technically feasible to extract the scientific knowledge in scholarly texts. Current methods, like text embeddings, fail to reliably preserve factual content, and simple paraphrasing may not be legally sound. We propose a new idea for the community to adopt: convert scholarly documents into knowledge preserving, but style agnostic representations we term Knowledge Units using LLMs. These units use structured data capturing entities, a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.19413","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.19413/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.19413","created_at":"2026-07-05T10:50:47.543517+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.19413v2","created_at":"2026-07-05T10:50:47.543517+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.19413","created_at":"2026-07-05T10:50:47.543517+00:00"},{"alias_kind":"pith_short_12","alias_value":"2KAFBVSE6NCP","created_at":"2026-07-05T10:50:47.543517+00:00"},{"alias_kind":"pith_short_16","alias_value":"2KAFBVSE6NCPT5AR","created_at":"2026-07-05T10:50:47.543517+00:00"},{"alias_kind":"pith_short_8","alias_value":"2KAFBVSE","created_at":"2026-07-05T10:50:47.543517+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23628","citing_title":"AutoSchemaKG: Autonomous Knowledge Graph Construction through Dynamic Schema Induction from Web-Scale Corpora","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL","json":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL.json","graph_json":"https://pith.science/api/pith-number/2KAFBVSE6NCPT5AR3ZAWWOIFPL/graph.json","events_json":"https://pith.science/api/pith-number/2KAFBVSE6NCPT5AR3ZAWWOIFPL/events.json","paper":"https://pith.science/paper/2KAFBVSE"},"agent_actions":{"view_html":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL","download_json":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL.json","view_paper":"https://pith.science/paper/2KAFBVSE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.19413&json=true","fetch_graph":"https://pith.science/api/pith-number/2KAFBVSE6NCPT5AR3ZAWWOIFPL/graph.json","fetch_events":"https://pith.science/api/pith-number/2KAFBVSE6NCPT5AR3ZAWWOIFPL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL/action/storage_attestation","attest_author":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL/action/author_attestation","sign_citation":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL/action/citation_signature","submit_replication":"https://pith.science/pith/2KAFBVSE6NCPT5AR3ZAWWOIFPL/action/replication_record"}},"created_at":"2026-07-05T10:50:47.543517+00:00","updated_at":"2026-07-05T10:50:47.543517+00:00"}