{"as_of":"2026-08-13T00:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8e4b587fe283e33a7633b436ed6119695d13bb88367be730b2cfde8fc90ebea","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T14:44:59.796239Z","state":"measured"},{"denominator":58,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":58,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.14957/citation-record","integrity":"/paper/2411.14957/integrity","json":"/paper/2411.14957/citation-record.json","paper":"/paper/2411.14957"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.591652Z","title":"Form2Seq : A framework for higher-order form structure extraction","venue":null,"work_id":"288411b6-1030-4e0d-b902-73493a50f13c","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.554116Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:bfb79dc0e03278e74aa05afcd823348dcba7422deca7cb910af7b87fa1c2e432","observation_id":"5d31a3c0-be01-4d2d-ae62-0c45da68db94","resolution":{"observed_at":"2026-08-12T14:45:00.596276Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.577445Z","title":"Data protection, 2024","venue":null,"work_id":"b2ca982d-3037-48bf-9c14-63eb19fa6c11","year":2024},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.559224Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:18dd8b1e6a5b1b32f6a579f0f5afec17c9f4709c9f652990a78fdad5ca70e6af","observation_id":"f0688c60-6869-4895-af93-ec34e1189337","resolution":{"observed_at":"2026-08-12T14:45:00.582008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.563767Z","title":"The claude 3 model family: Opus, sonnet, haiku,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.563767Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:d17557b5831de19c610f58614546ab9bcd16baa56c2a44508277d0d3beaa2584","observation_id":"969f4d54-cb0a-46df-8c34-ffd5c20d56a4","resolution":{"observed_at":"2026-08-12T14:44:59.563767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.554182Z","title":"Introducing the next generation of claude, 2024","venue":null,"work_id":"991fd242-22cb-4386-97dc-7d38a74ef4b1","year":2024},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.568285Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:e8fad7da0e359561725f608d76973db1ae0f6da2d96599f4a07105e88f3b4945","observation_id":"cd6701fd-e7cb-400e-b1f1-3e200082221d","resolution":{"observed_at":"2026-08-12T14:45:00.558517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.538646Z","title":"Prompt engineering, 2024","venue":null,"work_id":"27cfebcf-d49c-4bde-8c0b-2841e0c468c7","year":2024},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.572780Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:61d0def25a6b4081b1152cc91d33c35a156f1af470c4bb187e16e5fd13e6588e","observation_id":"14423e75-5e59-4599-9dc2-78bf1cd1bd25","resolution":{"observed_at":"2026-08-12T14:45:00.542804Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.525189Z","title":"Manmatha","venue":null,"work_id":"68c8313b-e5b9-4c0b-bd44-02d948f4c34d","year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.577688Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:aba7385df21b1c523dbd9c97bf03a4e5198956b4c157bf8f3e41444d25ee6982","observation_id":"dc1a055f-f90e-43d4-b0fe-b886d927305d","resolution":{"observed_at":"2026-08-12T14:45:00.529622Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.511458Z","title":"Wukong-reader: Multi-modal pre-training for fine-grained visual document understanding, 2022","venue":null,"work_id":"cb04c7e6-bca2-4b53-824f-60bfde5f976d","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.582258Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:09756b54ad02e949b0893545431788f9e1207c92e00ecc894455b34185c05307","observation_id":"19edde72-2833-48bc-bfa9-5d09f6b0085c","resolution":{"observed_at":"2026-08-12T14:45:00.516053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.497197Z","title":"Unilmv2: Pseudo-masked lan- guage models for unified language model pre-training, 2020","venue":null,"work_id":"c3292d04-b2e0-4a28-9d33-58f89ebee385","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.587219Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:07d742c22e1a5b67b9f20b5a86faaac7b676d43db7e58252601ae95d8d66db74","observation_id":"739c429b-932b-441b-b12a-e09549a628d3","resolution":{"observed_at":"2026-08-12T14:45:00.501847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.483646Z","title":null,"venue":null,"work_id":"f79b30ed-bb12-4f6f-b142-596a4d9aa7f9","year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.591460Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:2f6833d3e685e61e52ee934325786c868f3cfb86f459aea95fd5e47ab8e3c6c2","observation_id":"634fcd91-c06b-4a0a-923b-6fb233f6128b","resolution":{"observed_at":"2026-08-12T14:45:00.487979Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.469903Z","title":null,"venue":null,"work_id":"f8ef6a09-b852-4916-8813-0962fc5c1776","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.595444Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:56df5fcc8b625f68103c11e7e23b5dfd6dca88fe607ade03ddfe3d2b6cea6e48","observation_id":"6582a774-a4ab-40ab-9bf4-a2eab4a79659","resolution":{"observed_at":"2026-08-12T14:45:00.474371Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.456013Z","title":"Gonzalez, Ion Stoica, and Eric P","venue":null,"work_id":"e4611ab9-2bb6-45e6-94d1-5a8b3bf4d45e","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.599696Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:9e864b6a1a08d37fe48b7842b25e5dc62d78198e36007ba4e2418b2ce93c8a8d","observation_id":"642fa838-8a97-46f3-ba5a-43c602a88a96","resolution":{"observed_at":"2026-08-12T14:45:00.460416Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.441701Z","title":null,"venue":null,"work_id":"70aac3f1-8659-4440-993a-6eb37d39ba72","year":2013},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.603663Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:a695739b7ffcbfd1ec289920b78094470953c5eadf1948b11b5b0f36b810f7d1","observation_id":"34a96809-63f9-4272-8787-1b3607e3c0a1","resolution":{"observed_at":"2026-08-12T14:45:00.446143Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.427385Z","title":"Binarized neural networks: Training deep neural networks with weights and activations constrained to +1 or -1, 2016","venue":null,"work_id":"479e0dfd-943e-4497-a431-e7a92d285107","year":2016},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.607692Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:d42676b438e0e1c449ab6c759ab35d1545e5cdebc1e72d7647f8f8360e137628","observation_id":"353eca23-7166-4f60-9244-4a3a2b251558","resolution":{"observed_at":"2026-08-12T14:45:00.432335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.411594Z","title":"Denk and Christian Reisswig","venue":null,"work_id":"75dfc069-28f9-4cea-8a5d-a97d04b9e435","year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.611805Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:d64473456bc50ece1fd7a6450d2a25d392418a1850e6b5e91ed3b0d7cc9efebb","observation_id":"3908b9f7-0209-4bc1-af60-9820931bfc20","resolution":{"observed_at":"2026-08-12T14:45:00.416613Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.396477Z","title":"Bert: Pre-training of deep bidirectional trans- formers for language understanding, 2019","venue":null,"work_id":"8c687cc2-30ca-4bdb-a306-45ba1c1272d1","year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.615889Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:b9df1536d5fba734f5a478777e720cf689446271dedceeae3780b5b7c7ebeadd","observation_id":"d4a799f4-2594-45de-a892-726f7cde9c9d","resolution":{"observed_at":"2026-08-12T14:45:00.401341Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.381953Z","title":"Docparser: End-to-end ocr-free information extraction from visually rich documents, 2023","venue":null,"work_id":"d8744b85-4022-425c-a4bf-ede58cde4b2a","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.620019Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:853ef0c34baee9f76e5ebb4df1fb3891b2aca974c727ab22acb1bf9b08170a8f","observation_id":"ebbeb66c-b529-474f-ac5f-2b2989aa4484","resolution":{"observed_at":"2026-08-12T14:45:00.386865Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.368556Z","title":"Haralick, and I.T","venue":null,"work_id":"20482af7-7b42-4f14-9e79-57faf3ffdff8","year":null},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.623762Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:71521b0284b73b4995d82b9e2e651dbd1adb7989a8d68388968862814b611c04","observation_id":"c7447e3e-fd5f-4e0b-9a64-17da21d3eab6","resolution":{"observed_at":"2026-08-12T14:45:00.372423Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.355352Z","title":"A table detection method for pdf documents based on convo- lutional neural networks","venue":null,"work_id":"341d8c5f-4812-45b3-957f-e0e9cde67c56","year":2016},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.628249Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:2f1c7e5b4176e0debbb50f2331cab0fb7e4af7fbe349f4421b9f8f599d3cca98","observation_id":"f49174e5-b8ad-4d43-887d-36c84830980a","resolution":{"observed_at":"2026-08-12T14:45:00.359563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.341439Z","title":"Ocr with tesseract, amazon textract, and google document ai: a benchmarking experiment","venue":null,"work_id":"bca0e354-81c1-46a0-9f44-8dad88828aeb","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.632632Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:5b592625d01616c3a8461cdaf05ee6aad04f4e65e95e8a811b95f9cf0af0762c","observation_id":"37ddfbb0-d5f9-4b12-9f22-af3d742370b3","resolution":{"observed_at":"2026-08-12T14:45:00.346253Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.636887Z","title":"Distilling the knowledge in a neural network, 2015","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.636887Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:bb4d8f40c7736203c8879fdaf6d4bf03c3a5887eb4975f8f249579ac50d7b861","observation_id":"8376e333-6957-4c9c-8217-783bfe938823","resolution":{"observed_at":"2026-08-12T14:44:59.636887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.641189Z","title":"Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen- Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.641189Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:7cfda9a9207fd4bd76cb79f864774a91956d05f7497f4db4fdfcd71089bc8a73","observation_id":"51fb9bee-48c3-47b5-9a1e-3f2105e24667","resolution":{"observed_at":"2026-08-12T14:44:59.641189Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.307842Z","title":"Cre- ating something from nothing: Unsupervised knowledge dis- tillation for cross-modal hashing, 2020","venue":null,"work_id":"e9ccf08d-3caf-412d-9a7f-c581b0d3e7aa","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.645709Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:ca33b19cec176212b30327a2dc84331fb4f872d2eb83bc6c8d2bc60ebf0458d3","observation_id":"07d7e8c0-94e4-4dec-b1bb-addf2d59dc24","resolution":{"observed_at":"2026-08-12T14:45:00.313050Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.292719Z","title":"Layoutlmv3: Pre-training for document ai with unified text and image masking, 2022","venue":null,"work_id":"feb1f175-adad-4d99-afb9-0b6bbc872fb6","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.649936Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:f2863a57dc67669f2855ede7a08c4b58442e07acb65746a6b40bd7a8f2a298c5","observation_id":"e6126d0f-16fe-4100-a7c8-eba01a4114f5","resolution":{"observed_at":"2026-08-12T14:45:00.297324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.279252Z","title":"Spatial dependency parsing for semi-structured document information extraction, 2021","venue":null,"work_id":"e46f03c6-0815-4ea4-a7bf-4e5a5aba9892","year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.654042Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:b9d0bfbb9ad60e6acb3a6c14780b3b4cd8d295a8f658715d87c6054dbda4e031","observation_id":"9e5e1688-6369-4d38-827e-1808a561633a","resolution":{"observed_at":"2026-08-12T14:45:00.283659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.265573Z","title":"Funsd: A dataset for form understanding in noisy scanned documents, 2019","venue":null,"work_id":"904ddf10-29d3-44af-8936-1faf3c65766a","year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.658270Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:aecd9298a6cfc89c8abe8fefc0806910cfe39ff90124f898737209f080a7d0bc","observation_id":"a6dd6fd0-22b6-40b3-af57-48ce230cae56","resolution":{"observed_at":"2026-08-12T14:45:00.269920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.252953Z","title":null,"venue":null,"work_id":"b961b56f-eb24-4111-8aac-edc60f06689b","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.662597Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:382e85562c78d1f0d2c04c80f59cf69b48d7c19dc82b61de155043226fe8638f","observation_id":"31e5e396-2550-4bf4-8f17-c8b0e008d0b1","resolution":{"observed_at":"2026-08-12T14:45:00.256806Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.239692Z","title":"Chargrid: Towards understanding 2d documents, 2018","venue":null,"work_id":"e4da0c74-e216-464f-b47e-5442c04db952","year":2018},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.666835Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:314f49f1b1fc698ddd45bcc08983423a7d24df896a8286b27ee0f312448b762d","observation_id":"ef68e150-f49e-4636-94eb-0b35be8bd1f1","resolution":{"observed_at":"2026-08-12T14:45:00.244009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.671051Z","title":"A diagram is worth a dozen images, 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.671051Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:e00004c9673f12048a518572b5fc5f1b839e43365f01f8de84822eb8d23ad64e","observation_id":"3e6c28de-0d98-42ec-9100-300437f78235","resolution":{"observed_at":"2026-08-12T14:44:59.671051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.217643Z","title":"Formnetv2: Multimodal graph contrastive learning for form document information extraction, 2023","venue":null,"work_id":"96e55d1d-44d9-4a46-8ee3-f144833363fe","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.675495Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:3daf4519fab7db3cd24e8bda2253a6004c66e7a1c4d964b154fcbf1ba3f0bf08","observation_id":"c5ba4a83-72d6-45d3-80a4-2c382c0e5db4","resolution":{"observed_at":"2026-08-12T14:45:00.222100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.204029Z","title":"Lewis, G","venue":null,"work_id":"fe58781e-b1fa-4334-8959-ac51bf34de5b","year":2006},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.680112Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:5f1da7e89e1fafb615fcf9dafb6434f1d856bd0cc1aadffdcf0a79f72c681dec","observation_id":"d1c84f92-2054-45cb-8d9a-5ee887e2bb1c","resolution":{"observed_at":"2026-08-12T14:45:00.208223Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.684430Z","title":"Improved baselines with visual instruction tuning, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.684430Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:ab5d9e468f2d20c12409b07ac199afaf7198c3440b54ebfd861d3a526ec743ac","observation_id":"f64a1341-8213-4e38-85e8-61bbd1ffba4c","resolution":{"observed_at":"2026-08-12T14:44:59.684430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.181156Z","title":"Liu, Kevin Lin, John Hewitt, Ashwin Paranjape, Michele Bevilacqua, Fabio Petroni, and Percy Liang","venue":null,"work_id":"99d9caac-c676-4dc4-8b82-022b97c02e35","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.688496Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:ae724bf3be7e0e08e72c561c74595ed8d52f7e578564862d3ec44be0f77fb0ca","observation_id":"22e20805-2216-4cfa-b015-4e3d8024073b","resolution":{"observed_at":"2026-08-12T14:45:00.185835Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.692672Z","title":"Roberta: A robustly optimized bert pretraining approach, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.692672Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:ee174a6c5240af1463022be51e94a1f3d8263952fbb2a1cf0a7f22722796daa5","observation_id":"0672716b-4602-4034-a689-3fdbda6d1896","resolution":{"observed_at":"2026-08-12T14:44:59.692672Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.158076Z","title":"Repre- sentation learning for information extraction from form-like documents","venue":null,"work_id":"2b47e8df-d6d7-4e01-9511-33c20720eeef","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.696919Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:69e5cc0b03c8100c0cd9dbd57a9086b89e423140176406075621d0320b28862c","observation_id":"188b7b9a-4903-42e0-9b30-a08ee14a4d1d","resolution":{"observed_at":"2026-08-12T14:45:00.162465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.144250Z","title":"Marinai, M","venue":null,"work_id":"bae28621-4600-44c3-9531-72d2535e3993","year":null},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.701062Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:3917cdc2658a5475b20935bbefb65b31fc81d216eab59dbddabc8ed54fa7536c","observation_id":"fa6efbeb-5b2b-4d73-9d81-bc591faeebec","resolution":{"observed_at":"2026-08-12T14:45:00.148483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.130750Z","title":"Chartqa: A benchmark for question an- swering about charts with visual and logical reasoning, 2022","venue":null,"work_id":"d48604bf-5f3a-4f49-95ca-a6bf058d941c","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.705456Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:a206ffeceaa9363f6c7e4722a57aa8e65c8d6e285683f5d7ca16a9264f3efa2b","observation_id":"8a536668-a449-43cd-ab84-3a637ddc8027","resolution":{"observed_at":"2026-08-12T14:45:00.135258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.116189Z","title":null,"venue":null,"work_id":"8e1e6573-df03-4c66-a6e6-df524606a595","year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.709840Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:f0e9eb59571f1efb3a0725eaeb4c87300317cfb2cfd0cf1b8823a9cb340f301e","observation_id":"6c7d6891-2c9b-4471-81c4-78a0d7ae974f","resolution":{"observed_at":"2026-08-12T14:45:00.121435Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.102144Z","title":"Open language learning for information extraction","venue":null,"work_id":"fb4507bf-35f9-4ddb-8d16-40bd6142ff30","year":2012},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.714300Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:f8716e40e15890f85e5f3ca568afb90ae9440f7565232bfd621301ed67930f87","observation_id":"7c13a8b8-f3e5-49d0-9fa4-bbc19ecef613","resolution":{"observed_at":"2026-08-12T14:45:00.106520Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.087559Z","title":"doctr: Document text recognition","venue":null,"work_id":"f10f3575-7c9d-41a0-afa4-64944b469b18","year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.718391Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:837a527902a4102a8aac3b7c3ce86d8ae321a3dba07bfa6e9e99240f42295a48","observation_id":"a4ccb2ad-444f-43ed-ac53-8cff4fee6cf7","resolution":{"observed_at":"2026-08-12T14:45:00.091979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.074169Z","title":"O’Gorman","venue":null,"work_id":"77108509-a99c-44d4-b635-cef938f9e325","year":1993},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.722598Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:6a0cc95c542c7731bf0b84ea4dbb62113907352288fab050ca26b8e854304c0f","observation_id":"ffeb3918-4313-4d87-9519-cb4e1d539e11","resolution":{"observed_at":"2026-08-12T14:45:00.078048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.059532Z","title":"Gpt-4 is openai’s most advanced system, 2023","venue":null,"work_id":"7a2a88b6-6daa-487e-b56c-f47c9ff2b6f5","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.726706Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:471d009e3d9cef711d54be005de683e27b35a21500c4e69b11e0c85dd8d2128f","observation_id":"6a2380ef-13aa-4fe6-9b7f-e95b979ad99c","resolution":{"observed_at":"2026-08-12T14:45:00.064584Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.044092Z","title":"Revisiting the tree edit distance and its backtracing: A tutorial, 2022","venue":null,"work_id":"cc2d89e4-6ae4-476c-aec5-1c6a7e041933","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.730814Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:242e84da2b27ae87ab2aeeb81e4861babff905b7389ff552cef685edd9b07ce4","observation_id":"ef3566d2-4264-4d9b-949a-c0b034410c3f","resolution":{"observed_at":"2026-08-12T14:45:00.048737Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.028585Z","title":"Cloud- scan - a configuration-free invoice analysis system using re- current neural networks, 2017","venue":null,"work_id":"2843db11-6b58-4f41-8a29-700fcb1731be","year":2017},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.735003Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:adbbeff529c16625b31f460b10c4594ccb64509eb9e87a5ce9adb48d46eb6126","observation_id":"78894fef-321f-4922-8358-3973b32ebecd","resolution":{"observed_at":"2026-08-12T14:45:00.034542Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:45:00.013065Z","title":"Cord: A con- solidated receipt dataset for post-ocr parsing","venue":null,"work_id":"2cd07487-9c38-4553-a6d5-eb919e81470d","year":2019},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.739100Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:71927308e72e749786d17515cd3d13ea38d3371efc1fe4847b699cc797509498","observation_id":"ecaa873f-8758-4d08-8590-3e534a3ea5a3","resolution":{"observed_at":"2026-08-12T14:45:00.018187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.999182Z","title":"Ernie-layout: Layout knowledge en- hanced pre-training for visually-rich document understand- ing, 2022","venue":null,"work_id":"b12dd8fe-dc26-44be-9602-e8c370a918c4","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.743543Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:a4500cc599fd97826699235e5558cbc96a1013172394dcf34cc6bf9eb1dd4569","observation_id":"acafe7c5-00db-46ed-ba49-ffc5faa2fe72","resolution":{"observed_at":"2026-08-12T14:45:00.003908Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.984946Z","title":"Lmdx: Language model-based document information ex- traction and localization, 2024","venue":null,"work_id":"f8fd789a-1c74-41b8-96c6-bee2c14bef29","year":2024},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.747632Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:9478b13f3081ce8a8b19803b76ec3a52dc970c66a4553ebe4c3a4b3c99ea0f0b","observation_id":"08d06f40-fa1d-428e-9d11-6cf695eb09c5","resolution":{"observed_at":"2026-08-12T14:44:59.989477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.751400Z","title":"Learning transferable visual models from natural language supervision, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.751400Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:4e1d341b2663cbf24eddf2e8991f3c4156c3bbff6202ebc3053ac7fe5a828698","observation_id":"967348e0-bd89-4926-8d55-37c2f4bafb9a","resolution":{"observed_at":"2026-08-12T14:44:59.751400Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.961790Z","title":"Xnor-net: Imagenet classification using bi- nary convolutional neural networks, 2016","venue":null,"work_id":"b59034e6-d2e1-4312-b172-f1ff00e3b5c4","year":2016},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.755083Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:174c54f29ba2a1177e1fa6c5da0f92046023b86b557b2b4d86cf484210740a73","observation_id":"d815c8f5-8c8d-4d18-a8cc-d6d196ab1f02","resolution":{"observed_at":"2026-08-12T14:44:59.967070Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.946951Z","title":"Simon, J.-C","venue":null,"work_id":"1dbd3c98-137e-47ef-ae88-7cf74114febb","year":null},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.758835Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:6dd08f0d30afbed2a918c0d8b4db48e972b6e3074fb3fe9befad944a3ec14602","observation_id":"f6e6e4d8-2807-409c-88e3-2553b7b1ca52","resolution":{"observed_at":"2026-08-12T14:44:59.951961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.762683Z","title":"Llama 2: Open foundation and fine- tuned chat models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.762683Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:ddbb8e4edc1cb5f285484807ec38c34418bf9dee1f6b649f31abe4ca7f3fc041","observation_id":"cad7e608-e92a-47be-8b1d-04a0dcf130bc","resolution":{"observed_at":"2026-08-12T14:44:59.762683Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.924868Z","title":"Layout and task aware instruction prompt for zero-shot document image question answering, 2023","venue":null,"work_id":"6ea16072-ae71-4905-b1b2-0b42fd2ea898","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.766511Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:8befa5839be640427d637c11a8495aa8ccad4f3db0b51593a7b03e8e8a95eb21","observation_id":"9d633094-348b-4e56-b5a8-19debecbef84","resolution":{"observed_at":"2026-08-12T14:44:59.929187Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.910472Z","title":"Cogvlm: Visual expert for pretrained language models, 2024","venue":null,"work_id":"fe2db9ad-8144-4b60-b75f-7efcb36c87d8","year":2024},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.770271Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:f2d9f9cf0be20e4e8060c8f4824664fe455b598f7e5b14ce1102decf39277fa8","observation_id":"1ade8e34-f35a-4089-b70a-91659233de26","resolution":{"observed_at":"2026-08-12T14:44:59.915013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.896166Z","title":"Cnnpack: packing convolutional neural networks in the frequency domain","venue":null,"work_id":"514cf6b5-ea4e-4358-966e-3863b11a5fdd","year":2016},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.774443Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:c134321f507b37e98794cf67136e0b4bf66e5563105d7478479867f559d95cdd","observation_id":"a09d24d2-de2b-4e75-a13b-7c5560ec6727","resolution":{"observed_at":"2026-08-12T14:44:59.900947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.881502Z","title":"Queryform: A simple zero-shot form entity query framework, 2023","venue":null,"work_id":"ab4cef9c-12c1-49e2-90fd-8161cb942ed1","year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.778803Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:64bea52f6a61dd9c3a52e9fcead46979252656ba8e216a5ea16858f513d6d705","observation_id":"c031da7e-21ba-419f-8970-8f457e282546","resolution":{"observed_at":"2026-08-12T14:44:59.886091Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.783046Z","title":"Chain-of-thought prompting elicits reasoning in large language models, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.783046Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:bf0a91ee35efaf6448a95508bca862de9a1fc7c9635588564e0008f8335640bc","observation_id":"18dcfba2-5e07-48f2-8c6c-d6d3f29b3342","resolution":{"observed_at":"2026-08-12T14:44:59.783046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.857632Z","title":"Layoutlm: Pre-training of text and layout for document image understanding","venue":null,"work_id":"94679c60-f0a0-4b6e-8efe-d2c5f6f9ab89","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.787596Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:04b29b15d179525163325191439a1c3fbbc2a4680beec16862587e9490ced5d0","observation_id":"e9f5aa14-7f6a-4827-9071-c78e5c1b7c30","resolution":{"observed_at":"2026-08-12T14:44:59.862305Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.843315Z","title":"Layoutlmv2: Multi-modal pre-training for visually-rich document under- standing, 2022","venue":null,"work_id":"b2abb10c-f76b-49cd-a9e3-21f57b17fbd5","year":2022},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.792060Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:197041d5c0b02c7502b9c9c4cf2dc0f4c731bb4441f942e2e008694d8c9af2ec","observation_id":"1360fab2-f0dd-4841-a859-b2b3a464840b","resolution":{"observed_at":"2026-08-12T14:44:59.848026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:44:59.826729Z","title":"Data-free knowledge amalgamation via group-stack dual-gan, 2020","venue":null,"work_id":"9d675aa4-d54d-43fc-a3e2-3e6cc50b4cdd","year":2020},"citing_paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-12T14:44:59.796239Z"},"links":{"citing_paper":"/paper/2411.14957"},"observation_digest":"sha256:bead620f451089ad930d5b55de11875471255b1b1e781f976e899099b86347a8","observation_id":"f458db77-eaea-46c7-af01-01c4e2a876b3","resolution":{"observed_at":"2026-08-12T14:44:59.833301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2411.14957","last_updated":"2024-11-25T09:47:20Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-12T14:38:26.815181Z","submitted_at":"2024-11-22T14:16:09Z","title":"Information Extraction from Heterogeneous Documents without Ground Truth Labels using Synthetic Label Generation and Knowledge Distillation"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":14,"verified_exact":0,"verified_fuzzy":44},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 0 inbound Pith citation observations for arXiv:2411.14957."}