{"as_of":"2026-08-10T00:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:083fc16dca23bfcdd54c0732312805bb616260bc0da6ba64b11f9894838e3597","coverage":[{"denominator":29,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":29,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-12T06:14:42.321767Z","state":"measured"},{"denominator":29,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":29,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.02904/citation-record","integrity":"/paper/2607.02904/integrity","json":"/paper/2607.02904/citation-record.json","paper":"/paper/2607.02904"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Depressive disorder (depression),","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:ef42e3375cb7b71249fc245cd2bd8f8765300fa233ee3c795c0de02c7556219e","observation_id":"de41dc07-afb8-48ad-9bf8-2c0f5c33ba3e","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"A review of depression and suicide risk assessment using speech analysis,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:dc153b6670bf483767da7fc5cfe635c501bd51f3f38a68a23867ec2202b6fb5c","observation_id":"81be1974-b3f1-4ddc-9123-4449b0720bbd","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Automated assessment of psychiatric disorders using speech: A systematic review,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:a7f07f9416830dc1966201121517a46c68d1c3b80af970373f05a33c00f10c94","observation_id":"3e73c4da-4309-4495-a98a-5509ad875ca2","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"WavLM: Large-scale self-supervised pre-training for full stack speech processing,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:4fe799ee41776cdea12192d8da2adf5e9d6b0a9faf2bee969b608bb910f081b3","observation_id":"73b5a47f-5e83-4829-a16a-a55e1aca617b","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"The distress analysis interview corpus of human and computer interviews,","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:f2d0d5660bdf77fac1fc80843fbeb04df107908405150ba78cbbe7d3392bbda4","observation_id":"8511b00e-59bb-4ae5-a6e3-df843c6306be","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.09283","last_updated":"2020-03-05T03:43:31Z","snapshot_observed_at":"2026-08-08T16:59:40.659869Z","submitted_at":"2020-02-20T09:40:39Z","title":"MODMA dataset: a Multi-modal Open Dataset for Mental-disorder Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.09283","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"MODMA dataset: A multi-modal open dataset for mental-disorder analysis,","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2002.09283","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:1994ab724e28aec543153bfdd942e16565db3d8c034d03aa8bab863f294d8981","observation_id":"4f1af0eb-1202-49b0-9160-a424fcea8c81","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.07447","last_updated":"2021-06-14T14:14:28Z","snapshot_observed_at":"2026-08-08T09:38:45.288574Z","submitted_at":"2021-06-14T14:14:28Z","title":"HuBERT: Self-Supervised Speech Representation Learning by Masked Prediction of Hidden Units","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.07447","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2106.07447","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:09a6e3413495aeed000dd58b01c13ea35eb6f0210c7492b89a022a8520a939d6","observation_id":"28de5576-2bb0-49ca-9446-7276fb253d47","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Robust wav2vec 2.0: Analyzing domain shift in self-supervised pre-training,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:96c16b47216c7a7ecf5c6fd3eac4853efa241bb420e499e86b2b77b97145ea1e","observation_id":"65531f6e-da62-45c0-b5b8-2a255405c98c","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"data2vec: A general framework for self-supervised learning in speech, vision and language,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:46b6a3e7c0f03945b831ce490bfa157278cf9895dd3da19b4cdd2468db7379e6","observation_id":"905556f7-761f-4022-846c-18c036a4ba48","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2111.09296","last_updated":"2021-12-16T18:29:22Z","snapshot_observed_at":"2026-07-06T12:09:37.468149Z","submitted_at":"2021-11-17T18:49:42Z","title":"XLS-R: Self-supervised Cross-lingual Speech Representation Learning at Scale","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2111.09296","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"XLS-R: Self-supervised cross-lingual speech representation learning at scale,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2111.09296","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:37511ec332d2a6f1b4892c1ba5f97dc4f518b6a9e28169d087d0866b57f90f6d","observation_id":"5a231095-8343-43cd-9b28-7b84916812e4","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Inves- tigation of layer-wise speech representations in self-supervised learning models: A cross-lingual study in detecting depression,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:116124e4cdd8d47e66e4a032e68125c4f7ea2b83c8613fb8197abb377d92b232","observation_id":"d3e2531e-8dc2-4a71-9c8f-b5a86be9239b","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"SUPERB: Speech processing universal performance benchmark,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:473daac4517bbf7a79cc2f1403ac1b88d045a387a6dc56894c7a962448de12d3","observation_id":"216bf54f-d6ec-4ee1-8a76-8e3e26cc3b3b","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.13018","last_updated":"2024-03-12T07:13:02Z","snapshot_observed_at":"2026-08-06T13:32:53.436495Z","submitted_at":"2024-02-20T14:00:53Z","title":"EMO-SUPERB: An In-depth Look at Speech Emotion Recognition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.13018","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"EMO-SUPERB: An in-depth look at speech emotion recognition,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2402.13018","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:8c74d1fc31daed37b868ddd5777af3de7e4a05b1d00a0baedfe9115a5b9d8109","observation_id":"2499a599-837b-437c-b8b3-c8f878df8995","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.16920","last_updated":"2025-04-30T13:16:09Z","snapshot_observed_at":"2026-08-08T18:40:06.346575Z","submitted_at":"2024-09-25T13:27:17Z","title":"Cross-Lingual Speech Emotion Recognition: Humans vs. Self-Supervised Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.16920","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Cross- lingual speech emotion recognition: Humans vs. self-supervised mod- els,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2409.16920","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:3fd5f549b96d08fc02846a771a0e49ea28b6b6c7a959c3c0edd0cf555bdcb79a","observation_id":"be9de19d-e7db-4d1b-b876-96b7518b54f8","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Association between acoustic speech features and non-severe levels of anxiety and depression symptoms across lifespan,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:4efd2bbac9a4f70faa7e59425e2c74e865edbf8d066efdba48608a6aa0194293","observation_id":"fac8b026-cda0-476c-815a-b3e279d781a1","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Detecting depression with audio/text sequence modeling of interviews,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:7a53626b2ab691290f79fc5b5186bd7381129c03371712ba0e314476df533945","observation_id":"b803d68e-052b-43a7-8497-a8c47c319494","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"DEPA: Self-supervised audio embedding for depression detection,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:39d6e37edb196f32e41ffff317aaf041ec0b99a88d3b8d4b7497272162de9413","observation_id":"aac68f63-a811-498c-a970-1d96ab5b18c9","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.03502","last_updated":"2021-04-08T04:31:58Z","snapshot_observed_at":"2026-07-06T10:57:30.968546Z","submitted_at":"2021-04-08T04:31:58Z","title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.03502","snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Emotion recognition from speech using wav2vec 2.0 embeddings,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"cited_paper":"/paper/2104.03502","citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:a2daf481945501eecfbda3bb0d50b59e51098c0507a07ea9199270cd46ab7664","observation_id":"1c122ce8-1fa7-490d-8cb0-98cddfc424c3","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Tracking depression-related mental state using multimodal analysis,","venue":null,"work_id":null,"year":2013},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:068d950a4c3ca193c3652dfe81c6adf7d58bc8f4148a3156abea8d9b6bdb9195","observation_id":"3e669324-d417-452c-8453-f04e5dc5c7f2","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"A VEC 2019 Workshop and Challenge: State-of-Mind, Detecting Depression with AI, and Cross-Cultural Affect Recognition,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:32cc24a24fb949c06942e3ac0fd59876e194dc6641a497cd8407b0402740a5ea","observation_id":"5839823f-72e5-4472-a530-a8c012379279","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Speaker normalization for speech-based depression detection,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:463d04385a41f73c92c12ca000b8ba4efbd3346e287494908dc485505472f081","observation_id":"5d493f89-8b4c-4140-9310-868da5f53d7e","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"DepAudioNet: An efficient deep model for audio based depression classification,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:417d2de354296aa38a57cc3cc11a8907522082353101fdd0e31a7c8b4648dec9","observation_id":"48c12994-19b5-4d04-9eb9-a327c8075211","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Automatic speech emotion recognition using recurrent neural networks with local attention,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:3ef951e30d98d33df9e3adb1b3e8f1ef9c2d491443309462b05adbf9b8654282","observation_id":"9821d2d4-1c28-47d2-ad7b-327fde51af75","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"NetVLAD: CNN architecture for weakly supervised place recognition,","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:71f08f14fccf704895d59254a1b3a449084b41d6e2217c7f13136e7841deec0d","observation_id":"d65f4b48-aaf3-4486-bf76-0bd27993ef42","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Attentive statistics pooling for deep speaker embedding,","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:8b3432267b215b4a1952a53bec0b0e53ace37cc3bdc9be581bd767760dbac29e","observation_id":"b984236a-e221-4729-b69d-ddc92bdf11f7","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"BERT: Pre-training of deep bidirectional transformers for language understanding,","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:c79ed8cb438cc3544644b81fcb8e9599a5ab59e070f827e76fe706895695f169","observation_id":"451d3020-6fba-43ce-8160-632385d0d0c3","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"The INTERSPEECH 2021 computational paralinguistics challenge,","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:c07a91918a26cdf2b8474518a2ee33bbb86692c90aba02a6e5cf10913331e2a8","observation_id":"97bfd86e-6960-4b42-9e0e-37743235eece","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Prevalence of neural collapse during the terminal phase of deep learning training,","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:01f805c07578cfb40649a885d66d997b6f4f4e289108df26d3a66b34df5fab68","observation_id":"5da048d9-2d2d-46be-8479-e5e80e35bf29","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-12T06:14:42.321767Z","title":"Scikit-learn: Machine learning in Python,","venue":null,"work_id":null,"year":2011},"citing_paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-12T06:14:42.321767Z"},"links":{"citing_paper":"/paper/2607.02904"},"observation_digest":"sha256:4a9a97a2bee034d94584641b98fe213a587570246a73a1b82d04537a9add841d","observation_id":"81f31ee3-a202-4568-81d2-ee1d0e7560b2","resolution":{"observed_at":"2026-07-12T06:14:42.321767Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.02904","last_updated":"2026-07-03T03:02:48Z","latest_version":1,"primary_category":"eess.AS","snapshot_observed_at":"2026-08-08T09:39:08.965602Z","submitted_at":"2026-07-03T03:02:48Z","title":"Speaker-Aware Temporal Aggregation Strategies on Segment Representations for Depression Detection in Dyadic Interaction: A Benchmark Study"},"reference_resolution":{"displayed":29,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":29},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 29 of 29 outbound references and 0 inbound Pith citation observations for arXiv:2607.02904."}