Diverse Word Choices, Same Reference: Annotating Lexically-Rich Cross-Document Coreference. Zhukova, A., Hamborg, F., Donnay, K., Meuschke, N., & Gipp, B. February, 2026.
Diverse Word Choices, Same Reference: Annotating Lexically-Rich Cross-Document Coreference [link]Paper  doi  abstract   bibtex   1 download  
Cross-document coreference resolution (CDCR) identifies and links mentions of the same entities and events across related documents, enabling content analysis that aggregates information at the level of discourse participants. However, existing datasets primarily focus on event resolution and employ a narrow definition of coreference, which limits their effectiveness in analyzing diverse and polarized news coverage where wording varies widely. This paper proposes a revised CDCR annotation scheme of the NewsWCL50 dataset, treating coreference chains as discourse elements (DEs) and conceptual units of analysis. The approach accommodates both identity and near-identity relations, e.g., by linking "the caravan" - "asylum seekers" - "those contemplating illegal entry", allowing models to capture lexical diversity and framing variation in media discourse, while maintaining the fine-grained annotation of DEs. We reannotate the NewsWCL50 and a subset of ECB+ using a unified codebook and evaluate the new datasets through lexical diversity metrics and a same-head-lemma baseline. The results show that the reannotated datasets align closely, falling between the original ECB+ and NewsWCL50, thereby supporting balanced and discourse-aware CDCR research in the news domain.
@misc{ZhukovaHDM26,
  title = {Diverse Word Choices, Same Reference: Annotating Lexically-Rich Cross-Document Coreference},
  shorttitle = {Diverse Word Choices, Same Reference},
  author = {Zhukova, Anastasia and Hamborg, Felix and Donnay, Karsten and Meuschke, Norman and Gipp, Bela},
  year = 2026,
  month = feb,
  number = {arXiv:2602.17424},
  eprint = {2602.17424},
  primaryclass = {cs},
  publisher = {arXiv},
  doi = {10.48550/arXiv.2602.17424},
  url = {http://arxiv.org/abs/2602.17424},
  urldate = {2026-02-23},
  abstract = {Cross-document coreference resolution (CDCR) identifies and links mentions of the same entities and events across related documents, enabling content analysis that aggregates information at the level of discourse participants. However, existing datasets primarily focus on event resolution and employ a narrow definition of coreference, which limits their effectiveness in analyzing diverse and polarized news coverage where wording varies widely. This paper proposes a revised CDCR annotation scheme of the NewsWCL50 dataset, treating coreference chains as discourse elements (DEs) and conceptual units of analysis. The approach accommodates both identity and near-identity relations, e.g., by linking "the caravan" - "asylum seekers" - "those contemplating illegal entry", allowing models to capture lexical diversity and framing variation in media discourse, while maintaining the fine-grained annotation of DEs. We reannotate the NewsWCL50 and a subset of ECB+ using a unified codebook and evaluate the new datasets through lexical diversity metrics and a same-head-lemma baseline. The results show that the reannotated datasets align closely, falling between the original ECB+ and NewsWCL50, thereby supporting balanced and discourse-aware CDCR research in the news domain.},
  archiveprefix = {arXiv},
  file = {D:\Zotero\nmeuschke\Data\storage\VF58QKQK\Zhukova et al. - 2026 - Diverse Word Choices, Same Reference Annotating Lexically-Rich Cross-Document Coreference.pdf}
}

Downloads: 1