%Aigaion2 BibTeX export from Idiap Publications %Thursday 21 November 2024 12:33:24 PM @TECHREPORT{grangier:2005:idiap-05-21, author = {Grangier, David and Bengio, Samy}, projects = {Idiap}, title = {Inferring Document Similarity from Hyper-links}, type = {Idiap-RR}, number = {Idiap-RR-21-2005}, year = {2005}, institution = {IDIAP}, abstract = {Assessing semantic similarity between text documents is a crucial aspect in Information Retrieval systems. In this paper, we propose a technique to derive a similarity measure from hyper-link information. As linked documents are generally semantically closer than unlinked documents, we use a training corpus with hyper-links to infer a function $a,b \to sim(a,b)$ that assigns a higher value to linked documents than to unlinked ones. Two sets of experiments on different corpora show that this function compares favorably with {\em OKAPI} matching on document retrieval tasks.}, pdf = {https://publications.idiap.ch/attachments/reports/2005/grangier-rr-05-21.pdf}, postscript = {ftp://ftp.idiap.ch/pub/reports/2005/grangier-rr-05-21.ps.gz}, ipdmembership={speech}, }