@inproceedings{Tanguy-Fabre-Hathout-Ho-Dac:CORIA-TALN:2025,
    author = "Tanguy, Ludovic and Fabre, C\'ecile and Hathout, Nabil and Ho-Dac, Lydia-Mai",
    title = "Embeddings, topic models, LLM : un air de famille",
    booktitle = "Actes de CORIA-TALN-RJCRI-RECITAL 2025. Actes des 32\`eme Conf\'erence sur le Traitement Automatique des Langues Naturelles (TALN),  volume 1 : articles scientifiques originaux",
    month = "6",
    year = "2025",
    address = "Marseille, France",
    publisher = "Association pour le Traitement Automatique des Langues",
    pages = "295-312",
    note = "Cet article pr\'esente une \'etude portant sur les termes exprimant les relations familiales (fr\`ere, tante, etc",
    abstract = "Word embeddings, topic models, LLMs: a family affair This article presents a study on terms denoting family relationships (brother, aunt, etc.) in French using three approaches: word embeddings, topic modeling, and pre-trained language models. The first two types of representations are built from the French version of Wikipedia, while the third is derived through direct interaction with ChatGPT. The aim is to compare how these three methods represent such terms, in two main ways: by evaluating them against a structural definition of family relations (in terms of features such as gender, lineage, etc.), and by comparing the topics associated with each term. These methods reveal different modes of structuring family-related vocabulary, while also underscoring the continued necessity of corpus-based and controlled analyses to obtain reliable results.",
    keywords = "Word embeddings, topic modeling, LLMs, family lexicon.",
    url = "https://talnarchives.atala.org/TALN/TALN-2025/108.pdf"
}
