<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE root>
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.2" xml:lang="en"><front><journal-meta><journal-id journal-id-type="publisher-id">Computational nanotechnology</journal-id><journal-title-group><journal-title xml:lang="en">Computational nanotechnology</journal-title><trans-title-group xml:lang="kk"><trans-title>Computational nanotechnology</trans-title></trans-title-group><trans-title-group xml:lang="pt"><trans-title>Computational nanotechnology</trans-title></trans-title-group><trans-title-group xml:lang="ru"><trans-title>Computational nanotechnology</trans-title></trans-title-group><trans-title-group xml:lang="zh"><trans-title>Computational nanotechnology</trans-title></trans-title-group></journal-title-group><issn publication-format="print">2313-223X</issn><issn publication-format="electronic">2587-9693</issn><publisher><publisher-name xml:lang="en">YUR-VAK</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">688951</article-id><article-id pub-id-type="doi">10.33693/2313-223X-2025-12-2-19-27</article-id><article-id pub-id-type="edn">QPYWFS</article-id><article-categories><subj-group subj-group-type="toc-heading" xml:lang="en"><subject>ARTIFICIAL INTELLIGENCE AND MACHINE LEARNING</subject></subj-group><subj-group subj-group-type="toc-heading" xml:lang="ru"><subject>ИСКУССТВЕННЫЙ ИНТЕЛЛЕКТ И МАШИННОЕ ОБУЧЕНИЕ</subject></subj-group><subj-group subj-group-type="article-type"><subject>Research Article</subject></subj-group></article-categories><title-group><article-title xml:lang="en">Modification of the method for modeling the thematic environment of terms using the LDA approach</article-title><trans-title-group xml:lang="ru"><trans-title>Модификация метода моделирования тематического окружения терминов на основе подхода LDA</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-6917-9668</contrib-id><contrib-id contrib-id-type="scopus">57203129675</contrib-id><contrib-id contrib-id-type="researcherid">AAR-4461-2021</contrib-id><contrib-id contrib-id-type="spin">5231-7243</contrib-id><name-alternatives><name xml:lang="en"><surname>Zolotarev</surname><given-names>Oleg V.</given-names></name><name xml:lang="ru"><surname>Золотарев</surname><given-names>Олег Васильевич</given-names></name></name-alternatives><address><country country="RU">Russian Federation</country></address><bio xml:lang="en"><p>Cand. Sci. (Eng.), Associate Professor; Head, Department of Information Systems in Economics and Management</p></bio><bio xml:lang="ru"><p>кандидат технических наук, доцент; заведующий, кафедра информационных систем в экономике и управлении</p></bio><email>ol-zolot@yandex.ru</email><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-1362-802X</contrib-id><contrib-id contrib-id-type="researcherid">GZG-2909-2022</contrib-id><name-alternatives><name xml:lang="en"><surname>Yurchak</surname><given-names>Vladimir A.</given-names></name><name xml:lang="ru"><surname>Юрчак</surname><given-names>Владимир Александрович</given-names></name></name-alternatives><address><country country="RU">Russian Federation</country></address><bio xml:lang="en"><p>postgraduate student, lecturer, Department of Information Systems in Economics and Management</p></bio><bio xml:lang="ru"><p>аспирант, преподаватель, кафедра информационных систем в экономике и управлении</p></bio><email>yurchak.vladimir.1998@mail.ru</email><xref ref-type="aff" rid="aff1"/></contrib></contrib-group><aff-alternatives id="aff1"><aff><institution xml:lang="en">Russian New University</institution></aff><aff><institution xml:lang="ru">Российский новый университет</institution></aff></aff-alternatives><pub-date date-type="pub" iso-8601-date="2025-08-19" publication-format="electronic"><day>19</day><month>08</month><year>2025</year></pub-date><volume>12</volume><issue>2</issue><issue-title xml:lang="en"/><issue-title xml:lang="ru"/><fpage>19</fpage><lpage>27</lpage><history><date date-type="received" iso-8601-date="2025-08-11"><day>11</day><month>08</month><year>2025</year></date><date date-type="accepted" iso-8601-date="2025-08-11"><day>11</day><month>08</month><year>2025</year></date></history><permissions><copyright-statement xml:lang="en">Copyright ©; 2025, Yur-VAK</copyright-statement><copyright-statement xml:lang="ru">Copyright ©; 2025, Юр-ВАК</copyright-statement><copyright-year>2025</copyright-year><copyright-holder xml:lang="en">Yur-VAK</copyright-holder><copyright-holder xml:lang="ru">Юр-ВАК</copyright-holder><ali:free_to_read xmlns:ali="http://www.niso.org/schemas/ali/1.0/" start_date="2026-08-19"/><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/">https://journals.eco-vector.com/2313-223X/about/editorialPolicies</ali:license_ref></license></permissions><self-uri xlink:href="https://journals.eco-vector.com/2313-223X/article/view/688951">https://journals.eco-vector.com/2313-223X/article/view/688951</self-uri><abstract xml:lang="en"><p>Thematic modeling is an essential tool for analyzing large volumes of textual data, enabling the identification of latent semantic patterns. However, conventional approaches such as Latent Dirichlet Allocation (LDA) encounter difficulties when dealing with multi-valued and unigram tokens, resulting in reduced accuracy and clarity in the outcomes. This study aims to develop a technique for constructing a thematic structure based on refined LDA, which incorporates contextual features, vector representations of words, and external vocabularies. The objective is to address terminological ambiguity and enhance the clarity of thematic groups. The paper employs a mathematical model that integrates probabilistic thematic modeling with vector representations, facilitating the differentiation of word meanings and the establishment of precise connections between them. Using the corpus of Dimensions AI and PubMed publications, the study demonstrates an improved distribution of terms within thematic clusters. This involves frequency analysis and vector similarity, which are essential components of the study. The results emphasize the effectiveness of an integrated approach to dealing with complex linguistic structures in automated text analysis.</p></abstract><trans-abstract xml:lang="ru"><p>Тематическое моделирование является ключевым инструментом для анализа больших текстовых данных, позволяя выявлять скрытые смысловые структуры. Однако традиционные методы, такие как LDA, сталкиваются с проблемами при работе с многозначными и монолексемными токенами, что снижает точность и интерпретируемость результатов. Целью исследования является разработка метода моделирования тематического окружения терминов на основе модифицированного подхода LDA (Latent Dirichlet Allocation), интегрирующего контекстные признаки, векторные представления слов и внешние тезаурусы. Основные задачи включали: учет многозначности терминов, а также повышение интерпретируемости тематических кластеров. В работе используется математическая модель, объединяющая вероятностное тематическое моделирование с векторным представлением, что позволяет различать значения терминов и устанавливать точные связи между ними. Результаты, полученные на корпусах публикаций Dimensions AI и PubMed, демонстрируют улучшенное распределение терминов в тематических кластерах, включая анализ частоты встречаемости и векторное сходство. Исследование подтверждает эффективность комбинированного подхода для обработки сложных лингвистических конструкций в автоматизированном анализе текстов.</p></trans-abstract><kwd-group xml:lang="en"><kwd>LDA method</kwd><kwd>thesauri</kwd><kwd>multivalued tokens</kwd><kwd>monolex tokens</kwd><kwd>Dimensions AI</kwd><kwd>PubMed</kwd></kwd-group><kwd-group xml:lang="ru"><kwd>метод LDA</kwd><kwd>тезаурусы</kwd><kwd>многозначные токены</kwd><kwd>монолексемные токены</kwd><kwd>Dimensions AI</kwd><kwd>PubMed</kwd></kwd-group><funding-group/></article-meta></front><body></body><back><ref-list><ref id="B1"><label>1.</label><mixed-citation>Angelov D. Top2Vec: Distributed representations of topics. arXiv:2008.09470. 2020. URL: https://arxiv.org/abs/2008.09470 (дата обращения: 12.05.2025).</mixed-citation></ref><ref id="B2"><label>2.</label><mixed-citation>Grootendorst M. BERTopic: Neural topic modeling with a class-based TF-IDF procedure. arXiv:2203.05794. 2022. URL: https://arxiv.org/abs/2203.05794 (дата обращения: 12.05.2025).</mixed-citation></ref><ref id="B3"><label>3.</label><mixed-citation>Dieng A.B., Ruiz F.J.R., Blei D.M. Topic modeling in embedding spaces. Transactions of the Association for Computational Linguistics. 2020. Vol. 8. Pp. 439–453.</mixed-citation></ref><ref id="B4"><label>4.</label><mixed-citation>Bianchi F., Terragni S., Hovy D. Pre-training is a hot topic: Contextualized document embeddings improve topic coherence. Findings of EMNLP. 2024. Pp. 2346–2359.</mixed-citation></ref><ref id="B5"><label>5.</label><mixed-citation>Biggio M., Crippa F., Fumagalli A. et al. Joint document-token embeddings for hierarchical topic modeling. In: Contextualized-Top2Vec. Proceedings of the 2024 Conference on Neural Information Processing Systems. 2024. Pp. 10234–10246.</mixed-citation></ref><ref id="B6"><label>6.</label><mixed-citation>Bianchi F., Terragni S., Hovy D. Combined Topic Model (CTM): Integrating contextualized embeddings into LDA. In: Findings of ACL. 2021. Pp. 1175–1188.</mixed-citation></ref><ref id="B7"><label>7.</label><mixed-citation>Maheshwari K., Roberts M.E., Stewart B.M. Evaluating contextualized topic coherence for neural topic models. Journal of Machine Learning Research. 2022. Vol. 23. Pp. 1–20.</mixed-citation></ref><ref id="B8"><label>8.</label><mixed-citation>Angelov D., Inkpen D. Hierarchical topic modeling with contextual token representations. In: Contextualized-Top2Vec. Proceedings of the 2024 Conference on Neural Information Processing Systems. 2024. URL: https://github.com/ddangelov/Top2Vec (дата обращения: 12.05.2025).</mixed-citation></ref><ref id="B9"><label>9.</label><mixed-citation>Lee J., Yoon W., Kim S. et al. BioBERT: A pre-trained biomedical language representation model for biomedical text mining. Bioinformatics. 2020. Vol. 36. No. 4. Pp. 1234–1240.</mixed-citation></ref><ref id="B10"><label>10.</label><mixed-citation>Zaheer M., Guruganesh G., Dubey K.A. et al. Big Bird: Transformers for longer sequences. In: NeurIPS. 2020. Pp. 17283–17297.</mixed-citation></ref><ref id="B11"><label>11.</label><mixed-citation>Reimers N., Gurevych I. Sentence-BERT: Sentence embeddings using siamese BERT-Networks. In: Proceedings of EMNLP. 2019. Pp. 3980–3990.</mixed-citation></ref><ref id="B12"><label>12.</label><mixed-citation>Pethe M., Joshi S., Kulkarni P. et al. SciBERT: A pretrained language model for scientific text. In: Proceedings of EMNLP. 2019. Pp. 3615–3620.</mixed-citation></ref><ref id="B13"><label>13.</label><mixed-citation>McInnes L., Healy J., Melville J. UMAP: Uniform manifold approximation and projection for dimension reduction. arXiv:1802.03426. 2018. URL: https://arxiv.org/abs/1802.03426 (дата обращения: 12.05.2025).</mixed-citation></ref><ref id="B14"><label>14.</label><mixed-citation>Fraley C., Raftery A.E. Model-based clustering, discriminant analysis, and density estimation. Journal of the American Statistical Association. 2002. Vol. 97. No. 458. Pp. 611–631.</mixed-citation></ref><ref id="B15"><label>15.</label><mixed-citation>Bodenreider O. The Unified Medical Language System (UMLS): Integrating biomedical terminology. Nucleic Acids Research. 2004. Vol. 32. Suppl. 1. Pp. D267–D270.</mixed-citation></ref><ref id="B16"><label>16.</label><mixed-citation>Lowe H.J., Barnett G.O. Understanding and using the Medical Subject Headings (MeSH) vocabulary to perform literature searches. JAMA. 1994. Vol. 271. No. 14. Pp. 1103–1108.</mixed-citation></ref><ref id="B17"><label>17.</label><mixed-citation>Zolotarev O.V., Hakimova A.Kh., Agraval S. et al. Removing terms from biomedical publications-an approach based on n-grams. In: Civilization of Knowledge: Russian realities. 2023. Pp. 136–160. EDN: IRLOBR.</mixed-citation></ref></ref-list></back></article>
