<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">klj</journal-id><journal-title-group><journal-title xml:lang="ru">Казанский лингвистический журнал</journal-title><trans-title-group xml:lang="en"><trans-title>Kazan linguistic journal</trans-title></trans-title-group></journal-title-group><issn pub-type="epub">3033-8751</issn><publisher><publisher-name>Казанский (Приволжский) федеральный университет</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.26907/2658-3321.2025.8.2.204-217</article-id><article-id custom-type="elpub" pub-id-type="custom">klj-46</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>ФИЛОЛОГИЯ. ТЕОРЕТИЧЕСКАЯ, ПРИКЛАДНАЯ И СРАВНИТЕЛЬНО-СОПОСТАВИТЕЛЬНАЯ ЛИНГВИСТИКА</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>PHILOLOGICAL STUDIES. THEORETICAL, APPLIED AND COMPARATIVE LINGUISTICS</subject></subj-group></article-categories><title-group><article-title>Автоматическое распознавание лексических заимствований в корпусе текстов</article-title><trans-title-group xml:lang="en"><trans-title>Automatic Detection of Lexical Loanwords in a Text Corpus</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-3632-793X</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Дмитриев</surname><given-names>А. В.</given-names></name><name name-style="western" xml:lang="en"><surname>Dmitrijev</surname><given-names>A. V.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Дмитриев Александр Владиславович – Доцент</p><p>Санкт-Петербург</p></bio><bio xml:lang="en"><p>Dmitrijev Alexander Vladislavovich – Associate Professor</p><p>Saint-Petersburg</p></bio><email xlink:type="simple">avd84@list.ru</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0007-3127-2737</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Крупнова</surname><given-names>Е. С.</given-names></name><name name-style="western" xml:lang="en"><surname>Krupnova</surname><given-names>E. S.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Крупнова Елена Сергеевна – Специалист по учебно-методической работе 1 категории</p><p>Санкт-Петербург</p></bio><bio xml:lang="en"><p>Krupnova Elena Sergeevna – Specialist in educational and methodological work of first category</p><p>Saint-Petersburg</p></bio><email xlink:type="simple">krupnalena@mail.ru</email><xref ref-type="aff" rid="aff-1"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Санкт-Петербургский политехнический университет Петра Великого</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Peter the Great Saint-Petersburg polytechnic university</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2025</year></pub-date><pub-date pub-type="epub"><day>09</day><month>12</month><year>2025</year></pub-date><volume>8</volume><issue>2</issue><fpage>204</fpage><lpage>217</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Дмитриев А.В., Крупнова Е.С., 2025</copyright-statement><copyright-year>2025</copyright-year><copyright-holder xml:lang="ru">Дмитриев А.В., Крупнова Е.С.</copyright-holder><copyright-holder xml:lang="en">Dmitrijev A.V., Krupnova E.S.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://www.kljournal.ru/jour/article/view/46">https://www.kljournal.ru/jour/article/view/46</self-uri><abstract><p>В эпоху глобализации и активного взаимодействия носителей языка с представителями разных культур пополнение словарного состава иноязычными словами становится ключевым фактором развития и обогащения языковой системы. Однако в связи с увеличением объёма текстовых данных ручной анализ и поиск лексических единиц становятся менее эффективным и времязатратным. Это делает актуальным применение методов автоматической обработки естественного языка (NLP) для извлечения заимствований. Цель – рассмотреть несколько подходов автоматического извлечения лексических единиц с 1986 по настоящее время, а также разработать алгоритм для решения данной задачи. Материал исследования представлен собранным корпусом из 22348 английских предложений, полученных с сайтов 11 ведущих университетов Австрии, Германии и России. Для проверки результатов использовались 47 новых предложений. Дополнительно были сгенерированы 1318 новых предложений, содержащих немецкие заимствования с помощью чат-ботов. Методы – в исследовании использовалась мультиязыковая модель “bert-base-multilingual-cased”. Производилась разметка корпуса с использованием двух тегов, обозначающих наличие/отсутствие немецкого заимствования в предложении. Затем осуществлялось дообучение модели на размеченном корпусе и на дополнительно сгенерированных предложениях. Результаты исследования показывают, что современные подходы позволяют достичь высокой точности, однако остаются трудности, связанные с работой моделей с различными языковыми парами и улучшением их производительности. Кроме того, описан алгоритм автоматического извлечения немецких заимствований из английских предложений с помощью дообученной на 900 текстах модели BERT. Модель показала высокие результаты и смогла успешно распознать 30 из 43 слов немецкого происхождения.</p></abstract><trans-abstract xml:lang="en"><p>In the context of globalization and the dynamic interaction between language speakers and representatives of diverse cultures, the incorporation of foreign lexical items into vocabulary systems has become a fundamental driver of linguistic development and enrichment. However, the exponential growth in textual data has rendered manual analysis and lexical unit identification increasingly inefficient and time-consuming. This necessitates the implementation of automated natural language processing (NLP) methods for loanword extraction. This paper aims to examine various approaches to automatic lexical item extraction from 1986 to the present, while also developing an algorithm to address this challenge. The material includes a collected corpus of 22348 English sentences parsed from the websites of 11 leading universities in Austria, Germany and Russia. To verify the results, 47 new sentences were used. Additionally, 1318 new sentences including German loanwords were generated using chatbots. As for the methods, the “bert-base-multilingual-cased model” was used in the study. Corpus was annotated with two tags indicating the presence/absence of a German loanword in a sentence. The model was then retrained on the corpus and on additionally generated sentences. The findings demonstrate that while contemporary methods achieve high accuracy rates, significant challenges persist in model performance across different language pairs and in overall efficiency enhancement. Furthermore, the study describes an algorithm for automatic extraction of German loanwords from English sentences utilizing the BERT large language model trained on a corpus of 900 texts. The model demonstrated robust performance, successfully identifying 30 out of 43 words of German origin.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>заимствование</kwd><kwd>немецкие заимствования</kwd><kwd>автоматическая обработка текстов</kwd><kwd>NLP</kwd><kwd>мультиязыковая модель BERT</kwd></kwd-group><kwd-group xml:lang="en"><kwd>borrowing</kwd><kwd>Germanisms</kwd><kwd>natural language processing</kwd><kwd>NLP</kwd><kwd>multilingual BERT model</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Köllner M. Automatic loanword identification using tree reconciliation. Dissertation zur Erlangung des akademischen Grades Doktor der Philosophie in der Philosophischen Fakultat der Eberhard Karls. Universitat Tubingen; 2021. 216 p.</mixed-citation><mixed-citation xml:lang="en">Köllner M. Automatic loanword identification using tree reconciliation. Dissertation zur Erlangung des akademischen Grades Doktor der Philosophie in der Philosophischen Fakultat der Eberhard Karls. Universitat Tubingen; 2021. 216 p.</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Mennecier P., Nerbonne J., Heyer E., Manni F. A Central Asian Language Survey: Collecting Data, Measuring Relatedness and Detecting Loans. Language Dynamics and Change. 2016; 6: 57–98.</mixed-citation><mixed-citation xml:lang="en">Mennecier P., Nerbonne J., Heyer E., Manni F. A Central Asian Language Survey: Collecting Data, Measuring Relatedness and Detecting Loans. Language Dynamics and Change. 2016;6:57–98.</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Beatrice A. Comparing Corpus-based to Web-based Lookup Techniques for Automatic English Inclusion Detection. URL: http://www.lrec-conf.org/proceedings/lrec2008/pdf/674_paper.pdf [дата обращения: 20.01.2025].</mixed-citation><mixed-citation xml:lang="en">Beatrice A. Comparing Corpus-based to Web-based Lookup Techniques for Automatic English Inclusion Detection. Available from: http://www.lrec-conf.org/proceedings/lrec2008/pdf/674_paper.pdf  [accessed: 20.01.2025].</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Álvarez-Mellado E. An Annotated Corpus of Emerging Anglicisms in Spanish Newspaper Headlines; 2020. URL: https://arxiv.org/pdf/2004.02929.pdf [дата обращения: 25.01.2025].</mixed-citation><mixed-citation xml:lang="en">Álvarez-Mellado E. An Annotated Corpus of Emerging Anglicisms in Spanish Newspaper Headlines; 2020. Available from: https://arxiv.org/pdf/2004.02929.pdf  [accessed: 25.01.2025].</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Shengyi J., Tong C., Yingwen F., Nankai L. and Jieyi X. BERT4EVER at ADoBo 2021: Detection of Borrowings in the Spanish Language Using Pseudolabel Technology; 2021. URL: https://ceur-ws.org/Vol-2943/adobo_paper1.pdf (дата обращения: 23.01.2025)</mixed-citation><mixed-citation xml:lang="en">Shengyi J., Tong C., Yingwen F., Nankai L. and Jieyi X. BERT4EVER at ADoBo 2021: Detection of Borrowings in the Spanish Language Using Pseudolabel Technology; 2021. Available from: https://ceur-ws.org/Vol-2943/adobo_paper1.pdf (accessed: 23.01.2025)</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Nath A., Saravani S.M., Khebour I., Mannan S., Liand Z., et al. A Generalized Method for Automated Multilingual Loanword Detection. Proceedings of the 29th International Conference on Computational Linguistics; 2022. Pp. 4996–5013.</mixed-citation><mixed-citation xml:lang="en">Nath A., Saravani S.M., Khebour I., Mannan S., Liand Z., et al. A Generalized Method for Automated Multilingual Loanword Detection. Proceedings of the 29th International Conference on Computational Linguistics; 2022. Pp. 4996–5013.</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Miller J.E., Tresoldi T., Zariquiey R., Beltrán Castañon C.A., et al. Using lexical language models to detect borrowings in monolingual wordlists. 2020; 15(12):1–23.</mixed-citation><mixed-citation xml:lang="en">Miller J.E., Tresoldi T., Zariquiey R., Beltrán Castañon C.A., et al. Using lexical language models to detect borrowings in monolingual wordlists. 2020; 15(12):1–23.</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Кортегосо В.Н., Захаров В.П. Два метода выявления русских заимствований в якутских текстах. International Journal of Open Information Technologies. 2022;10 (11):26–34.</mixed-citation><mixed-citation xml:lang="en">Vissio N. Cortegoso, Zakharov V.P. Two methods for identifying Russian words in Yakut texts. International Journal of Open Information Technologies. 2022;10(11):26–34. (In Russ.)</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Падерина Т. С. Методы извлечения терминов в научных текстах (на материале статей по направлению науки о земле). Казанский лингвистический журнал. 2023; 6(3): 388–396.</mixed-citation><mixed-citation xml:lang="en">Paderina T.S. Methods for Terminology Extraction in Scientific Texts (Based on Articles of Earth Sciences). Kazan Linguistic Journal. 2023;6(3):388–396. (In Russ.)</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Devlin J., Chang M.W., Kenton L., Toutanova K. Pre-training of Deep Bidirectional Transformers for Language. URL: http://arxiv.org/abs/1810.04805 (accessed: 27.01.2025)</mixed-citation><mixed-citation xml:lang="en">Devlin J., Chang M.W., Kenton L., Toutanova K. Pre-training of Deep Bidirectional Transformers for Language. URL: http://arxiv.org/abs/1810.04805 (accessed: 27.01.2025)</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
