<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">lingngu</journal-id><journal-title-group><journal-title xml:lang="ru">Вестник НГУ. Серия: Лингвистика и межкультурная коммуникация</journal-title><trans-title-group xml:lang="en"><trans-title>NSU Vestnik. Series: Linguistics and Intercultural Communication</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">1818-7935</issn><publisher><publisher-name>Новосибирский государственный университет</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.25205/1818-7935-2021-19-3-57-68</article-id><article-id custom-type="elpub" pub-id-type="custom">lingngu-276</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>ТЕОРЕТИЧЕСКАЯ И ПРИКЛАДНАЯ ЛИНГВИСТИКА</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>THEORETICAL AND APPLIED LINGUISTICS</subject></subj-group></article-categories><title-group><article-title>Автоматическое обнаружение и исправление деривационных ошибок в письменной речи на русском как иностранном</article-title><trans-title-group xml:lang="en"><trans-title>A New Approach to Automatic Detection and Correction of Derivational Errors in L2 Russian</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-1707-7525</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Выренкова</surname><given-names>А. С.</given-names></name><name name-style="western" xml:lang="en"><surname>Vyrenkova</surname><given-names>A. S.</given-names></name></name-alternatives><email xlink:type="simple">anastasia.marushkina@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-8361-0282</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Смирнов</surname><given-names>И. Ю.</given-names></name><name name-style="western" xml:lang="en"><surname>Smirnov</surname><given-names>I. Yu.</given-names></name></name-alternatives><email xlink:type="simple">smirnof.van@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Национальный исследовательский университет «Высшая школа экономики»</institution><country>Россия</country></aff><aff xml:lang="en"><institution>HSE University</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2021</year></pub-date><pub-date pub-type="epub"><day>15</day><month>10</month><year>2021</year></pub-date><volume>19</volume><issue>3</issue><fpage>57</fpage><lpage>68</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Выренкова А.С., Смирнов И.Ю., 2021</copyright-statement><copyright-year>2021</copyright-year><copyright-holder xml:lang="ru">Выренкова А.С., Смирнов И.Ю.</copyright-holder><copyright-holder xml:lang="en">Vyrenkova A.S., Smirnov I.Y.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://lingngu.elpub.ru/jour/article/view/276">https://lingngu.elpub.ru/jour/article/view/276</self-uri><abstract><p>Учебные корпуса представляют собой один из наиболее ценных источников статистических данных об ошибках учащихся. Например, информация из корпусов учащихся, которые изучают язык как иностранный, используется для исследований в области усвоения второго языка [Granger, 1996]. Однако достоверность содержащихся в корпусах данных зависит от качества разметки ошибок, которая чаще всего выполняется вручную и, таким образом, представляет собой трудоемкую и кропотливую процедуру для аннотаторов. Чтобы облегчить процесс разметки, в корпусах используются дополнительные инструменты, в частности спеллчекеры. В данной статье основное внимание уделяется созданию системы автоматического поиска и исправления словообразовательных ошибок. Этот тип ошибок, почти никогда не возникающий у взрослых носителей русского языка, но появляющийся у изучающих русский язык как иностранный [Chernigovskaya, Gor, 2000], был выбран потому, что их исправление вызывает большие сложности у существующих спеллчекеров. В рамках работы на материале Русского учебного корпуса (Russian Learner Corpus, http://www.web-corpora.net/RLC/) было протестировано два подхода, помогающих в решении данной проблемы. Первый, который основывается на принципе конечных автоматов [Dickinson, Herring, 2008], имеет целью обнаружить морфологические нарушения в текстах изучающих русский как иностранный. Второй, в основе работы которого лежит модель шумного канала [Brill and Moore, 2000], обеспечивает исправление выявленных ошибок. После тестирования эффективности этих двух подходов с учетом результатов их работы была предложена собственная система автокоррекции словообразовательных ошибок. В ней используются алгоритм обнаружения морфологических ошибок из подхода Dickinson, Herring и модель Continuous Bag of Words FastText, которая основывается на теории дистрибутивной семантики [Harris, 1954]. В дополнение к ним вводятся правила исправления для распространенных случаев словотворчества, а также словарь парадигм для приведения слова к той грамматической фор-ме, в которой было употреблено исправляемое слово. Результаты работы авторской системы были апробированы на данных Русского учебного корпуса и показали свою валидность.</p></abstract><trans-abstract xml:lang="en"><p>Learner corpora serve as one of the most valuable sources of statistical data on learners' errors. For instance, data from foreign-language learners’ corpora can be used for the Second Language Acquisition research. However, corpora representativity strongly depends on the quality of its error markup, which is most frequently carried out manually and thus presents a time-consuming and painstaking routine for the annotators. To make annotation process easier, additional tools, such as spellcheckers, are usually used. This paper focuses on developing a program for automatic correction of derivational errors made by learners of Russian as a foreign language. Derivational errors, which are not common for adult Russian native speakers (L1), but occur quite often in written texts or speech of Russian as foreign language learners (L2) [Chernigovskaya, Gor, 2000], were chosen as scope of our research because correction of such mistakes presents a formidable challenge for existing spellcheckers. Using the data from the Russian Learner Corpus (http://www.web-corpora.net/RLC/), we tested two already existing approaches to solve such kind of problems. The first one is based on a finite state automaton principle developed by Dickinson and Herring 2008, and it was test-ed as algorithm for derivational errors detection. The second one which relies on the Noisy Channel model by Brill and Moore, 2000, was used for studying errors correction. After we analyzed effectiveness of these tests, we developed our own system for autocorrection of derivational errors. In our program the algorithm of Dickinson and Herring was used as word-formation error detection module. The Noisy Channel model has been rejected, and we decided to use instead the Continuous Bag of Words FastText model, based on Harris distributional semantics theory [<xref ref-type="bibr" rid="cit1954">1954</xref>]. In addition, filtering rules have been developed for correcting frequent errors that the model is unable to handle. To restore automatically the correct grammatical word form, dictionary of word paradigms is used. Model results were validated on the data of Russian Learner Corpus.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>словообразовательные ошибки</kwd><kwd>словотворчество</kwd><kwd>машинное обучение</kwd><kwd>автоматическое обнаружение ошибок</kwd><kwd>автоматическое исправление ошибок</kwd><kwd>русский как иностранный</kwd><kwd>учебный корпус</kwd><kwd>разметка</kwd></kwd-group><kwd-group xml:lang="en"><kwd>derivational errors</kwd><kwd>machine learning</kwd><kwd>automatic error detection</kwd><kwd>automatic error correction</kwd><kwd>Russian as a foreign language</kwd><kwd>learner corpus</kwd><kwd>corpus annotation</kwd></kwd-group><funding-group><funding-statement xml:lang="ru">Исследование осуществлено в рамках Программы фундаментальных исследований НИУ ВШЭ.</funding-statement><funding-statement xml:lang="en">Research has been completed as a part of the Higher School of Economics Basic Research Program.</funding-statement></funding-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Копотев М. Введение в корпусную лингвистику: электрон. учеб. пособие для студентов филологических и лингвистических специальностей университетов. Praha: Animedia, 2014.</mixed-citation><mixed-citation xml:lang="en">Копотев М. Введение в корпусную лингвистику: электрон. учеб. пособие для студентов филологических и лингвистических специальностей университетов. Praha: Animedia, 2014.</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Amaral, L., Detmar, M. Where does ICALL Fit into Foreign Language Teaching? In: Talk given at CALICO Conference. University of Hawaii, 2006.</mixed-citation><mixed-citation xml:lang="en">Amaral, L., Detmar, M. Where does ICALL Fit into Foreign Language Teaching? In: Talk given at CALICO Conference. University of Hawaii, 2006.</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Bojanowski, P., Grave, E., Joulin, A., Mikolov, T. Enriching word vectors with subword information. Transactions of the Association for Computational Linguistics, 2017, vol. 5, p. 135–146.</mixed-citation><mixed-citation xml:lang="en">Bojanowski, P., Grave, E., Joulin, A., Mikolov, T. Enriching word vectors with subword information. Transactions of the Association for Computational Linguistics, 2017, vol. 5, p. 135–146.</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Brill, E., Moore, R. An Improved Error Model for Noisy Channel Spelling Correction. Proceedings of the 38th Annual Meeting of the Association for Computational Linguistics, 2000, p. 286–293.</mixed-citation><mixed-citation xml:lang="en">Brill, E., Moore, R. An Improved Error Model for Noisy Channel Spelling Correction. Proceedings of the 38th Annual Meeting of the Association for Computational Linguistics, 2000, p. 286–293.</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Chernigovskaya T., Gor K. The Complexity of Paradigm and Input Frequencies in Native and Second Language Verbal Processing: Evidence from Russian. Language and Language Behavior (Eds. Erling Wande &amp; Tatiana Chernigovskaya), 2000, p. 20–37.</mixed-citation><mixed-citation xml:lang="en">Chernigovskaya T., Gor K. The Complexity of Paradigm and Input Frequencies in Native and Second Language Verbal Processing: Evidence from Russian. Language and Language Behavior (Eds. Erling Wande &amp; Tatiana Chernigovskaya), 2000, p. 20–37.</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Church, K., Gale, W. Probability scoring for spelling correction. Statistics and Computing, 1991, vol. 1, p. 93–103</mixed-citation><mixed-citation xml:lang="en">Church, K., Gale, W. Probability scoring for spelling correction. Statistics and Computing, 1991, vol. 1, p. 93–103</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Dickinson, M., Herring, J. Developing Online ICALL Resources for Russian. The 3rd workshop on innovative use of NLP for building educational applications, Columbus, OH, 2008, p. 1–9.</mixed-citation><mixed-citation xml:lang="en">Dickinson, M., Herring, J. Developing Online ICALL Resources for Russian. The 3rd workshop on innovative use of NLP for building educational applications, Columbus, OH, 2008, p. 1–9.</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Granger, S. From CA to CIA and back: An integrated contrastive approach to computerized bilingual and learner corpora. In: Languages in Contrast. Text-based cross-linguistic studies, Lund University Press, 1996, p. 37–51.</mixed-citation><mixed-citation xml:lang="en">Granger, S. From CA to CIA and back: An integrated contrastive approach to computerized bilin-gual and learner corpora. In: Languages in Contrast. Text-based cross-linguistic studies, Lund University Press, 1996, p. 37–51.</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Granger, S. Learner Corpora in Foreign Language Education. In: Language, Education and Technology, 2017, p. 1–14. DOI 10.1007/978-3-319-02328-1_33-1.</mixed-citation><mixed-citation xml:lang="en">Granger, S. Learner Corpora in Foreign Language Education. In: Language, Education and Technology, 2017, p. 1–14. DOI 10.1007/978-3-319-02328-1_33-1.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Harris, Z. Distributional Structure. WORD, 1954, vol. 10, iss. 2–3, p. 146–162. DOI 10.1080/00437956.1954.11659520</mixed-citation><mixed-citation xml:lang="en">Harris, Z. Distributional Structure. WORD, 1954, vol. 10, iss. 2–3, p. 146–162. DOI 10.1080/00437956.1954.11659520</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Heift, T., Devlan, N. Web delivery of adaptive and interactive language tutoring. International Journal of Artificial Intelligence in Education, 2001, vol. 12 (4), p. 310–325. Kernighan, M., Church, K., Gale, W. A Spelling Correction Program Based on a Noisy Channel Model. COLING-90, 1990, p. 205–210. DOI 10.3115/997939.997975.</mixed-citation><mixed-citation xml:lang="en">Heift, T., Devlan, N. Web delivery of adaptive and interactive language tutoring. International Journal of Artificial Intelligence in Education, 2001, vol. 12 (4), p. 310–325.</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Kernighan, M., Church, K., Gale, W. A Spelling Correction Program Based on a Noisy Channel Model. COLING-90, 1990, p. 205–210. DOI 10.3115/997939.997975.</mixed-citation><mixed-citation xml:lang="en">Kernighan, M., Church, K., Gale, W. A Spelling Correction Program Based on a Noisy Channel Model. COLING-90, 1990, p. 205–210. DOI 10.3115/997939.997975.</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Kopotev, M. Introduction to Corpus linguistics: Course-book for students of arts subjects with emphasis on the Russian language. Praha, Animedia, 2014. (in Russ.)</mixed-citation><mixed-citation xml:lang="en">Kopotev, M. Introduction to Corpus linguistics: Course-book for students of arts subjects with emphasis on the Russian language. Praha, Animedia, 2014. (in Russ.)</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">Kutuzov, A., Kuzmenko, E. WebVectors: A Toolkit for Building Web Interfaces for Vector Semantic Models. Ignatov D. et al. (eds) Analysis of Images, Social Networks and Texts. AIST 2016. Communications in Computer and Information Science, 2017, vol. 661. Springer, Cham.</mixed-citation><mixed-citation xml:lang="en">Kutuzov, A., Kuzmenko, E. WebVectors: A Toolkit for Building Web Interfaces for Vector Semantic Models. Ignatov D. et al. (eds) Analysis of Images, Social Networks and Texts. AIST 2016. Communications in Computer and Information Science, 2017, vol. 661. Springer, Cham.</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Leacock, C., Chodorow, M., Gamon, M., Tetreau, J. Automated Grammatical Error Detection for Language Learners, 2nd ed. Synthesis Lectures on Human Language Technologies. 2014, vol. 7, p. 1–185. DOI 10.2200/S00562ED1V01Y201401HLT025</mixed-citation><mixed-citation xml:lang="en">Leacock, C., Chodorow, M., Gamon, M., Tetreau, J. Automated Grammatical Error Detection for Language Learners, 2nd ed. Synthesis Lectures on Human Language Technologies. 2014, vol. 7, p. 1–185. DOI 10.2200/S00562ED1V01Y201401HLT025</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Nagata, N. An Effective Application of Natural Language. Processing in Second Language Instruction. CALICO Journal, 1995.</mixed-citation><mixed-citation xml:lang="en">Nagata, N. An Effective Application of Natural Language. Processing in Second Language Instruction. CALICO Journal, 1995.</mixed-citation></citation-alternatives></ref><ref id="cit17"><label>17</label><citation-alternatives><mixed-citation xml:lang="ru">Paquot, M., Jarvis, S. Learner corpora and native language identification, 2015. DOI 10.1017/CBO9781139649414.027.</mixed-citation><mixed-citation xml:lang="en">Paquot, M., Jarvis, S. Learner corpora and native language identification, 2015. DOI 10.1017/CBO9781139649414.027.</mixed-citation></citation-alternatives></ref><ref id="cit18"><label>18</label><citation-alternatives><mixed-citation xml:lang="ru">Rudzewitz, B., Ziai, R., De Kuthy, K., Möller, V., Nuxoll, F., Detmar, M. Generating Feedback for English Foreign Language Exercises. Proceedings of the Thirteenth Workshop on Innova-tive Use of NLP for Building Educational Applications (BEA), 2018, p. 127–136.</mixed-citation><mixed-citation xml:lang="en">Rudzewitz, B., Ziai, R., De Kuthy, K., Möller, V., Nuxoll, F., Detmar, M. Generating Feedback for English Foreign Language Exercises. Proceedings of the Thirteenth Workshop on Innovative Use of NLP for Building Educational Applications (BEA), 2018, p. 127–136.</mixed-citation></citation-alternatives></ref><ref id="cit19"><label>19</label><citation-alternatives><mixed-citation xml:lang="ru">Shannon, C. A Mathematical Theory of Communication. Bell System Technical Journal, 1948, vol. 27, p. 379–423.</mixed-citation><mixed-citation xml:lang="en">Shannon, C. A Mathematical Theory of Communication. Bell System Technical Journal, 1948, vol. 27, p. 379–423.</mixed-citation></citation-alternatives></ref><ref id="cit20"><label>20</label><citation-alternatives><mixed-citation xml:lang="ru">Shavrina, T., Shapovalova, O. To the methodology of corpus construction for machine learning: «Taiga» syntax tree corpus and parser. Proceedings of international conference CORPO-RA2017, 2017, p. 78–84.</mixed-citation><mixed-citation xml:lang="en">Shavrina, T., Shapovalova, O. To the methodology of corpus construction for machine learning: «Taiga» syntax tree corpus and parser. Proceedings of international conference CORPO-RA2017, 2017, p. 78–84.</mixed-citation></citation-alternatives></ref><ref id="cit21"><label>21</label><citation-alternatives><mixed-citation xml:lang="ru">Sorokin, A., Baytin, A., Galinskaya, I., Rykunova, E., Shavrina, T. SpellRuEval: the First Competition on Automatic Spelling Correction for Russian. Computational Linguistics and Intellectual Technologies Proceedings of the Annual International Conference “Dialogue”, 2016, p. 660–673.</mixed-citation><mixed-citation xml:lang="en">Sorokin, A., Baytin, A., Galinskaya, I., Rykunova, E., Shavrina, T. SpellRuEval: the First Com-petition on Automatic Spelling Correction for Russian. Computational Linguistics and Intellec-tual Technologies Proceedings of the Annual International Conference “Dialogue”, 2016, p. 660–673.</mixed-citation></citation-alternatives></ref><ref id="cit22"><label>22</label><citation-alternatives><mixed-citation xml:lang="ru">Valdes, G. The teaching of heritage languages: an introduction for Slavicteaching professionals. Slavica, Bloomington, 2000, p. 375–403.</mixed-citation><mixed-citation xml:lang="en">Valdes, G. The teaching of heritage languages: an introduction for Slavic-teaching professionals. Slavica, Bloomington, 2000, p. 375–403.</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
