<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">lingngu</journal-id><journal-title-group><journal-title xml:lang="ru">Вестник НГУ. Серия: Лингвистика и межкультурная коммуникация</journal-title><trans-title-group xml:lang="en"><trans-title>NSU Vestnik. Series: Linguistics and Intercultural Communication</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">1818-7935</issn><publisher><publisher-name>Новосибирский государственный университет</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.25205/1818-7935-2026-24-1-102-112</article-id><article-id custom-type="elpub" pub-id-type="custom">lingngu-1200</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>КОМПЬЮТЕРНАЯ И ПРИКЛАДНАЯ ЛИНГВИСТИКА</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>COMPUTER AND APPLIED LINGUISTICS</subject></subj-group></article-categories><title-group><article-title>Авторские инварианты в английских художественных текстах</article-title><trans-title-group xml:lang="en"><trans-title>Author’s invariants in English literary texts</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-5808-3134</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ковалевский</surname><given-names>А. П.</given-names></name><name name-style="western" xml:lang="en"><surname>Kovalevskii</surname><given-names>A. P.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Ковалевский Артем Павлович, доктор физико-математических наук, доцент, ведущий научный сотрудник </p></bio><bio xml:lang="en"><p>Artyom P. Kovalevskii, DSc., Associate Professor, Leading Researcher</p></bio><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0002-1781-2630</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Павлова</surname><given-names>Ю. В.</given-names></name><name name-style="western" xml:lang="en"><surname>Pavlova</surname><given-names>Yu. V.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Павлова Юлия Викторовна, студентка </p></bio><bio xml:lang="en"><p>Yulia V. Pavlova, Student</p></bio><xref ref-type="aff" rid="aff-2"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Институт математики СО РАН</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Sobolev Institute of Mathematics</institution><country>Russian Federation</country></aff></aff-alternatives><aff-alternatives id="aff-2"><aff xml:lang="ru"><institution>Новосибирский государственный университет</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Novosibirsk State University</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>27</day><month>05</month><year>2026</year></pub-date><volume>24</volume><issue>1</issue><fpage>102</fpage><lpage>112</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Ковалевский А.П., Павлова Ю.В., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Ковалевский А.П., Павлова Ю.В.</copyright-holder><copyright-holder xml:lang="en">Kovalevskii A.P., Pavlova Y.V.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://lingngu.elpub.ru/jour/article/view/1200">https://lingngu.elpub.ru/jour/article/view/1200</self-uri><abstract><p>В работе предложена и протестирована методика обнаружения стилистической разладки в художественных текстах на английском языке, основанная на статистическом анализе бинарной последовательности, формируемой по признаку принадлежности слов к авторскому инварианту. В качестве авторского инварианта рассматривалось множество служебных слов, текст отображался в виде последовательности нулей и единиц, служившей для исследования стилистических особенностей текста. На основе этой последовательности был построен эмпирический мост, максимальное отклонение которого использовалось в качестве статистики нормы эмпирического моста для выявления возможной разладки. Метод был апробирован на корпусе из 100 художественных текстов британских и американских авторов XIX–XXI веков, а также на 9 000 попарных комбинаций склеенных текстов. Были выделены пороговые значения статистики, позволяющие отличать однородные тексты от текстов с возможной стилистической границей. Для оценки статистической значимости использовалась аппроксимация распределения Колмогорова, на основе которой рассчитывались р-значения. Результаты экспериментов показали: устойчивую способность метода выявлять стилистическую границу между произведениями разных авторов; эффективность алгоритма при различении одного текста одного автора и комбинации двух текстов одного автора; эффективность алгоритма при различении одного текста одного автора от пары текстов разных авторов; независимость метода от длины текста: высокие и низкие значения статистики и р-значения наблюдались при разных объемах. Анализ эмпирических мостов и поведения отдельных служебных слов показал, что стилистические особенности текста могут проявляться как в масштабах всего произведения, так и внутри отдельных его частей. Метод эффективно фиксирует подобные изменения. Разработанный подход может быть использован в рамках корпусной лингвистики, прикладной стилистики, обработки естественного языка, а также при создании интеллектуальных систем анализа текста.</p></abstract><trans-abstract xml:lang="en"><p>The paper proposes and tests a method for detecting stylistic change points in English-language ﬁction texts based on statistical analysis of a binary sequence taking into account the words belonging to the author’s invariant. A set of function words was considered as the author’s invariant; texts were displayed as sequences of zeros and ones used to study the stylistic features of texts. An empirical bridge was constructed based on this sequence. The maximum deviation of the empirical bridge served as statistical norm to detect possible change points. The method was tested on a corpus of 100 novels by British and American authors of the 19th–21st centuries, as well as on 9,000 pairwise combinations of concatenated texts. Threshold statistical values are identiﬁed that make it possible to distinguish homogeneous texts and texts with a possible stylistic change point. An approximation of the Kolmogorov distribution is used to assess statistical signiﬁcance. P-values are calculated on the basis of the limiting Kolmogorov distribution. The results of the experiments show a stable ability of the method to identify a stylistic change point between the novels by diﬀerent authors; the eﬃciency of the algorithm in distinguishing one text by one author and a combination of two texts by one author; the eﬃciency of the algorithm in distinguishing one text by one author and a pair of texts by diﬀerent authors; independence of the method of the text length, that is, high and low values of statistics and p-values are observed for diﬀerent text lengths. The analysis of empirical bridges and the behavior of individual function words show that stylistic features of the text can manifest themselves both on the scale of the entire work and within its individual parts. The method eﬀectively records such changes. The developed approach can be used in the framework of corpus linguistics, applied stylistics, natural language processing, as well as in the creation of intelligent text analysis systems.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>служебные слова</kwd><kwd>авторский инвариант</kwd><kwd>эмпирический мост</kwd><kwd>обнаружение разладки</kwd><kwd>склейка текстов</kwd></kwd-group><kwd-group xml:lang="en"><kwd>function words</kwd><kwd>author’s invariant</kwd><kwd>empirical bridge</kwd><kwd>change point detection</kwd><kwd>concatenation</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Гусарова Г. В., Ковалевский А. П., Макаренко А. Г. Критерии наличия разладки // Сибирский журнал индустриальной математики. 2005. Т. 8, № 4. С. 18–33. URL: https://www.mathnet.ru/sjim273</mixed-citation><mixed-citation xml:lang="en">Gusarova G. V., Kovalevskii A. P., Makarenko A. G. Criteria for the existence of a change point.</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Abebe B., Chebunin M., Kovalevskii A. Text Segmentation Via Processes that Count the Number of Different Words Forward and Backward // Journal of Quantitative Linguistics. 2024. Vol. 31. No. 1. P. 1–18. DOI: https://doi.org/10.1080/09296174.2023.2275342</mixed-citation><mixed-citation xml:lang="en">Sibirskii Zhurnal Industrial’noi Matematiki, 2005, vol. 8, no. 4, pp. 18–33. (in Russ.)</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Abebe B., Chebunin M., Kovalevskii A., Zakrevskaya N. Statistical tests for text homogeneity: Using forward and backward processes of numbers of different words // Glottometrics. 2022. Vol. 53. No. 1. P. 42–58. DOI: https://doi.org/10.53482/2022_53_401</mixed-citation><mixed-citation xml:lang="en">Abebe B., Chebunin M., Kovalevskii A. Text Segmentation Via Processes that Count the Number of Diﬀerent Words Forward and Backward. Journal of Quantitative Linguistics, vol. 31, no. 1, 2024, pp. 1–18. DOI: https://doi.org/10.1080/09296174.2023.2275342</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Beeferman D., Berger A., Lafferty J. D. Statistical models for text segmentation // Machine Learning. 1999. Vol. 34. No. 1–3. P. 177–210. DOI: https://doi.org/10.1023/A:1007506220214</mixed-citation><mixed-citation xml:lang="en">Abebe B., Chebunin M., Kovalevskii A., Zakrevskaya N. Statistical tests for text homogeneity: Using forward and backward processes of numbers of diﬀerent words. Glottometrics, vol. 53, no. 1, 2022, pp. 42–58. DOI: https://doi.org/10.53482/2022_53_401</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Choi F. Y. Y. Advances in domain independent linear text segmentation // 1st Meeting of the North American Chapter of the Association for Computational Linguistics. URL: https://aclanthology.org/A00-2004.pdf</mixed-citation><mixed-citation xml:lang="en">Beeferman D., Berger A., Laﬀerty J. D. Statistical models for text segmentation. Machine Learning, vol. 34, no. 1–3, 1999, pp. 177–210. DOI: https://doi.org/10.1023/A:1007506220214</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Dembele S., Lo G. S. Probabilistic, Statistical and Algorithmic Aspects of the Similarity of Texts and Application to Gospels Comparison // Journal of Data Analysis and Information Processing. 2015. Vol. 3. P. 112–127. DOI: https://doi.org/10.4236/jdaip.2015.34012</mixed-citation><mixed-citation xml:lang="en">Choi F. Y. Y. Advances in domain independent linear text segmentation. 1st Meeting of the North American Chapter of the Association for Computational Linguistics. URL: https://aclanthology.org/A00-2004.pdf</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Hearst M. A. Text tiling: Segmenting text into multi-paragraph subtopic passages // Computational Linguistics. 1997. Vol. 23. No. 1. P. 33–64.</mixed-citation><mixed-citation xml:lang="en">Dembele S., Lo G. S. Probabilistic, Statistical and Algorithmic Aspects of the Similarity of Texts and Application to Gospels Comparison. Journal of Data Analysis and Information Processing, 2015, vol. 3, pp. 112–127. DOI: https://doi.org/10.4236/jdaip.2015.34012</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Itoh N., Kurths J. Change-point detection of climate time series by nonparametric method // Proceedings of the World Congress on Engineering and Computer Science. 2010. Vol. 1. P. 445–448.</mixed-citation><mixed-citation xml:lang="en">Hearst M. A. Text tiling: Segmenting text into multi-paragraph subtopic passages. Computational Linguistics, 1997, vol. 23, no. 1, pp. 33–64.</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Mikolov T., Sutskever I., Chen K., Corrado G. S., Dean J. Distributed representations of words and phrases and their compositionality // Advances in Neural Information Processing Systems. 2013. Vol. 26. P. 1–9.</mixed-citation><mixed-citation xml:lang="en">Itoh N., Kurths J. Change-point detection of climate time series by nonparametric method. Proceedings of the World Congress on Engineering and Computer Science, 2010, vol. 1, pp. 445–448.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Nanni G., Glavaš F., Ponzetto S. P. Unsupervised text segmentation using semantic relatedness graphs // In: Proceedings of the Fifth Joint Conference on Lexical and Computational Semantics (*SEM). Berlin, Germany, 2016.</mixed-citation><mixed-citation xml:lang="en">Mikolov T., Sutskever I., Chen K., Corrado G. S., Dean J. Distributed representations of words and phrases and their compositionality. Advances in Neural Information Processing Systems, 2013, vol. 26, pp. 1–9.</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Németh G., Zainkó C. Multilingual Statistical Text Analysis, Zipf’s Law and Hungarian Speech Generation // Acta Linguistica Hungarica. 2002. Vol. 49. No. 3–4. P. 385–405.</mixed-citation><mixed-citation xml:lang="en">Nanni G., Glavaš F., Ponzetto S. P. Unsupervised text segmentation using semantic relatedness graphs. In: Proceedings of the Fifth Joint Conference on Lexical and Computational Semantics (*SEM). Berlin, Germany, 2016, pp. 125–130.</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Pevzner L., Hearst M. A. A critique and improvement of an evaluation metric for text segmentation // Computational Linguistics. 2002. Vol. 28. No. 1. P. 19–36. DOI: https://doi.org/10.1162/089120102317341756</mixed-citation><mixed-citation xml:lang="en">Németh G., Zainkó C. Multilingual Statistical Text Analysis, Zipf’s Law and Hungarian Speech Generation. Acta Linguistica Hungarica, 2002, vol. 49, no. 3–4, pp. 385–405.</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Pevzner L., Hearst M. A. A critique and improvement of an evaluation metric for text segmentation. Computational Linguistics, 2002, vol. 28, no. 1, pp. 19–36. DOI: https://doi.org/10.1162/089120102317341756</mixed-citation><mixed-citation xml:lang="en">Pevzner L., Hearst M. A. A critique and improvement of an evaluation metric for text segmentation. Computational Linguistics, 2002, vol. 28, no. 1, pp. 19–36. DOI: https://doi.org/10.1162/089120102317341756</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
