<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">inform</journal-id><journal-title-group><journal-title xml:lang="ru">Информатика</journal-title><trans-title-group xml:lang="en"><trans-title>Informatics</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">1816-0301</issn><issn pub-type="epub">2617-6963</issn><publisher><publisher-name>UIIP NASB</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.37661/1/1816-0301-2026-23-3-76-91</article-id><article-id custom-type="elpub" pub-id-type="custom">inform-1416</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>ИНФОРМАЦИОННЫЕ ТЕХНОЛОГИИ</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>INFORMATION TECHNOLOGY</subject></subj-group></article-categories><title-group><article-title>Трехвекторный алгоритм обнаружения русскоязычных нейросетевых текстовых фрагментов на основе вероятностного анализа</article-title><trans-title-group xml:lang="en"><trans-title>A three-vector algorithm for detecting Russian-language neural-network-generated text fragments based on probabilistic analysis</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Крез</surname><given-names>К. С.</given-names></name><name name-style="western" xml:lang="en"><surname>Krez</surname><given-names>K. S.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Крез Карина Сергеевна , аспирант, ассистент кафедры проектирования информационно-компьютерных систем</p><p>ул. П. Бровки, 6, Минск, 220013</p></bio><bio xml:lang="en"><p>Karyna S. Krez , Postgraduate Student, Assistant of the Department of Information and Computer Systems Design</p><p>st. P. Brovki, 6, Minsk, 220013</p></bio><email xlink:type="simple">karinakrez04@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Шнейдеров</surname><given-names>Е. Н.</given-names></name><name name-style="western" xml:lang="en"><surname>Shneiderov</surname><given-names>Y. N.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Шнейдеров Евгений Николаевич , кандидат технических наук, доцент, доцент кафедры проектирования информационно-компьютерных систем, проректор по учебной работе</p><p>ул. П. Бровки, 6, Минск, 220013</p></bio><bio xml:lang="en"><p>Yevgeny N. Shneiderov , Cand. Sci. (Eng.), Assoc. Prof., Assoc. Prof. of the Department of Information and Computer Systems Design, Vice-Rector for Academic Affairs</p><p>st. P. Brovki, 6, Minsk, 220013</p></bio><email xlink:type="simple">shneiderov@bsuir.by</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Галяк</surname><given-names>И. П.</given-names></name><name name-style="western" xml:lang="en"><surname>Galyak</surname><given-names>I. P.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Галяк Иван Павлович , студент кафедры проектирования информационно-компьютерных систем</p><p>ул. П. Бровки, 6, Минск, 220013</p></bio><bio xml:lang="en"><p>Ivan P. Galyak , Student at the Department of Information and Computer Systems Design</p><p>st. P. Brovki, 6, Minsk, 220013</p></bio><email xlink:type="simple">haliak.ivanba@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ефименко</surname><given-names>Д. Д.</given-names></name><name name-style="western" xml:lang="en"><surname>Efimenko</surname><given-names>D. D.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Ефименко Даниил Дмитриевич , студент кафедры проектирования информационно-компьютерных систем</p><p>ул. П. Бровки, 6, Минск, 220013</p></bio><bio xml:lang="en"><p>Daniil D. Efimenko , Student at the Department of Information and Computer Systems Design</p><p>st. P. Brovki, 6, Minsk, 220013</p></bio><email xlink:type="simple">Dhmainer@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Белорусский государственный университет информатики и радиоэлектроники</institution><country>Беларусь</country></aff><aff xml:lang="en"><institution>Belarusian State University of Informatics and Radioelectronics</institution><country>Belarus</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>26</day><month>09</month><year>2026</year></pub-date><volume>23</volume><issue>3</issue><fpage>76</fpage><lpage>91</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Крез К.С., Шнейдеров Е.Н., Галяк И.П., Ефименко Д.Д., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Крез К.С., Шнейдеров Е.Н., Галяк И.П., Ефименко Д.Д.</copyright-holder><copyright-holder xml:lang="en">Krez K.S., Shneiderov Y.N., Galyak I.P., Efimenko D.D.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://inf.grid.by/jour/article/view/1416">https://inf.grid.by/jour/article/view/1416</self-uri><abstract><sec><title>Цели</title><p>Цели. Целями работы являются разработка и программная реализация трехвекторного алгоритма обнаружения русскоязычных нейросетевых текстовых фрагментов на основе вероятностного анализа. Объект исследования русскоязычные тексты, предмет статистические признаки их происхождения. Особое внимание уделяется количественному сравнению предложенного подхода с базовыми одновекторными методами обнаружения детектором на основе перплексии и методом рангового анализа GLTR.</p></sec><sec><title>Методы</title><p>Методы. Предложенный алгоритм объединяет три вектора признаков: предсказуемость токенов, долю редких лексем и информационную плотность текста, оцениваемую через коэффициент алгоритмического сжатия. В качестве вероятностного ядра использована авторегрессионная языковая модель rugpt3small. Реализация выполнена на языке Python с применением библиотек transformers, PyTorch, NLTK и zlib. Экспериментальная проверка проводилась на выборке из 100 смешанных документов (16 000 предложений) пяти тематических групп, нейросетевые фрагменты которых сгенерированы четырьмя языковыми моделями: DeepSeek, GPT, Qwen и YandexAI. Качество оценивалось по отклонению от эталонной доли нейросетевого текста и классификационным метрикам.</p></sec><sec><title>Результаты</title><p>Результаты. После применения калибровочной поправки средний индекс по выборке составил 45,2 балла при эталонном уровне 50, а значения для всех четырех моделей находились в узком диапазоне 44,4–45,9 балла, что свидетельствует об устойчивости алгоритма к выбору генеративной модели. Отклонение среднего индекса от эталона составило 4,8 балла против 11,1 балла у метода на основе перплексии и 7,6 балла у GLTR; F1-мера на уровне предложений достигла 0,76 против 0,63 и 0,66 соответственно.</p></sec><sec><title>Заключение</title><p>Заключение. Трехвекторный алгоритм подтвердил свою эффективность в задаче обнаружения нейросетевых фрагментов и может служить основой для инструментов проверки академических текстов и систем антиплагиата. Дальнейшее развитие исследования связано с расширением выборки, проверкой устойчивости к перефразированию и редактированию, а также проведением абляционного анализа.</p></sec></abstract><trans-abstract xml:lang="en"><sec><title>Objectives</title><p>Objectives. The purpose of the work is to develop and programmatically implement a three-vector algorithm for detecting Russian-language neural network text fragments based on probabilistic analysis. The object of the study is Russian-language texts, the subject is statistical signs of their origin. Special attention is paid to the quantitative comparison of the proposed approach with the basic single-vector detection methods the perplexy detector and the GLTR rank analysis method.</p></sec><sec><title>Methods</title><p>Methods. The proposed algorithm combines three feature vectors: token predictability, the proportion of rare lexemes, and the information density of the text, estimated via the algorithmic compression ratio. The author's regression language model rugpt3small is used as a probabilistic core. The implementation is made in Python using the transformers, PyTorch, NLTK, and zlib libraries. The experimental validation was conducted on a sample of 100 mixed documents (16,000 sentences) of five thematic groups, the neural network fragments of which were generated by four language models DeepSeek, GPT, Qwen and YandexAI. The quality was assessed by the deviation from the reference proportion of the neural network text and by classification metrics.</p></sec><sec><title>Results</title><p>Results. After applying the calibration adjustment, the average index for the sample was 45.2 points against a reference level of 50, and the values for all four models were in the narrow range of 44.445.9 points, which indicates the algorithm's stability to the choice of a generative model. The deviation of the average index from the standard was 4.8 points versus 11.1 points for the method based on perplexity and 7.6 points for GLTR; the F1 measure at the sentence level reached 0.76 versus 0.63 and 0.66, respectively.</p></sec><sec><title>Conclusion</title><p>Conclusion. The three-vector algorithm has proven its effectiveness in the task of detecting neural network fragments and can serve as the basis for tools for verifying academic texts and anti-plagiarism systems. Further development of the research is related to the expansion of the sample, testing the resistance to paraphrasing and editing, as well as conducting ablative analysis</p></sec></trans-abstract><kwd-group xml:lang="ru"><kwd>нейросетевой текст</kwd><kwd>вероятностный анализ</kwd><kwd>энтропия</kwd><kwd>частотный анализ</kwd><kwd>язык программирования Python</kwd><kwd>плагиат</kwd></kwd-group><kwd-group xml:lang="en"><kwd>neural-network text</kwd><kwd>probabilistic analysis</kwd><kwd>entropy</kwd><kwd>frequency analysis</kwd><kwd>Python programming language</kwd><kwd>plagiarism</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Ивахненко, Е. Н. ChatGPT в высшем образовании и науке: угроза или ценный ресурс? / Е. Н. Ивахненко, В. С. Никольский // Высшее образование в России. – 2023. – Т. 32, № 4. – С. 9–22.</mixed-citation><mixed-citation xml:lang="en">Ivakhnenko E. N., Nikolsky V. S. ChatGPT in higher education and science: A threat or a valuable resource? Vysshee obrazovanie v Rossii [Higher Education in Russia], 2023, vol. 32, no. 4, рр. 9–22 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Федотова, А. М. Методика идентификации текстов, сгенерированных большими языковыми моделями / А. М. Федотова, А. С. Романов // Информатика и автоматизация. – 2025. – Т. 24, № 5. – С. 1444–1470.</mixed-citation><mixed-citation xml:lang="en">Fedotova A. M., Romanov A. S. Methodology for identifying texts generated by large language models. Informatika i avtomatizatsiya [Computer Science and Automation], 2025, vol. 24, no. 5, рр. 1444–1470 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Мачковская, Л. Я. Генеративные модели искусственного интеллекта в преподавании русского языка как иностранного: возможности, ограничения и риски / Л. Я. Мачковская, А. В. Фатина, О. А. Ветошкина // Международный журнал гуманитарных и естественных наук. – 2025. – № 11-1. – С. 73–78.</mixed-citation><mixed-citation xml:lang="en">Machkovskaya L. Ya., Fatina A. V., Vetoshkina O. A. Generative artificial intelligence models in teaching Russian as a foreign language: Opportunities, limitations, and risks. Mezhdunarodnyy zhurnal gumanitarnykh i estestvennykh nauk [International Journal of Humanities and Natural Sciences], 2025, no. 11-1, pp. 73–78 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Айдагулова, А. Р. Особенности текстов, сгенерированных искусственным интеллектом / А. Р. Айдагулова // Вестник Башкирского государственного педагогического университета им. М. Акмуллы. – 2023. – № 4 (72). – С. 154–156.</mixed-citation><mixed-citation xml:lang="en">Aydagulova A. R. Features of texts generated by artificial intelligence. Vestnik Bashkirskogo gosudarstvennogo pedagogicheskogo universiteta im. M. Akmully [Bulletin of M. Akmulla Bashkir State Pedagogical University], 2023, no. 4 (72), рр. 154–156 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Брызгалина, Е. В. Вызовы технологий искусственного интеллекта для этической экспертизы исследований живых систем / Е. В. Брызгалина // Теоретическая и прикладная этика: традиции и перспективы : материалы XVI Междунар. конф., Санкт-Петербург, 16–18 нояб. 2023 г. – СПб. : Изд-во С.-Петерб. гос. ун-та, 2024. – С. 57–58.</mixed-citation><mixed-citation xml:lang="en">Bryzgalina E. V. Challenges of artificial intelligence technologies for the ethical review of living systems research. Teoreticheskaya i prikladnaya etika: traditsii i perspektivy: materialy XVI Mezhdunarodnoj konferencii, Sankt-Peterburg, 16–18 nojabrja 2023 g. [Theoretical and Applied Ethics: Traditions and Prospects: Proceedings of the 16th International Conference, Saint Petersburg, 16–18 November 2023]. Saint Petersburg, Izdatel'stvo Sankt-Peterburgskogo gosudarstvennogo universiteta, 2024, pp. 57–58 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Вейс, И. А. Распознавание паттернов применения искусственного интеллекта при создании текстов / И. А. Вейс // Современные инновации, системы и технологии. – 2025. – Т. 5, № 1. – С. 1033–1040.</mixed-citation><mixed-citation xml:lang="en">Veys I. A. Recognition of artificial intelligence application patterns in text creation. Sovremennye innovatsii, sistemy i tekhnologii [Modern Innovations, Systems and Technologies], 2025, vol. 5, no. 1, рр. 1033–1040 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Василевская, А. В. ИИ в количественном исследовании: проверка устойчивости к манипуляциям и искажениям фактов / А. В. Василевская // Теоретическая и прикладная этика: традиции и перспективы : материалы XVI Междунар. конф., Санкт-Петербург, 16–18 нояб. 2023 г. – СПб. : Изд-во С.-Петерб. гос. ун-та, 2024. – С. 67–68.</mixed-citation><mixed-citation xml:lang="en">Vasilevskaya A. V. AI in quantitative research: Testing resilience to manipulation and factual distortions. Teoreticheskaya i prikladnaya etika: traditsii i perspektivy: materialy XVI Mezhdunarodnoj konferencii, Sankt-Peterburg, 16–18 nojabrja 2023 g. [Theoretical and Applied Ethics: Traditions and Prospects: Proceedings of the 16th International Conference, Saint Petersburg, 16–18 November 2023]. Saint Petersburg, Izdatel'stvo Sankt-Peterburgskogo gosudarstvennogo universiteta, 2024, рр. 67–68 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Никитина, А. С. Человек или AI: к вопросу об авторстве / А. С. Никитина // Вестник молодых ученых и специалистов Самарского университета. – 2024. – № 1 (24). – С. 182–186.</mixed-citation><mixed-citation xml:lang="en">Nikitina A. S. Human or AI: On the issue of authorship. Vestnik molodykh uchenykh i spetsialistov Samarskogo universiteta [Bulletin of Young Scientists and Specialists of Samara University], 2024, no. 1 (24), рр. 182–186 (In Russ.).</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Gehrmann, S. GLTR: Statistical detection and visualization of generated text / S. Gehrmann, H. Strobelt, A. M. Rush // Proc. of the 57th Annual Meeting of the Association for Computational Linguistics: System Demonstrations, Florence, Italy, 28 July – 2 Aug. 2019. – Florence, 2019. – P. 111–116.</mixed-citation><mixed-citation xml:lang="en">Gehrmann S., Strobelt H., Rush A. M. GLTR: Statistical detection and visualization of generated text. Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics: System Demonstrations, Florence, Italy, 28 July – 2 August 2019. Florence, 2019, рр. 111–116.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">DetectGPT: Zero-shot machine-generated text detection using probability curvature / E. Mitchell, Y. Lee, A. Khazatsky [et al.] // Proc. of the 40th Intern. Conf. on Machine Learning (ICML 2023), Honolulu, Hawaii, USA, 23–29 July 2023. – Honolulu, 2023. – P. 24950–24962.</mixed-citation><mixed-citation xml:lang="en">Mitchell E., Lee Y., Khazatsky A., Manning C. D., Finn C. DetectGPT: Zero-shot machine-generated text detection using probability curvature. Proceedings of the 40th International Conference on Machine Learning (ICML 2023), Honolulu, Hawaii, USA, 23–29 July 2023. Honolulu, 2023, рр. 24950–24962.</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Watermark for large language models / J. Kirchenbauer, J. Geiping, Y. Wen [et al.] // Proc. of the 40th Intern. Conf. on Machine Learning (ICML 2023), Honolulu, Hawaii, USA, 23–29 July 2023. – Honolulu, 2023. – P. 17061–17084.</mixed-citation><mixed-citation xml:lang="en">Kirchenbauer J., Geiping J., Wen Y., Katz J., Miers I., Goldstein T. A watermark for large language models. Proceedings of the 40th International Conference on Machine Learning (ICML 2023), Honolulu, Hawaii, USA, 23–29 July 2023. Honolulu, 2023, рр. 17061–17084.</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
