<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE root>
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.2" xml:lang="en"><front><journal-meta><journal-id journal-id-type="publisher-id">Discrete and Continuous Models and Applied Computational Science</journal-id><journal-title-group><journal-title xml:lang="en">Discrete and Continuous Models and Applied Computational Science</journal-title><trans-title-group xml:lang="ru"><trans-title>Discrete and Continuous Models and Applied Computational Science</trans-title></trans-title-group></journal-title-group><issn publication-format="print">2658-4670</issn><issn publication-format="electronic">2658-7149</issn><publisher><publisher-name xml:lang="en">Peoples' Friendship University of Russia named after Patrice Lumumba (RUDN University)</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">51919</article-id><article-id pub-id-type="doi">10.22363/2658-4670-2026-34-2-160-174</article-id><article-id pub-id-type="edn">JHMSMG</article-id><article-categories><subj-group subj-group-type="toc-heading" xml:lang="en"><subject>Computer Science</subject></subj-group><subj-group subj-group-type="toc-heading" xml:lang="ru"><subject>Информатика и вычислительная техника</subject></subj-group><subj-group subj-group-type="article-type"><subject>Research Article</subject></subj-group></article-categories><title-group><article-title xml:lang="en">Modeling authorship attribution for short texts under training corpus degradation</article-title><trans-title-group xml:lang="ru"><trans-title>Моделирование атрибуции авторства коротких текстов в условиях деградации обучающего корпуса</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0000-1985-1401</contrib-id><name-alternatives><name xml:lang="en"><surname>Khvostenko</surname><given-names>Viktor M.</given-names></name><name xml:lang="ru"><surname>Хвостенко</surname><given-names>В. М.</given-names></name></name-alternatives><bio xml:lang="en"><p>Researcher of Cybersecurity CPS Department of HSE University</p></bio><email>vkhvostenko@hse.ru</email><xref ref-type="aff" rid="aff1"/></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0007-8777-8622</contrib-id><name-alternatives><name xml:lang="en"><surname>Kovalenko</surname><given-names>Andrey P.</given-names></name><name xml:lang="ru"><surname>Коваленко</surname><given-names>А. П.</given-names></name></name-alternatives><bio xml:lang="en"><p>Professor, Doctor of Technical Sciences, Full member of Academy of Cryptography of the Russian</p></bio><email>kovalenko.ap@cryptoacademy.gov.ru</email><xref ref-type="aff" rid="aff2"/></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-9023-9896</contrib-id><name-alternatives><name xml:lang="en"><surname>Melnikov</surname><given-names>Sergey Yu.</given-names></name><name xml:lang="ru"><surname>Мельников</surname><given-names>С. Ю.</given-names></name></name-alternatives><bio xml:lang="en"><p>Doctor of Physical and Mathematical Sciences, Professor of Department of Probability Theory and Cyber Security of RUDN University; Chief Researcher of Cybersecurity CPS Department of HSE University</p></bio><email>melnikov-syu@rudn.ru</email><xref ref-type="aff" rid="aff1"/><xref ref-type="aff" rid="aff3"/></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-1129-8434</contrib-id><name-alternatives><name xml:lang="en"><surname>Meshcheryakov</surname><given-names>Roman V.</given-names></name><name xml:lang="ru"><surname>Мещеряков</surname><given-names>Р. В.</given-names></name></name-alternatives><bio xml:lang="en"><p>Professor, Doctor of Technical Sciences, Chief Researcher of Cybersecurity CPS Department of HSE University; Head of the Laboratory, Institute of Control Science of RAS</p></bio><email>mrv@ieee.org</email><xref ref-type="aff" rid="aff1"/><xref ref-type="aff" rid="aff4"/></contrib></contrib-group><aff-alternatives id="aff1"><aff><institution xml:lang="en">HSE University</institution></aff><aff><institution xml:lang="ru">Национальный исследовательский университет «Высшая школа экономики»</institution></aff></aff-alternatives><aff-alternatives id="aff2"><aff><institution xml:lang="en">Academy of Cryptography of the Russian Federation</institution></aff><aff><institution xml:lang="ru">Академия криптографии Российской Федерации</institution></aff></aff-alternatives><aff-alternatives id="aff3"><aff><institution xml:lang="en">RUDN University</institution></aff><aff><institution xml:lang="ru">Российский университет дружбы народов</institution></aff></aff-alternatives><aff-alternatives id="aff4"><aff><institution xml:lang="en">Institute of Control Science of RAS, ICS RAS</institution></aff><aff><institution xml:lang="ru">Институт проблем управления им. В. А. Трапезникова Российской академии наук</institution></aff></aff-alternatives><pub-date date-type="pub" iso-8601-date="2026-08-15" publication-format="electronic"><day>15</day><month>08</month><year>2026</year></pub-date><volume>34</volume><issue>2</issue><issue-title xml:lang="en">VOL 34, NO2 (2026)</issue-title><issue-title xml:lang="ru">ТОМ 34, №2 (2026)</issue-title><fpage>160</fpage><lpage>174</lpage><history><date date-type="received" iso-8601-date="2026-08-20"><day>20</day><month>08</month><year>2026</year></date></history><permissions><copyright-statement xml:lang="en">Copyright ©; 2026, Khvostenko V.M., Kovalenko A.P., Melnikov S.Y., Meshcheryakov R.V.</copyright-statement><copyright-statement xml:lang="ru">Copyright ©; 2026, Хвостенко В.М., Коваленко А.П., Мельников С.Ю., Мещеряков Р.В.</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="en">Khvostenko V.M., Kovalenko A.P., Melnikov S.Y., Meshcheryakov R.V.</copyright-holder><copyright-holder xml:lang="ru">Хвостенко В.М., Коваленко А.П., Мельников С.Ю., Мещеряков Р.В.</copyright-holder><ali:free_to_read xmlns:ali="http://www.niso.org/schemas/ali/1.0/"/><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/">https://creativecommons.org/licenses/by-nc/4.0</ali:license_ref></license></permissions><self-uri xlink:href="https://journals.rudn.ru/miph/article/view/51919">https://journals.rudn.ru/miph/article/view/51919</self-uri><abstract xml:lang="en"><p>Attributing authorship of short texts is a complex task, and its accuracy heavily depends on the quality and quantity of training data. In many practical scenarios, however, this data is subject to degradation. This study examines how two specific types of degradation - textual corruption and reduction in corpus size - affect the performance of six standard classifiers for authorship attribution of English-language tweets. We model textual corruption as random and independent character-level distortions applied to the texts in the training corpora. The test set (the texts whose authorship is to be attributed) remains unchanged. The experimental setup involves 50 authors, with 200 test texts. The volume of training texts per author varies from 800 to 50 (across 16 incremental steps), and the distortion level ranges from 0 to 20\% (6 gradations). For each combination of corpus size and distortion level, we train and test six classifiers: Support Vector Machines (SVM), Logistic Regression, Naive Bayes, Random Forest, Decision Tree, and $k$-Nearest Neighbors (kNN). Authorial style is represented using Bag of Words (BOW), TF-IDF, word token $N$-grams, and their combinations. The results show that for most classifiers, degradation in the training sets leads to comparable reductions in attribution accuracy. The highest overall performance was achieved by the Support Vector Machine with TF-IDF features. It proved to be the most robust method both when the number of training texts was reduced and under high distortion levels. In the absence of distortions, SVM with TF-IDF achieved an accuracy of 0.92; at distortion levels of 5 and 20\%, its accuracy reached 0.89 and 0.75, respectively. Logistic Regression with Bag of Words ranked second, delivering accuracies of 0.9, 0.87, and 0.74 under the same conditions. The kNN method demonstrated the lowest accuracy among the classifiers examined.</p></abstract><trans-abstract xml:lang="ru"><p>Точность атрибуции авторства текстов критически зависит от качества и объема обучающих данных. Изучается, как два типа деградации обучающих авторских коллекций, искажение текста и сокращение размера корпуса, влияют на точность шести стандартных классификаторов при атрибуции авторства англоязычных твитов. Рассматриваются случайные независимые искажения типа «замена символов», применяемые к текстам в обучающих корпусах. Тестовая выборка остается неизменной. Эксперименты проводились на текстах 50 авторов (по 200 тестовых текстов). Количество обучающих текстов одного автора варьировалось от 800 до 50 (16 промежуточных шагов), а уровень искажения - от 0 до 20\% (6 градаций). Для каждой комбинации размера корпуса и уровня искажения обучались и тестировались шесть классификаторов: метод опорных векторов (SVM), логистическую регрессию, наивный байесовский классификатор, случайный лес, дерево решений и метод $k$-ближайших соседей (kNN). В качестве признаков авторского стиля использовались мешок слов (BoW) и TF-IDF для токенов и $N$-грамм. Установлено, что для большинства классификаторов деградация обучающих данных приводит к сопоставимому снижению точности атрибуции. Наилучшие результаты показал метод опорных векторов с TF-IDF. Он оказался наиболее устойчивым как при сокращении количества обучающих текстов, так и при высоких уровнях искажений. В отсутствие искажений метод SVM с TF-IDF достиг точности 0.92; при уровнях искажений 5 и 20\% его точность составила 0.89 и 0.75 соответственно. Логистическая регрессия с с мешком слов заняла второе место, показав точность 0.9, 0.87 и 0.74 в тех же условиях. Метод kNN продемонстрировал наименьшую точность среди рассмотренных классификаторов.</p></trans-abstract><kwd-group xml:lang="en"><kwd>Authorship Attribution</kwd><kwd>Short Texts</kwd><kwd>Training Corpus  Degradation</kwd><kwd>Textual Corruption</kwd><kwd>Classifier Performance</kwd></kwd-group><kwd-group xml:lang="ru"><kwd>атрибуция авторства</kwd><kwd>короткие тексты</kwd><kwd>ухудшение качества обучающего корпуса</kwd><kwd>искажение текста</kwd><kwd>производительность классификатора</kwd></kwd-group><funding-group><award-group><funding-source><institution-wrap><institution xml:lang="en">The study was supported by the Russian Science Foundation grant No. 24-11-00340 (https://rscf.ru/project/24-11- 00340/).</institution></institution-wrap></funding-source></award-group></funding-group></article-meta><fn-group/></front><body></body><back><ref-list><ref id="B1"><label>1.</label><mixed-citation>M. Eder, “Does size matter? Authorship attribution, small samples, big problem,” Digital Scholarship in the Humanities, vol. 30, pp. 167-182, 2015. DOI: 10.1093/llc/fqt065</mixed-citation></ref><ref id="B2"><label>2.</label><mixed-citation>G. Franzini, M. Kestemont, G. Rotari, M. Jander, J. Ochab, J. Byszuk, and M. Eder, “Attributing authorship in the noisy digitized correspondence of Jacob and Wilhelm Grimm,” Frontiers in Digital Humanities, vol. 5, p. 4, 2018. DOI: 10.3389/fdigh.2018.00004</mixed-citation></ref><ref id="B3"><label>3.</label><mixed-citation>K. Luyckx and W. Daelemans, “Authorship attribution and verification with many authors and limited data,” in Proceedings of the 22nd International Conference on Computational Linguistics (COLING 2008), 2008, pp. 513-520. DOI: 10.3115/1599081.1599147</mixed-citation></ref><ref id="B4"><label>4.</label><mixed-citation>D. Lopresti, “Optical character recognition errors and their effects on natural language processing,” in Proceedings of the 3rd Workshop on Analytics for Noisy Unstructured Text Data, 2009, pp. 115-122. DOI: 10.1145/1568293.1568312</mixed-citation></ref><ref id="B5"><label>5.</label><mixed-citation>M. Kubis, P. Szymański, M. Ziółko, and K. Wróbel, “Open challenge for correcting errors of speech recognition systems,” in Language and Technology Conference, 2019, pp. 322-337. DOI: 10.1007/978-3-030-33241-5_27</mixed-citation></ref><ref id="B6"><label>6.</label><mixed-citation>A. A. Goncharov, N. V. Buntman, and V. A. Nuriyev, “Errors in machine translation: Classification problems,” Sistemy i Sredstva Informatiki, vol. 29, no. 3, pp. 92-103, 2019, (in Russian). DOI: 10.14357/08696527190308</mixed-citation></ref><ref id="B7"><label>7.</label><mixed-citation>E. Dawson and L. Nielsen, “Automated cryptanalysis of XOR plaintext strings,” Cryptologia, vol. 20, no. 2, pp. 165-181, 1996. DOI: 10.1080/0161-119691884871</mixed-citation></ref><ref id="B8"><label>8.</label><mixed-citation>L. V. Subramaniam, S. Roy, T. A. Faruquie, and S. Negi, “A survey of types of text noise and techniques to handle noisy text,” in Proceedings of the 3rd Workshop on Analytics for Noisy Unstructured Text Data, 2009, pp. 115-122. DOI: 10.1145/1568293.1568312</mixed-citation></ref><ref id="B9"><label>9.</label><mixed-citation>G. Sperduti and A. Moreo, Misspellings in natural language processing: A survey, 2025. arXiv: 2501.16836.</mixed-citation></ref><ref id="B10"><label>10.</label><mixed-citation>H. Jing, D. Lopresti, and C. Shih, “Summarizing noisy documents,” in Proceedings of the Symposium on Document Image Understanding Technology, 2003, pp. 111-119.</mixed-citation></ref><ref id="B11"><label>11.</label><mixed-citation>N. Moratanch and S. Chitrakala, “A survey on extractive text summarization,” in International Conference on Computer, Communication and Signal Processing (ICCCSP), 2017, pp. 1-6. DOI: 10.1109/ICCCSP.2017.7944061</mixed-citation></ref><ref id="B12"><label>12.</label><mixed-citation>D. D. Walker, W. B. Lund, and E. K. Ringger, “Evaluating models of latent document semantics in the presence of OCR errors,” in Proceedings of the 2010 Conference on Empirical Methods in Natural Language Processing, 2010, pp. 1-10.</mixed-citation></ref><ref id="B13"><label>13.</label><mixed-citation>E. Stamatatos, “On the robustness of authorship attribution based on character n-gram features,” Journal of Law and Policy, vol. 21, p. 7, 2013.</mixed-citation></ref><ref id="B14"><label>14.</label><mixed-citation>M. Eder, “Mind your corpus: Systematic errors in authorship attribution,” Literary and Linguistic Computing, vol. 28, no. 4, pp. 603-614, 2013. DOI: 10.1093/llc/fqt039</mixed-citation></ref><ref id="B15"><label>15.</label><mixed-citation>N. Mamaev, X. Piotrowska, M. Marusenko, and A. Ronzhin, “Burrows’s delta for authorship attribution of Russian literary texts,” in CEUR Workshop Proceedings, vol. 2233, 2017.</mixed-citation></ref><ref id="B16"><label>16.</label><mixed-citation>H. Sayoud et al., “Automatic authorship attribution of noisy documents,” in Proceedings of the FLAIRS Conference, 2017, pp. 202-205.</mixed-citation></ref><ref id="B17"><label>17.</label><mixed-citation>H. Benzerroug and S. Khennouf, “Author identification of corrupted OCR-based texts,” HDSKD Journal, vol. 3, pp. 91-99, 2017.</mixed-citation></ref><ref id="B18"><label>18.</label><mixed-citation>Z. Hamadache and H. Sayoud, “Authorship attribution of noisy text data with a comparative study of clustering methods,” International Journal of Knowledge and Systems Science, vol. 9, no. 2, pp. 45-69, 2018. DOI: 10.4018/IJKSS.2018040103</mixed-citation></ref><ref id="B19"><label>19.</label><mixed-citation>N. Bindal, P. Singh, V. Singh, and D. Gupta, “A systematic review of state-of-the-art noise removal techniques in digital images,” Multimedia Tools and Applications, vol. 81, pp. 31 529- 31 552, 2022. DOI: 10.1007/s11042-022-12980-7</mixed-citation></ref><ref id="B20"><label>20.</label><mixed-citation>S. Colutto, P. Kahle, S. Guenter, and G. Muehlberger, “Transkribus: A platform for automated text recognition and searching of historical documents,” in 15th International Conference on eScience (eScience), 2019, pp. 463-466. DOI: 10.1109/eScience.2019.00073</mixed-citation></ref><ref id="B21"><label>21.</label><mixed-citation>F. Pedregosa, G. Varoquaux, A. Gramfort, V. Michel, B. Thirion, O. Grisel, M. Blondel, P. Prettenhofer, R. Weiss, V. Dubourg, et al., “Scikit-learn: Machine learning in Python,” Journal of Machine Learning Research, vol. 12, pp. 2825-2830, 2011.</mixed-citation></ref><ref id="B22"><label>22.</label><mixed-citation>R. Schwartz, O. Tsur, A. Rappoport, and M. Koppel, “Authorship attribution of micro-messages,” in Proceedings of the 2013 Conference on Empirical Methods in Natural Language Processing, 2013, pp. 1880-1891.</mixed-citation></ref><ref id="B23"><label>23.</label><mixed-citation>C. Suman, S. Saha, and P. Bhattacharyya, “Authorship attribution of microtext using capsule networks,” IEEE Transactions on Computational Social Systems, vol. 9, no. 4, pp. 1038-1047, 2021. DOI: 10.1109/TCSS.2021.3073690</mixed-citation></ref><ref id="B24"><label>24.</label><mixed-citation>T. Alsanoosy, B. Shalbi, and A. Noor, “Authorship attribution for English short texts,” Engineering Technology and Applied Science Research, vol. 14, no. 5, pp. 16 419-16 426, 2024. DOI: 10.48084/etasr.7794</mixed-citation></ref><ref id="B25"><label>25.</label><mixed-citation>V. M. Khvostenko et al., “Two-parameter model of synthetic distortions in the problem of assessing the readability of distorted texts,” The European Physical Journal Special Topics, vol. 234, pp. 3865-3870, 2025. DOI: 10.1140/epjs/s11734-025-01567-8</mixed-citation></ref><ref id="B26"><label>26.</label><mixed-citation>K. P. Murphy, Probabilistic Machine Learning: An Introduction. MIT Press, 2022.</mixed-citation></ref></ref-list></back></article>
