<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">kaz29</journal-id><journal-title-group><journal-title xml:lang="ru">Вестник Казахстанско-Британского технического университета</journal-title><trans-title-group xml:lang="en"><trans-title>Herald of the Kazakh-British Technical University</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">1998-6688</issn><issn pub-type="epub">2959-8109</issn><publisher><publisher-name>Казахстанско-Британский Технический Университет</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.55452/1998-6688-2026-23-3-336-346</article-id><article-id custom-type="elpub" pub-id-type="custom">kaz29-3196</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>КОМПЬЮТЕРНЫЕ НАУКИ</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>COMPUTER SCIENCE</subject></subj-group></article-categories><title-group><article-title>БЕНЧМАРКИНГ МОДЕЛЕЙ ГЛУБОКОГО ОБУЧЕНИЯ ДЛЯ КЛАССИФИКАЦИИ КАЗАХСКИХ НОВОСТНЫХ ТЕКСТОВ В УСЛОВИЯХ ОГРАНИЧЕННОЙ РАЗМЕТКИ</article-title><trans-title-group xml:lang="en"><trans-title>BENCHMARKING DEEP LEARNING MODELS FOR FEW-SHOT CLASSIFICATION OF KAZAKH NEWS TEXTS UNDER LIMITED ANNOTATION</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0005-3266-2126</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Марламбеков</surname><given-names>Д.</given-names></name><name name-style="western" xml:lang="en"><surname>Marlambekov</surname><given-names>D.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Докторант</p><p>Алматы</p></bio><bio xml:lang="en"><p>PhD-student</p><p>Almaty</p></bio><email xlink:type="simple">dmarlambekov@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-7718-6478</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ахметова</surname><given-names>A.</given-names></name><name name-style="western" xml:lang="en"><surname>Akhmetova</surname><given-names>A.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Докторант</p><p>Алматы</p></bio><bio xml:lang="en"><p>PhD-student</p><p>Almaty</p></bio><email xlink:type="simple">baltabekova1994@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0009-0003-8161-533X</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Төрекул</surname><given-names>С.</given-names></name><name name-style="western" xml:lang="en"><surname>Torekul</surname><given-names>S.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Докторант</p><p>Алматы</p></bio><bio xml:lang="en"><p>PhD-studentAlmaty</p></bio><email xlink:type="simple">saule.torekul@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-3853-8896</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Уалиева</surname><given-names>И. М.</given-names></name><name name-style="western" xml:lang="en"><surname>Ualiyeva</surname><given-names>M.</given-names></name></name-alternatives><bio xml:lang="ru"><p>К.ф.-м.н., ассоциированный профессор</p><p>Алматы</p></bio><bio xml:lang="en"><p>Cand. Phys.-Math. Sc., Associate Professor</p><p>Almaty</p></bio><email xlink:type="simple">i.ualiyeva@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Казахский национальный университет им. аль-Фараби</institution><country>Казахстан</country></aff><aff xml:lang="en"><institution>Al-Farabi Kazakh National University</institution><country>Kazakhstan</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>26</day><month>09</month><year>2026</year></pub-date><volume>23</volume><issue>3</issue><fpage>336</fpage><lpage>346</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Марламбеков Д., Ахметова A., Төрекул С., Уалиева И.М., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Марламбеков Д., Ахметова A., Төрекул С., Уалиева И.М.</copyright-holder><copyright-holder xml:lang="en">Marlambekov D., Akhmetova A., Torekul S., Ualiyeva M.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://vestnik.kbtu.edu.kz/jour/article/view/3196">https://vestnik.kbtu.edu.kz/jour/article/view/3196</self-uri><abstract><p>Стремительное развитие методов глубокого обучения и появление трансформерных архитектур кардинально изменили ландшафт обработки естественного языка, однако эти достижения остаются неравномерно распределенными. В то время как для английского и других высокоресурсных языков существуют масштабные бенчмарки, для низкоресурсных языков, таких как казахский, проблема качественной классификации текстов остается острым вызовом. Авторами проведен сравнительный анализ классической модели эмбеддингов Word2Vec, рекуррентных моделей BiLSTM, трансформеров BERT и XLM-R, генеративной модели mT5, а также классических и статистических бейзлайнов для задачи few-shot классификации казахских новостных текстов в условиях ограниченной разметки. Эксперименты проведены на датасете KazNews для few-shot классификации, собранном из статей tengrinews.kz и inform.kz и параллельных корпусов HuffPost. Датасеты содержат пять категорий (спорт, политика, бизнес, путешествия и др.). Эксперименты в режимах 1-shot и 5-shot продемонстрировали, что ни одна из моделей не доминирует одновременно на всех датасетах и режимах: Word2Vec показал лучшие результаты на датасетах на корпусах HuffPost (0.48 и 0.53) при 5-shot режиме; mT5-Seq2Seq лучший на KazNews при 5-shot (0.71); BERT в целом конкурентоспособен при 5-shot на датасете KazNews с длинными текстами. Авторы также выявили резкий скачок качества для модели mT5-Seq2Seq в 5-shot. Для режима 1-shot ни одна из моделей не показала приемлемого результата качества.</p></abstract><trans-abstract xml:lang="en"><p>Deep learning methods and transformer architectures have fundamentally reshaped the field of natural language processing; however, these advancements remain unevenly distributed. While high-resource languages like English benefit from large-scale benchmarks, robust text classification for low-resource languages, such as Kazakh, continues to pose a significant challenge. This paper presents a comparative analysis of the classical Word2Vec embedding, BiLSTM recurrent models, BERT and XLM-R transformers, the generative mT5 model, alongside classical and statistical baselines for few-shot Kazakh news text classification under limited annotation constraints. Experiments were conducted on the KazNews dataset, curated for few-shot classification from tengrinews.kz and inform.kz articles, as well as parallel HuffPost corpora. The datasets comprise five categories (sports, politics, business, travel, etc.). The experimental findings in 1-shot and 5-shot settings demonstrate that no single model consistently dominates across all datasets and regimes: Word2Vec achieved the best performance on the HuffPost corpora in the 5-shot regime (0.48 and 0.53); mT5-Seq2Seq outperformed others on KazNews under 5-shot (0.71); while BERT remained competitive on the long-text KazNews dataset in the 5-shot setup. The authors also observed a sharp performance gain for the mT5-Seq2Seq model in the 5-shot setup. Conversely, none of the evaluated models yielded acceptable quality metrics in the 1-shot regime.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>обработка естественного языка</kwd><kwd>низкоресурсные языки</kwd><kwd>классификация текстов</kwd><kwd>fewshot learning</kwd><kwd>рекуррентные модели</kwd><kwd>трансформерные модели</kwd><kwd>генеративные модели</kwd></kwd-group><kwd-group xml:lang="en"><kwd>few-shot learning</kwd><kwd>text classification</kwd><kwd>Kazakh language</kwd><kwd>low-resource languages</kwd><kwd>transformer models</kwd><kwd>BERT</kwd><kwd>XLM-R</kwd><kwd>BiLSTM</kwd><kwd>Word2Vec</kwd><kwd>deep learning</kwd><kwd>natural language processing</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Pakray, P., Gelbukh, A., &amp; Bandyopadhyay, S. (2025). Natural language processing applications for low-resource languages. Natural Language Processing, 31(2), 183–197. https://doi.org/10.1017/nlp.2024.33</mixed-citation><mixed-citation xml:lang="en">Pakray, P., Gelbukh, A., &amp; Bandyopadhyay, S. (2025). Natural language processing applications for low-resource languages. Natural Language Processing, 31(2), 183–197. https://doi.org/10.1017/nlp.2024.33</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Kowsari, K., Jafari Meimandi, K., Heidarysafa, M., Mendu, S., Barnes, L., &amp; Brown, D. (2019). Text classification algorithms: A survey. Information, 10(4), 150. https://doi.org/10.3390/info10040150</mixed-citation><mixed-citation xml:lang="en">Kowsari, K., Jafari Meimandi, K., Heidarysafa, M., Mendu, S., Barnes, L., &amp; Brown, D. (2019). Text classification algorithms: A survey. Information, 10(4), 150. https://doi.org/10.3390/info10040150</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">McCallum, A., &amp; Nigam, K. (1998). A comparison of event models for naive Bayes text classification. AAAI-98 Workshop on Learning for Text Categorization. https://aaai.org/papers/041-ws98-05-007/</mixed-citation><mixed-citation xml:lang="en">McCallum, A., &amp; Nigam, K. (1998). A comparison of event models for naive Bayes text classification. AAAI-98 Workshop on Learning for Text Categorization. https://aaai.org/papers/041-ws98-05-007/</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Joachims, T. (1998). Text categorization with support vector machines: Learning with many relevant features. In C. Nédellec &amp; C. Rouveirol (Eds.), Machine Learning: ECML-98 (pp. 137–142). Springer. https://doi.org/10.1007/BFb0026683</mixed-citation><mixed-citation xml:lang="en">Joachims, T. (1998). Text categorization with support vector machines: Learning with many relevant features. In C. Nédellec &amp; C. Rouveirol (Eds.), Machine Learning: ECML-98 (pp. 137–142). Springer. https://doi.org/10.1007/BFb0026683</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Cox, D. R. (1958). The regression analysis of binary sequences. Journal of the Royal Statistical Society: Series B (Methodological), 20(2), 215–242.</mixed-citation><mixed-citation xml:lang="en">Cox, D. R. (1958). The regression analysis of binary sequences. Journal of the Royal Statistical Society: Series B (Methodological), 20(2), 215–242.</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Salton, G., Wong, A., &amp; Yang, C.S. (1975). A vector space model for automatic indexing. Communications of the ACM, 18(11), 613–620. https://doi.org/10.1145/361219.361220</mixed-citation><mixed-citation xml:lang="en">Salton, G., Wong, A., &amp; Yang, C.S. (1975). A vector space model for automatic indexing. Communications of the ACM, 18(11), 613–620. https://doi.org/10.1145/361219.361220</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Spärck Jones, K. (2004). A statistical interpretation of term specificity in retrieval. Journal of Documentation, 60(5), 493–502. https://doi.org/10.1108/00220410410560573</mixed-citation><mixed-citation xml:lang="en">Spärck Jones, K. (2004). A statistical interpretation of term specificity in retrieval. Journal of Documentation, 60(5), 493–502. https://doi.org/10.1108/00220410410560573</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Hochreiter, S., &amp; Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735– 1780. https://doi.org/10.1162/neco.1997.9.8.1735</mixed-citation><mixed-citation xml:lang="en">Hochreiter, S., &amp; Schmidhuber, J. (1997). Long short-term memory. Neural Computation, 9(8), 1735– 1780. https://doi.org/10.1162/neco.1997.9.8.1735</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Graves, A., &amp; Schmidhuber, J. (2005). Framewise phoneme classification with bidirectional LSTM and other neural network architectures. Neural Networks, 18(5), 602–610. https://doi.org/10.1016/j.neunet.2005.06.042</mixed-citation><mixed-citation xml:lang="en">Graves, A., &amp; Schmidhuber, J. (2005). Framewise phoneme classification with bidirectional LSTM and other neural network architectures. Neural Networks, 18(5), 602–610. https://doi.org/10.1016/j.neunet.2005.06.042</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Joulin, A., Grave, E., Bojanowski, P., &amp; Mikolov, T. (2016). Bag of tricks for efficient text classification. arXiv. https://doi.org/10.48550/arXiv.1607.01759</mixed-citation><mixed-citation xml:lang="en">Joulin, A., Grave, E., Bojanowski, P., &amp; Mikolov, T. (2016). Bag of tricks for efficient text classification. arXiv. https://doi.org/10.48550/arXiv.1607.01759</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Mikolov, T., Chen, K., Corrado, G., &amp; Dean, J. (2013). Efficient estimation of word representations in vector space. arXiv. https://doi.org/10.48550/arXiv.1301.3781</mixed-citation><mixed-citation xml:lang="en">Mikolov, T., Chen, K., Corrado, G., &amp; Dean, J. (2013). Efficient estimation of word representations in vector space. arXiv. https://doi.org/10.48550/arXiv.1301.3781</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Devlin, J., Chang, M.-W., Lee, K., &amp; Toutanova, K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Vol. 1, pp. 4171– 4186). Association for Computational Linguistics. https://doi.org/10.18653/v1/N19-1423</mixed-citation><mixed-citation xml:lang="en">Devlin, J., Chang, M.-W., Lee, K., &amp; Toutanova, K. (2019). BERT: Pre-training of deep bidirectional transformers for language understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (Vol. 1, pp. 4171– 4186). Association for Computational Linguistics. https://doi.org/10.18653/v1/N19-1423</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Conneau, A., Khandelwal, K., Goyal, N., Chaudhary, V., Wenzek, G., Guzmán, F., ... &amp; Zettlemoyer, L. (2019). Unsupervised cross-lingual representation learning at scale. arXiv. https://arxiv.org/abs/1911.02116</mixed-citation><mixed-citation xml:lang="en">Conneau, A., Khandelwal, K., Goyal, N., Chaudhary, V., Wenzek, G., Guzmán, F., ... &amp; Zettlemoyer, L. (2019). Unsupervised cross-lingual representation learning at scale. arXiv. https://arxiv.org/abs/1911.02116</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">Xue, L., Constant, N., Roberts, A., Kale, K., Al-Rfou, R., Siddhant, A., ... &amp; Raffel, C. (2021). mT5: A massively multilingual pre-trained text-to-text transformer. In Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (pp. 483–498). https://doi.org/10.18653/v1/2021.naacl-main.41</mixed-citation><mixed-citation xml:lang="en">Xue, L., Constant, N., Roberts, A., Kale, K., Al-Rfou, R., Siddhant, A., ... &amp; Raffel, C. (2021). mT5: A massively multilingual pre-trained text-to-text transformer. In Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (pp. 483–498). https://doi.org/10.18653/v1/2021.naacl-main.41</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Latief, A. D., Jarin, A., Yuyun, Hidayati, N. N., Afra, D. I. N., &amp; Riza, H. (2025). A systematic review of few-shot and zero-shot learning for NLP in low-resource languages: Insights and challenges. IEEE Xplore. https://ieeexplore.ieee.org/abstract/document/11325582</mixed-citation><mixed-citation xml:lang="en">Latief, A. D., Jarin, A., Yuyun, Hidayati, N. N., Afra, D. I. N., &amp; Riza, H. (2025). A systematic review of few-shot and zero-shot learning for NLP in low-resource languages: Insights and challenges. IEEE Xplore. https://ieeexplore.ieee.org/abstract/document/11325582</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Latief, A. D., Jarin, A., Yuyun, Hidayati, N. N., Afra, D. I. N., &amp; Riza, H. (2025). A systematic review of few-shot and zero-shot learning for NLP in low-resource languages: Insights and challenges. In 2025 International Conference on Computer, Control, Informatics and its Applications (IC3INA) (pp. 358–363). IEEE. https://doi.org/10.1109/IC3INA68387.2025.11325582</mixed-citation><mixed-citation xml:lang="en">Latief, A. D., Jarin, A., Yuyun, Hidayati, N. N., Afra, D. I. N., &amp; Riza, H. (2025). A systematic review of few-shot and zero-shot learning for NLP in low-resource languages: Insights and challenges. In 2025 International Conference on Computer, Control, Informatics and its Applications (IC3INA) (pp. 358–363). IEEE. https://doi.org/10.1109/IC3INA68387.2025.11325582</mixed-citation></citation-alternatives></ref><ref id="cit17"><label>17</label><citation-alternatives><mixed-citation xml:lang="ru">Toleu, A., Tolegen, G., &amp; Ualiyeva, I. (2025). Fine-Tuning Large Language Models for Kazakh Text Simplification. Applied Sciences, 15(15), 8344. https://doi.org/10.3390/app15158344</mixed-citation><mixed-citation xml:lang="en">Toleu, A., Tolegen, G., &amp; Ualiyeva, I. (2025). Fine-Tuning Large Language Models for Kazakh Text Simplification. Applied Sciences, 15(15), 8344. https://doi.org/10.3390/app15158344</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
