<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">kaz44</journal-id><journal-title-group><journal-title xml:lang="ru">Вестник Университета Шакарима. Серия технические науки</journal-title><trans-title-group xml:lang="en"><trans-title>Bulletin of Shakarim University. Technical Sciences</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">2788-7995</issn><issn pub-type="epub">3006-0524</issn><publisher><publisher-name>«Шәкәрім университеті» КеАҚ</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.53360/2788-7995-2026-1(21)-14</article-id><article-id custom-type="elpub" pub-id-type="custom">kaz44-2398</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>АВТОМАТИЗАЦИЯ И ИНФОРМАЦИОННЫЕ ТЕХНОЛОГИИ (ОРИГИНАЛЬНАЯ СТАТЬЯ)</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>AUTOMATION AND INFORMATION TECHNOLOGY (ORIGINAL ARTICLE)</subject></subj-group></article-categories><title-group><article-title>ПОСТРОЕНИЕ И ЛИНГВИСТИЧЕСКИЙ АНАЛИЗ ВОПРОСНО-ОТВЕТНОГО КОРПУСА KAZAKH HISTORYQA</article-title><trans-title-group xml:lang="en"><trans-title>CONSTRUCTION AND LINGUISTIC ANALYSIS OF THE QUESTION-ANSWER CORPUS OF KAZAKH HISTORYQA</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-2180-293X</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ергеш</surname><given-names>М. Ж.</given-names></name><name name-style="western" xml:lang="en"><surname>Yergesh</surname><given-names>M. Zh.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Манас Жантуғанұлы Ергеш – докторант кафедры «Технологий искусственного интеллекта» </p><p>010000, Астана, ул. Сатпаева, 2</p></bio><bio xml:lang="en"><p>Manas Yergesh – PhD doctoral student, Department of Artificial Intelligence Technologies </p><p>010000, Astana, 2 Satpaev Street</p></bio><email xlink:type="simple">shymkent90@mail.ru</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-8967-2625</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ергеш</surname><given-names>Б. Ж.</given-names></name><name name-style="western" xml:lang="en"><surname>Yergesh</surname><given-names>B. Zh.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Бану Жантуғанқызы Ергеш – PhD, заместитель директора департамента цифрового развития </p><p>010000, Астана, ул. Сатпаева, 2</p></bio><bio xml:lang="en"><p>Banu Yergesh – PhD, deputy director of the Department of Digital Development </p><p>010000, Astana, 2 Satpaev Street</p></bio><email xlink:type="simple">b.yergesh@gmail.com</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-2899-9886</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Оразаева</surname><given-names>А. Р.</given-names></name><name name-style="western" xml:lang="en"><surname>Orazayeva</surname><given-names>A.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Айнур Ришатовна Оразаева – PhD, ассистент-профессор кафедры «Информационные технологии» </p><p>010000, Астана, ул. Мухамедханова, 37А</p></bio><bio xml:lang="en"><p>Ainur Orazayeva – K. Kulazhanov Kazakh University of Technology and Business, Department of Information Technology, Head of the Department </p><p>010000, Astana, 37 A Mukhamedhanov Street</p></bio><email xlink:type="simple">oar_is@mail.com</email><xref ref-type="aff" rid="aff-2"/></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-5800-9134</contrib-id><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Талгат</surname><given-names>А.</given-names></name><name name-style="western" xml:lang="en"><surname>Talgat</surname><given-names>А.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Амангуль Талгат – сеньор-лектор кафедры «Информационные технологии» </p><p>010000, Астана, ул. Мухамедханова, 37А</p></bio><bio xml:lang="en"><p>Amangul Talgat – K. Kulazhanov Kazakh University of Technology and Business, Department of Information Technology, Head of the Department </p><p>010000, Astana, 37 A Mukhamedhanov Street</p></bio><email xlink:type="simple">amangul_talgat81@mail.ru</email><xref ref-type="aff" rid="aff-2"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Евразийский национальный университет им. Л.Н. Гумилёва</institution><country>Казахстан</country></aff><aff xml:lang="en"><institution>L.N. Gumilyov Eurasian National University</institution><country>Kazakhstan</country></aff></aff-alternatives><aff-alternatives id="aff-2"><aff xml:lang="ru"><institution>АО «Казахский университет технологии и бизнеса им. К. Кулажанова»</institution><country>Казахстан</country></aff><aff xml:lang="en"><institution>Kazakh University of Technology and Business named after K. Kulazhanov</institution><country>Kazakhstan</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>25</day><month>05</month><year>2026</year></pub-date><volume>1</volume><issue>1(21)</issue><fpage>126</fpage><lpage>135</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Ергеш М.Ж., Ергеш Б.Ж., Оразаева А.Р., Талгат А., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Ергеш М.Ж., Ергеш Б.Ж., Оразаева А.Р., Талгат А.</copyright-holder><copyright-holder xml:lang="en">Yergesh M.Z., Yergesh B.Z., Orazayeva A., Talgat А.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://tech.vestnik.shakarim.kz/jour/article/view/2398">https://tech.vestnik.shakarim.kz/jour/article/view/2398</self-uri><abstract><p>Цифровизация среднего образования в Казахстане усиливает потребность в специализированных казахскоязычных ресурсах, поддерживающих интеллектуальные вопросноответные системы по школьным предметам. Однако существующие QA-датасеты в основном ориентированы на высокоресурсные языки и не охватывают содержание учебных дисциплин, в частности истории Казахстана. В данной работе представлен и подробно проанализирован корпус Kazakh HistoryQA – первый специализированный вопросно-ответный ресурс на казахском языке, созданный на основе шести официальных учебников по истории Казахстана для 5-10 классов. Корпус включает 208 тематических секций, около 180 тысяч токенов и 661 QA-пару с привязкой к точным координатам ответного спана, что обеспечивает совместимость со стандартными пайплайнами обучения моделей машинного чтения, включая библиотеку HuggingFace. Проведён многоуровневый анализ корпуса: статистика длины контекстов, вопросов и ответов, распределение материалов по классам и учебникам, типологизация вопросов на фактологические, хронологические, процедурные, причинно-следственные и сравнительные. Дополнительно выполнено морфологическое профилирование текста, построен частотный словарь и выделены предметные термины. Для оценки надежности корпуса как бенчмарка реализованы базовые модели выбора предложений (случайный выбор, TF-IDF, BM25), среди которых BM25 показал лучшие результаты. Представленный корпус создаёт основу для последующих исследований и разработки гибридных онтологически обогащённых QA-систем в сфере образования.</p></abstract><trans-abstract xml:lang="en"><p>The digitalization of secondary education in Kazakhstan increases the need for specialized Kazakhlanguage resources that support intelligent question-and-answer systems for school subjects. However, existing QA datasets are mainly focused on high-resource languages and do not cover the content of academic disciplines, particularly the history of Kazakhstan. This paper presents and analyzes in detail the Kazakh HistoryQA corpus, the first specialized question-and-answer resource in the Kazakh language, which is based on six official textbooks on the history of Kazakhstan for grades 5-10. The corpus includes 208 thematic sections, about 180 thousand tokens, and 661 QA-pairs with exact coordinates of the response span, which ensures compatibility with standard machine reading model training pipelines, including the HuggingFace library. A multi-level analysis of the corpus has been conducted: statistics on the length of contexts, questions, and answers, distribution of materials by classes and textbooks, and typology.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>Казахстанская история</kwd><kwd>OCR</kwd><kwd>QA</kwd><kwd>онтология</kwd><kwd>образовательная NLP</kwd><kwd>RAG</kwd><kwd>тезаурус</kwd><kwd>онтология</kwd></kwd-group><kwd-group xml:lang="en"><kwd>History of Kazakhstan</kwd><kwd>corpus</kwd><kwd>OCR</kwd><kwd>retrieval</kwd><kwd>QA</kwd><kwd>ontology</kwd><kwd>educational NLP</kwd><kwd>RAG</kwd><kwd>question-answer</kwd><kwd>thesaurus</kwd><kwd>ontology</kwd><kwd>evaluation</kwd></kwd-group><funding-group><funding-statement xml:lang="ru">Данное исследование финансируется Комитетом науки Министерства науки и высшего образования Республики Казахстан (грант № AP22787194).</funding-statement></funding-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Vartiainen T. Engaging with bad (meta) data in historical corpus linguistics / T. Vartiainen, T. Säily // Challenges in Corpus Linguistics. – John Benjamins Publishing Company, 2024. – P. 9-34.</mixed-citation><mixed-citation xml:lang="en">Vartiainen T. Engaging with bad (meta) data in historical corpus linguistics / T. Vartiainen, T. Säily // Challenges in Corpus Linguistics. – John Benjamins Publishing Company, 2024. – P. 9-34.</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Durrell M. ‘Representativeness’, ‘Bad Data’, and legitimate expectations. What can an electronic historical corpus tell us that we didn’t actually know already (and how)? / M. Durrell // Historical Corpora. Challenges and Perspectives. – Narr, 2024. – P. 13-33.</mixed-citation><mixed-citation xml:lang="en">Durrell M. ‘Representativeness’, ‘Bad Data’, and legitimate expectations. What can an electronic historical corpus tell us that we didn’t actually know already (and how)? / M. Durrell // Historical Corpora. Challenges and Perspectives. – Narr, 2024. – P. 13-33.</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Historical insights at scale: A corpus-wide machine learning analysis of early modern astronomic tables / O. Eberle et al // Science Advances. – 2024. – Vol. 10, № 43. – P. eadj1719.</mixed-citation><mixed-citation xml:lang="en">Historical insights at scale: A corpus-wide machine learning analysis of early modern astronomic tables / O. Eberle et al // Science Advances. – 2024. – Vol. 10, № 43. – P. eadj1719.</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Václav C. When Data Meet Tools: Using the Monitor Corpus for the Analysis of Language Development / C. Václav, S. Martin, P. Klára // Jazykovedný Časopis. – 2025. – Vol. 76, № 1. – P.</mixed-citation><mixed-citation xml:lang="en">Václav C. When Data Meet Tools: Using the Monitor Corpus for the Analysis of Language Development / C. Václav, S. Martin, P. Klára // Jazykovedný Časopis. – 2025. – Vol. 76, № 1. – P.</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Gorozhanov A.I. Natural Language Processing and Fiction Text: Basis for Corpus Research / A.I. Gorozhanov, I.A. Guseinova, D.V. Stepanova // Vestnik Rossiiskogo universiteta druzhby narodov. Seriya: Teoriya yazyka. Semiotika. Semantika. – 2024. – Vol. 15, № 1. – P. 195-210.</mixed-citation><mixed-citation xml:lang="en">Gorozhanov A.I. Natural Language Processing and Fiction Text: Basis for Corpus Research / A.I. Gorozhanov, I.A. Guseinova, D.V. Stepanova // Vestnik Rossiiskogo universiteta druzhby narodov. Seriya: Teoriya yazyka. Semiotika. Semantika. – 2024. – Vol. 15, № 1. – P. 195-210.</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">TIGQA: An Expert Annotated Question Answering Dataset in Tigrinya / H. Teklehaymanot et al // arXiv preprint arXiv:2404. – 17194. – 2024.</mixed-citation><mixed-citation xml:lang="en">TIGQA: An Expert Annotated Question Answering Dataset in Tigrinya / H. Teklehaymanot et al // arXiv preprint arXiv:2404. – 17194. – 2024.</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Nyarko G.S. Leaderships' Role in Quality Assurance Moderating Curriculum Implementation: Perspectives and Preferences of Headteachers and Teachers in Basic Schools in Ghana / G.S. Nyarko // Education Quarterly Reviews. – 2025. – Vol. 8, № 3.</mixed-citation><mixed-citation xml:lang="en">Nyarko G.S. Leaderships' Role in Quality Assurance Moderating Curriculum Implementation: Perspectives and Preferences of Headteachers and Teachers in Basic Schools in Ghana / G.S. Nyarko // Education Quarterly Reviews. – 2025. – Vol. 8, № 3.</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Veitsman Y. Recent advancements and challenges of Turkic Central Asian language processing / Y. Veitsman, M. Hartmann // Proceedings of the First Workshop on Language Models for Low-Resource Languages. – 2025. – P. 309-324.</mixed-citation><mixed-citation xml:lang="en">Veitsman Y. Recent advancements and challenges of Turkic Central Asian language processing / Y. Veitsman, M. Hartmann // Proceedings of the First Workshop on Language Models for LowResource Languages. – 2025. – P. 309-324.</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Biyik H. Turkish Delights: A Dataset on Turkish Euphemisms / H. Biyik, P. Lee, A. Feldman // Proceedings of the First Workshop on Natural Language Processing for Turkic Languages (SIGTURK 2024). – 2024. – P. 71-80.</mixed-citation><mixed-citation xml:lang="en">Biyik H. Turkish Delights: A Dataset on Turkish Euphemisms / H. Biyik, P. Lee, A. Feldman // Proceedings of the First Workshop on Natural Language Processing for Turkic Languages (SIGTURK 2024). – 2024. – P. 71-80.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Kardeş-NLU: Transfer to Low-Resource Languages with the Help of a High-Resource Cousin – A Benchmark and Evaluation for Turkic Languages / L.K. Senel et al // Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics. – 2024. – Vol. 1. – P. 1672-1688.</mixed-citation><mixed-citation xml:lang="en">Kardeş-NLU: Transfer to Low-Resource Languages with the Help of a High-Resource Cousin – A Benchmark and Evaluation for Turkic Languages / L.K. Senel et al // Proceedings of the 18th Conference of the European Chapter of the Association for Computational Linguistics. – 2024. – Vol. 1. – P. 1672-1688.</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Tutar K. Turkish Question-Answer Dataset Evaluated with Deep Learning / K. Tutar, O.T. Yildiz // 2024 9th International Conference on Computer Science and Engineering (UBMK). – IEEE, 2024. – P. 101-105.</mixed-citation><mixed-citation xml:lang="en">Tutar K. Turkish Question-Answer Dataset Evaluated with Deep Learning / K. Tutar, O.T. Yildiz // 2024 9th International Conference on Computer Science and Engineering (UBMK). – IEEE, 2024. – P. 101-105.</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Setting Standards in Turkish NLP: TR-MMLU for Large Language Model Evaluation / M.A. Bayram et al // arXiv preprint arXiv:2501.00593. – 2024.</mixed-citation><mixed-citation xml:lang="en">Setting Standards in Turkish NLP: TR-MMLU for Large Language Model Evaluation / M.A. Bayram et al // arXiv preprint arXiv:2501.00593. – 2024.</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Kaya Y.B. Effect of Tokenization Granularity for Turkish Large Language Models / Y.B. Kaya, A.C. Tantung // Intelligent Systems with Applications. – 2024. – Vol. 21. – Art. 200335.</mixed-citation><mixed-citation xml:lang="en">Kaya Y.B. Effect of Tokenization Granularity for Turkish Large Language Models / Y.B. Kaya, A.C. Tantung // Intelligent Systems with Applications. – 2024. – Vol. 21. – Art. 200335.</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">Development and Evaluation of a Small Kazakh Language Corpus to Improve the Efficiency of Multilingual NLP Systems in Low-Resource Environments / A. Tleubayeva et al // 2025 IEEE 5th International Conference on Smart Information Systems and Technologies (SIST). – IEEE, 2025. – P. 1-6.</mixed-citation><mixed-citation xml:lang="en">Development and Evaluation of a Small Kazakh Language Corpus to Improve the Efficiency of Multilingual NLP Systems in Low-Resource Environments / A. Tleubayeva et al // 2025 IEEE 5th International Conference on Smart Information Systems and Technologies (SIST). – IEEE, 2025. – P. 1-6.</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Aitim A. A Comparison of Kazakh Language Processing Models for Improving Semantic Search Results / A. Aitim, R. Satybaldiyeva // Eastern-European Journal of Enterprise Technologies. – 2025. – Vol. 133, № 2.</mixed-citation><mixed-citation xml:lang="en">Aitim A. A Comparison of Kazakh Language Processing Models for Improving Semantic Search Results / A. Aitim, R. Satybaldiyeva // Eastern-European Journal of Enterprise Technologies. – 2025. – Vol. 133, № 2.</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Machine Learning Methods for Kazakh Morphology: A Comprehensive Overview / I. Akhmetov et al // 2024 IEEE 3rd International Conference on Problems of Informatics, Electronics and Radio Engineering (PIERE). – IEEE, 2024. – P. 1880-1884.</mixed-citation><mixed-citation xml:lang="en">Machine Learning Methods for Kazakh Morphology: A Comprehensive Overview / I. Akhmetov et al // 2024 IEEE 3rd International Conference on Problems of Informatics, Electronics and Radio Engineering (PIERE). – IEEE, 2024. – P. 1880-1884.</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
