<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">inttrans</journal-id><journal-title-group><journal-title xml:lang="ru">Интеллектуальный транспорт</journal-title><trans-title-group xml:lang="en"><trans-title>Intelligent transport</trans-title></trans-title-group></journal-title-group><issn pub-type="epub">3033-6007</issn><publisher><publisher-name>АО «НИИАС»</publisher-name></publisher></journal-meta><article-meta><article-id custom-type="elpub" pub-id-type="custom">inttrans-24</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>ИСКУССТВЕННЫЙ ИНТЕЛЛЕКТ И МАШИННОЕ ОБУЧЕНИЕ</subject></subj-group></article-categories><title-group><article-title>OMLS-Bench: многоуровневый бенчмарк LLM для программной инженерии</article-title><trans-title-group xml:lang="en"><trans-title>OMLS-Bench: a multi-level benchmark for engineering LLMS</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ярушев</surname><given-names>С. А.</given-names></name><name name-style="western" xml:lang="en"><surname>Yarushev</surname><given-names>S. A.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Ярушев Сергей Александрович, к.т.н., директор научного центра</p><p>Москва</p></bio><bio xml:lang="en"><p>Sergey A. Yarushev, Ph.D. in Engineering, Director of the Research Center</p><p>Moscow</p></bio><email xlink:type="simple">Yarushev.SA@rea.ru</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Ануров</surname><given-names>А. О.</given-names></name><name name-style="western" xml:lang="en"><surname>Anurov</surname><given-names>A. O.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Ануров Александр Олегович, лаборант-исследователь</p><p>Москва</p></bio><bio xml:lang="en"><p>Alexandr O. Anurov, research assistant</p><p>Moscow</p></bio><email xlink:type="simple">Anurov.AO@rea.ru</email><xref ref-type="aff" rid="aff-2"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Булгаков</surname><given-names>Г. Г.</given-names></name><name name-style="western" xml:lang="en"><surname>Bulgakov</surname><given-names>G. G.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Булгаков Геннадий Геннадьевич, аспирант</p><p>Москва</p></bio><bio xml:lang="en"><p>Gennadii G. Bulgakov, postgraduate student</p><p>Moscow</p></bio><email xlink:type="simple">g.bulgak0v@list.ru</email><xref ref-type="aff" rid="aff-2"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Российский Экономический Университет им. Г.В. Плеханова</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Federal Research Center «Computer Science and Control» of the Russian Academy of Sciences</institution><country>Russian Federation</country></aff></aff-alternatives><aff-alternatives id="aff-2"><aff xml:lang="ru"><institution>Российский Экономический Университет им. Г.В. Плеханова</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Plekhanov Russian University of Economics</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2025</year></pub-date><pub-date pub-type="epub"><day>11</day><month>09</month><year>2026</year></pub-date><volume>0</volume><issue>3(35)</issue><fpage>96</fpage><lpage>112</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Ярушев С.А., Ануров А.О., Булгаков Г.Г., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Ярушев С.А., Ануров А.О., Булгаков Г.Г.</copyright-holder><copyright-holder xml:lang="en">Yarushev S.A., Anurov A.O., Bulgakov G.G.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://www.intelligent-transport.ru/jour/article/view/24">https://www.intelligent-transport.ru/jour/article/view/24</self-uri><abstract><p>В данной работе представлена система оценки OMLS-Bench (Open Multi-Level Skills Benchmark) двухэтапный фреймворк для всесторонней оценки больших языковых моделей в задачах программной инженерии. Цель предложенного подхода преодолеть ограничения существующих методик, которые либо измеряют узкие подзадачи, либо применяют единый уровень сложности и не фиксируют динамику инженерного мастерства и интерактивную природу диагностического рассуждения. Описываемая система охватывает девять доменов (Back-end, Front-end, Mobile, DevOps, Data Analysis, Machine Learning, Big Data, IoT, встроенные системы) и пять уровней сложности, что позволяет стратифицировать качество по доменам и уровням. Предлагается двухэтапная процедура: Этап I стандартизированные задания с множественным выбором и строгим форматом ответа , Этап II сценарные задания с пошаговыми контрольными списками и независимой моделью-судьёй. Формализованы метрики Tier Accuracy и Domain Accuracy, введён интегральный показатель OPS; раскрыты переменные формулы (1). Публикуются артефакты для воспроизводимости: JSON-схемы заданий, русскоязычные шаблоны промптов и скрипт оценки eval_mc.py с описанием входных/выходных параметров. Эксперименты показывают: неоднородность качества между доменами; снижение результатов при переходе от тестов с фиксированными вариантами к сценарным задачам; детальные диагностические отчёты по невыполненным пунктам контрольных списков. OMLS-Bench может служить практическим инструментом сравнения LLM в инженерных задачах и основой для целенаправленной донастройки моделей под конкретные области. Проведённая стартовая крупномасштабная оценка десяти современных больших языковых моделей выявила чёткую стратификацию результатов по уровням сложности и по сферам: модели большего масштаба демонстрировали высокую точность в широко представленных веб-ориентированных областях, тогда как специализированные направления (мобильная разработка, встроенные системы) показали существенно худшие показатели. Эти наблюдения подчёркивают важность как размера модели, так и разнообразия предметных данных при обучении. OMLS-Bench обеспечивает воспроизводимое и расширяемое средство оценки, которое может служить основой для разработки более надёжных и предметно ориентированных моделей-помощников инженера. В перспективе планируется развитие интерактивной фазы, повышение реалистичности сценариев и доработка контрольных чек-листов для приближения тестирования к профессиональной практике.</p></abstract><trans-abstract xml:lang="en"><p>This paper presents the OMLS-Benchmark (Open Multi-Level Skills Benchmark) assessment system, a two–stage framework for the comprehensive assessment of large language models in software engineering tasks. The aim of the proposed approach is to overcome the limitations of existing techniques that either measure narrow subtasks or apply a single level of complexity and do not capture the dynamics of engineering skill and the interactive nature of diagnostic reasoning. The described system covers nine domains (Back-end, Front-end, Mobile, DevOps, Data Analysis, Machine Learning, Big Data, IoT, embedded systems) and five levels of complexity, which allows you to stratify quality by domains and levels. A two–stage procedure is proposed: Stage I standardized tasks with multiple choice and a strict response format, Stage II scenario tasks with step-by-step checklists and an independent judge model. The Tier Accuracy and Domain Accuracy metrics have been formalized, the OPS integral indicator has been introduced; the variables of formula (1) have been disclosed. Artifacts are published for reproducibility: JSON schemas of tasks, Russianlanguage templates of projects and an evaluation script. eval_mc.py with a description of the input/output parameters. Experiments show: heterogeneity of quality between domains; decreased results when switching from tests with fixed options to scenario tasks; detailed diagnostic reports on outstanding checklist items. The OMLS-Bench can serve as a practical tool for comparing LLMs in engineering tasks and as a basis for purposefully fine-tuning models to specific areas. The initial large-scale assessment of ten modern large language models revealed a clear stratification of results by complexity levels and by area: larger-scale models demonstrated high accuracy in widely represented web-oriented areas, while specialized areas (mobile development, embedded systems) showed significantly worse performance. These observations highlight the importance of both the size of the model and the variety of subject data in training. OMLS-Bench provides a reproducible and extensible evaluation tool that can serve as a basis for the development of more reliable and domain-specific assistant engineer models. In the future, it is planned to develop the interactive phase, increase the realism of scenarios and finalize control checklists to bring testing closer to professional practice.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>Оценка больших языковых моделей</kwd><kwd>Бенчмарк для программной инженерии</kwd><kwd>Многоуровневая оценка</kwd><kwd>Тестирование с множественным выбором</kwd><kwd>Диагностика с участием LLM (LLM-in-the-loop)</kwd><kwd>Итоговый балл производительности (Overall Performance Score</kwd><kwd>OPS)</kwd><kwd>Интерактивные диагностические протоколы</kwd></kwd-group><kwd-group xml:lang="en"><kwd>Large Language Model Evaluation</kwd><kwd>Software Engineering Benchmark</kwd><kwd>Multi-Tier Assessment</kwd><kwd>MultipleChoice Testing</kwd><kwd>LLM-in-the-Loop Diagnostics</kwd><kwd>Overall Performance Score (OPS)</kwd><kwd>Interactive Diagnostic Protocols</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Ануров А. О., Булгаков Г. Г., Ярушев С. А. OMLS-Bench [Электронный ресурс]. – URL: https://github.com/Blgkff/OMLS-BENCH (дата обращения: 14.06.2025).</mixed-citation><mixed-citation xml:lang="en">Ануров А. О., Булгаков Г. Г., Ярушев С. А. OMLS-Bench [Электронный ресурс]. – URL: https://github.com/Blgkff/OMLS-BENCH (дата обращения: 14.06.2025).</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">D. Hendrycks et al., “Measuring Massive Multitask Language Understanding,” in Proc. Int. Conf. Learn. Represent. (ICLR), 2021. [Online]. Available: https://arxiv.org/abs/2009.03300</mixed-citation><mixed-citation xml:lang="en">D. Hendrycks et al., “Measuring Massive Multitask Language Understanding,” in Proc. Int. Conf. Learn. Represent. (ICLR), 2021. [Online]. Available: https://arxiv.org/abs/2009.03300</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">T. B. Brown et al., “Language Models are Few-Shot Learners,” Adv. Neural Inf. Process. Syst. (NeurIPS), vol. 33, pp. 1877–1901, 2020. [Online]. Available: https://arxiv.org/abs/2005.14165</mixed-citation><mixed-citation xml:lang="en">T. B. Brown et al., “Language Models are Few-Shot Learners,” Adv. Neural Inf. Process. Syst. (NeurIPS), vol. 33, pp. 1877–1901, 2020. [Online]. Available: https://arxiv.org/abs/2005.14165</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">A. Wang et al., “GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding,” in *Proc. 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Jt. Conf. Nat. Lang. Process. (EMNLP-IJCNLP)*, pp. 353–361, 2019. [Online]. Available: https://arxiv.org/abs/1804.07461</mixed-citation><mixed-citation xml:lang="en">A. Wang et al., “GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding,” in *Proc. 2019 Conf. Empir. Methods Nat. Lang. Process. 9th Int. Jt. Conf. Nat. Lang. Process. (EMNLP-IJCNLP)*, pp. 353–361, 2019. [Online]. Available: https://arxiv.org/abs/1804.07461</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">A. Wang et al., “SuperGLUE: A Stickier Benchmark for General-Purpose Language Understanding Systems,” in Proc. 3rd Workshop Eval. Compar. NLP Syst., 2020. [Online]. Available: https://arxiv.org/abs/1905.00537</mixed-citation><mixed-citation xml:lang="en">A. Wang et al., “SuperGLUE: A Stickier Benchmark for General-Purpose Language Understanding Systems,” in Proc. 3rd Workshop Eval. Compar. NLP Syst., 2020. [Online]. Available: https://arxiv.org/abs/1905.00537</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">S. Lu et al., “CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation,” in Proc. 2021 Conf. Empir. Methods Nat. Lang. Process. (EMNLP), 2021. [Online]. Available: https://arxiv.org/abs/2102.04664</mixed-citation><mixed-citation xml:lang="en">S. Lu et al., “CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation,” in Proc. 2021 Conf. Empir. Methods Nat. Lang. Process. (EMNLP), 2021. [Online]. Available: https://arxiv.org/abs/2102.04664</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">M. Chen et al., “Evaluating Large Language Models Trained on Code,” arXiv preprint arXiv:2107.03374, 2021. [Online]. Available: https://arxiv.org/abs/2107.03374</mixed-citation><mixed-citation xml:lang="en">M. Chen et al., “Evaluating Large Language Models Trained on Code,” arXiv preprint arXiv:2107.03374, 2021. [Online]. Available: https://arxiv.org/abs/2107.03374</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">M. Mitchell and D. C. Krakauer, “The Debate Over Understanding in AI’s Large Language Models,” arXiv preprint arXiv:2210.13966, 2022. [Online]. Available: https://arxiv.org/abs/2210.13966</mixed-citation><mixed-citation xml:lang="en">M. Mitchell and D. C. Krakauer, “The Debate Over Understanding in AI’s Large Language Models,” arXiv preprint arXiv:2210.13966, 2022. [Online]. Available: https://arxiv.org/abs/2210.13966</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">D. Hendrycks et al., “Aligning AI With Shared Human Values,” in Proc. Int. Conf. Learn. Represent. (ICLR), 2021. [Online]. Available: https://arxiv.org/abs/2008.02275</mixed-citation><mixed-citation xml:lang="en">D. Hendrycks et al., “Aligning AI With Shared Human Values,” in Proc. Int. Conf. Learn. Represent. (ICLR), 2021. [Online]. Available: https://arxiv.org/abs/2008.02275</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">M. Suzgun et al., “Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them,” in Find. Assoc. Comput. Linguist.: ACL, pp. 4670–4685, 2022. [Online]. Available: https://arxiv.org/abs/2210.09261</mixed-citation><mixed-citation xml:lang="en">M. Suzgun et al., “Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them,” in Find. Assoc. Comput. Linguist.: ACL, pp. 4670–4685, 2022. [Online]. Available: https://arxiv.org/abs/2210.09261</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">M. Kazemi et al., “BIG-Bench Extra Hard,” arXiv preprint arXiv:2502.19187, 2025. [Online]. Available: https://arxiv.org/abs/2502.19187</mixed-citation><mixed-citation xml:lang="en">M. Kazemi et al., “BIG-Bench Extra Hard,” arXiv preprint arXiv:2502.19187, 2025. [Online]. Available: https://arxiv.org/abs/2502.19187</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">H. Li et al., “CMMLU: Measuring Massive Multitask Language Understanding in Chinese,” in Find. Assoc. Comput. Linguist.: ACL, pp. 6543–6558, 2024. [Online]. Available: https://aclanthology.org/2024.findings-acl.671</mixed-citation><mixed-citation xml:lang="en">H. Li et al., “CMMLU: Measuring Massive Multitask Language Understanding in Chinese,” in Find. Assoc. Comput. Linguist.: ACL, pp. 6543–6558, 2024. [Online]. Available: https://aclanthology.org/2024.findings-acl.671</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Y. Wang et al., “MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark,” arXiv preprint arXiv:2406.01574, 2024. [Online]. Available: https://arxiv.org/abs/2406.01574</mixed-citation><mixed-citation xml:lang="en">Y. Wang et al., “MMLU-Pro: A More Robust and Challenging Multi-Task Language Understanding Benchmark,” arXiv preprint arXiv:2406.01574, 2024. [Online]. Available: https://arxiv.org/abs/2406.01574</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">L. Austin et al., “Program Synthesis with Large Language Models,” in NeurIPS Workshop Mach. Learn. Syst., 2021. [Online]. Available: https://arxiv.org/abs/2108.07732</mixed-citation><mixed-citation xml:lang="en">L. Austin et al., “Program Synthesis with Large Language Models,” in NeurIPS Workshop Mach. Learn. Syst., 2021. [Online]. Available: https://arxiv.org/abs/2108.07732</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">K. Cobbe et al., “Training Verifiers to Solve Math Word Problems,” Adv. Neural Inf. Process. Syst. (NeurIPS), vol. 34, 2021. [Online]. Available: https://arxiv.org/abs/2110.14168</mixed-citation><mixed-citation xml:lang="en">K. Cobbe et al., “Training Verifiers to Solve Math Word Problems,” Adv. Neural Inf. Process. Syst. (NeurIPS), vol. 34, 2021. [Online]. Available: https://arxiv.org/abs/2110.14168</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">A. Radford et al., “Language Models are Unsupervised Multitask Learners,” OpenAI Tech. Rep., 2019. [Online]. Available: https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf</mixed-citation><mixed-citation xml:lang="en">A. Radford et al., “Language Models are Unsupervised Multitask Learners,” OpenAI Tech. Rep., 2019. [Online]. Available: https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
