<?xml version="1.0" encoding="UTF-8"?>
<?xml-stylesheet type="text/xsl" href="https://iereview.ru/lib/pkp/xml/oai2.xsl" ?>
<OAI-PMH xmlns="http://www.openarchives.org/OAI/2.0/"
	xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
	xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/
		http://www.openarchives.org/OAI/2.0/OAI-PMH.xsd">
	<responseDate>2026-08-23T22:40:18Z</responseDate>
	<request identifier="oai:ojs2.iereview.ru:article/235" metadataPrefix="jats" verb="GetRecord">https://iereview.ru/index.php/IE/oai</request>
	<GetRecord>
		<record>
			<header>
				<identifier>oai:ojs2.iereview.ru:article/235</identifier>
				<datestamp>2026-04-19T16:51:57Z</datestamp>
				<setSpec>IE:ARI</setSpec>
			</header>
			<metadata>
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns="" dtd-version="1.4" xsi:noNamespaceSchemaLocation="https://jats.nlm.nih.gov/archiving/1.4/xsd/JATS-archivearticle1.xsd" xml:lang="ru">
			<front>
			<journal-meta>
				<journal-id journal-id-type="publisher">IE</journal-id><journal-id journal-id-type="ojs">IE</journal-id>
				<journal-title-group>
			<journal-title xml:lang="ru">СТРОИТЕЛЬНЫЕ И ДОРОЖНЫЕ МАШИНЫ</journal-title><trans-title-group xml:lang="en"><trans-title>STROITEL'NYE I DOROZHNYE MASHINY</trans-title></trans-title-group>
</journal-title-group>			<issn pub-type="ppub">0039-2391</issn>			<publisher><publisher-name>ИП Подколзин М.М.</publisher-name></publisher>
			<self-uri xlink:href="https://iereview.ru/index.php/IE"/>
		</journal-meta>
		<article-meta>
			<article-id pub-id-type="publisher-id">235</article-id>
			<article-categories><subj-group subj-group-type="heading" xml:lang="en"><subject>APPLIED RESEARCH</subject></subj-group><subj-group subj-group-type="heading" xml:lang="ru"><subject>ПРИКЛАДНЫЕ ИССЛЕДОВАНИЯ</subject></subj-group></article-categories>
			<title-group><article-title xml:lang="ru">Разработка и исследование агента на основе алгоритма deep q-network для задач управления в динамических средах</article-title><trans-title-group xml:lang="en"><trans-title>Development and investigation of an agent based on Deep Q-Network algorithm for control tasks in dynamic environments</trans-title></trans-title-group></title-group>
			<contrib-group content-type="author">
				<contrib contrib-type="author">
					<name-alternatives>
						<name name-style="western" specific-use="primary" xml:lang="ru">
							<surname>Курташ</surname>
							<given-names>Никита Сергеевич</given-names>
						</name>
						<name name-style="western" xml:lang="en">
							<surname>Kurtash</surname>
							<given-names>Nikita S.</given-names>
						</name>
					</name-alternatives>
					<xref ref-type="aff" rid="aff-1"/>
					<email>nikitakurtash@gmail.com</email>
				</contrib>
				<contrib contrib-type="author">
					<name-alternatives>
						<name name-style="western" specific-use="primary" xml:lang="ru">
							<surname>Брютова</surname>
							<given-names>София Даниловна</given-names>
						</name>
						<name name-style="western" xml:lang="en">
							<surname>Bryutova</surname>
							<given-names>Sofia D.</given-names>
						</name>
					</name-alternatives>
					<xref ref-type="aff" rid="aff-1"/>
					<email>britovasofia@gmail.com</email>
				</contrib>
			</contrib-group>
			<aff-alternatives id="aff-1">
				<aff xml:lang="ru"><institution content-type="orgname">Санкт-Петербургский политехнический университет Петра Великого, 195251, г. Санкт-Петербург, Политехническая ул., д. 29</institution></aff>
				<aff xml:lang="en"><institution content-type="orgname">Peter the Great St. Petersburg Polytechnic University, 195251, St. Petersburg, Polytekhnicheskaya str., 29</institution></aff>
			</aff-alternatives>
			<pub-date date-type="collection"><year>2025</year></pub-date><pub-date date-type="pub" publication-format="epub">
				<day>30</day>
				<month>12</month>
				<year>2025</year>
			</pub-date>
			<volume seq="3">69</volume>
			<issue>12</issue>
				<issue-id>21</issue-id><issue-title xml:lang="ru">Строительные и дорожные машины</issue-title><issue-title xml:lang="en">Stroitel'nye i dorozhnye mashiny</issue-title><fpage>136</fpage>
				<lpage>144</lpage>
			<permissions>
				<copyright-statement xml:lang="ru">© 2025 СТРОИТЕЛЬНЫЕ И ДОРОЖНЫЕ МАШИНЫ. Все права защищены.</copyright-statement>
				<copyright-statement xml:lang="en">© 2025 STROITEL'NYE I DOROZHNYE MASHINY. All rights reserved.</copyright-statement>
				<copyright-year>2025</copyright-year>
				<copyright-holder xml:lang="ru">СТРОИТЕЛЬНЫЕ И ДОРОЖНЫЕ МАШИНЫ</copyright-holder>
				<copyright-holder xml:lang="en">STROITEL'NYE I DOROZHNYE MASHINY</copyright-holder>
				<license license-type="open-access" specific-use="metadata" xlink:href="https://creativecommons.org/publicdomain/zero/1.0/" xml:lang="ru">
					<license-p>Метаданные настоящей записи распространяются на условиях Creative Commons CC0 1.0 (передача в общественное достояние).</license-p>
				</license>
				<license license-type="open-access" specific-use="metadata" xlink:href="https://creativecommons.org/publicdomain/zero/1.0/" xml:lang="en">
					<license-p>The metadata of this record are distributed under the Creative Commons CC0 1.0 Universal Public Domain Dedication.</license-p>
				</license>
			</permissions>
			
			<self-uri xlink:href="https://iereview.ru/index.php/IE/article/view/235"/>
			
			
			
			<abstract xml:lang="ru"><p>Глубокое обучение с подкреплением (Deep Reinforcement Learning, Deep RL) является одним из наиболее перспективных направлений машинного обучения. Глубокое обучение с подкреплением является важным направлением машинного обучения, находящим применение в автономном управлении, робототехнике и игровых системах. В работе представлена реализация алгоритма Deep Q-Network (DQN) для обучения агента управлению в игре Змейка. Методика включает формирование компактного векторного представления состояния среды из 11 бинарных признаков, применение полносвязной нейронной сети для аппроксимации Q-функции, использование механизма воспроизведения опыта (Experience Replay) с буфером на 100 000 записей и стратегии эпсилон-жадного выбора (ε-greedy) для обеспечения баланса между исследованием среды и эксплуатацией полученных знаний. В ходе исследования проведено обучение агента на протяжении 500 игровых эпизодов с различными конфигурациями среды. Результаты показали устойчивый рост среднего счёта с 0 до 23.7 в процессе обучения и достижение среднего счёта 25.5 при тестировании, что на 7.6% выше финального показателя обучения. Максимальный достигнутый счёт составил 74 очка. Результаты подтверждают применимость Deep RL для решения задач управления в стохастических средах и демонстрируют способность DQN-агента к обобщению выученной стратегии.</p></abstract><trans-abstract xml:lang="en"><p>Deep reinforcement learning (Deep RL) is one of the most promising areas of machine learning. Deep reinforcement learning is an important area of machine learning that finds application in autonomous control, robotics, and gaming systems. This paper presents the implementation of the Deep Q-Network (DQN) algorithm for training an agent to control the Snake game. The methodology includes the formation of a compact vector representation of the environment state from 11 binary features, the use of a fully connected neural network to approximate the Q-function, the use of an experience replay mechanism with a buffer of 100,000 records, and an ε-greedy strategy to ensure a balance between exploring the environment and exploiting the knowledge gained. During the study, the agent was trained over 500 game episodes with different environment configurations. The results showed a steady increase in the average score from 0 to 23.7 during training and an average score of 25.5 during testing, which is 7.6% higher than the final training score. The maximum score achieved was 74 points. The results confirm the applicability of Deep RL for solving control problems in stochastic environments and demonstrate the ability of the DQN agent to generalize the learned strategy.</p></trans-abstract><kwd-group xml:lang="en"><title>Keywords</title><kwd>reinforcement learning</kwd><kwd>Deep Q-Network</kwd><kwd>neural networks</kwd><kwd>Q-learning</kwd><kwd>game agents</kwd><kwd>machine learning</kwd></kwd-group><kwd-group xml:lang="ru"><title>Ключевые слова</title><kwd>обучение с подкреплением</kwd><kwd>Deep Q-Network</kwd><kwd>нейронные сети</kwd><kwd>Q-обучение</kwd><kwd>игровые агенты</kwd><kwd>машинное обучение</kwd></kwd-group><funding-group>
				<funding-statement xml:lang="ru">Исследование выполнено без внешнего финансирования.</funding-statement>
				<funding-statement xml:lang="en">The study was conducted without external funding.</funding-statement>
			</funding-group>
			<counts><page-count count="9"/></counts>
			<custom-meta-group>
				<custom-meta>
					<meta-name>metadata-license</meta-name>
					<meta-value><ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/publicdomain/zero/1.0/">CC0 1.0</ext-link></meta-value>
				</custom-meta>
			<custom-meta><meta-name>issue-cover</meta-name><meta-value><inline-graphic xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://iereview.ru/public/journals/1/cover_issue_21_ru_RU.jpg"/></meta-value></custom-meta></custom-meta-group>
		</article-meta>
	</front>
	<back>
		<ref-list xml:lang="ru">
			<title>Список литературы</title>
			<ref id="R1"><mixed-citation>Chen C., Ying V., Laird D. Глубокое Q-обучение с рекуррентными нейронными сетями: отчет о проекте // CS229 Final Project Report. Stanford University. 2016. С. 1–6. URL: http://cs229.stanford.edu/proj2016/report/ChenYingLairdDeepQLearningWithRecurrentNeuralNetwords-report.pdf (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R2"><mixed-citation>Mnih V., Kavukcuoglu K., Silver D. и др. Обучение игре Atari с использованием глубокого обучения с подкреплением // arXiv.org. 2013. С. 1–9. URL: https://arxiv.org/abs/1312.5602 (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R3"><mixed-citation>Mnih V., Kavukcuoglu K., Silver D. и др. Управление на уровне человека с использованием глубокого обучения с подкреплением // Nature. 2015. Т. 518. № 7540. С. 529–533. DOI: 10.1038/nature14236.</mixed-citation></ref>
			<ref id="R4"><mixed-citation>Osband I., Blundell C., Pritzel A., Van Roy B. Глубокое исследование среды с использованием Bootstrapped DQN // Advances in Neural Information Processing Systems (NeurIPS). 2016. Т. 29. С. 4026–4034. URL: https://arxiv.org/abs/1602.04621 (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R5"><mixed-citation>Schaul T., Quan J., Antonoglou I., Silver D. Приоритетное воспроизведение опыта // arXiv.org. 2015. С. 1–21. URL: https://arxiv.org/abs/1511.05952 (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R6"><mixed-citation>Schulman J., Wolski F., Dhariwal P. и др. Алгоритмы проксимальной оптимизации политики // arXiv.org. 2017. С. 1–12. URL: https://arxiv.org/abs/1707.06347 (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R7"><mixed-citation>Sutton R.S., Barto A.G. Обучение с подкреплением: введение. 2-е изд. Кембридж: MIT Press, 2018. 548 с. URL: http://incompleteideas.net/book/the-book-2nd.html (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R8"><mixed-citation>Van Hasselt H., Guez A., Silver D. Глубокое обучение с подкреплением с использованием Double Q-learning // Proceedings of the AAAI Conference on Artificial Intelligence. 2016. Т. 30. № 1. С. 2094–2100. URL: https://arxiv.org/abs/1509.06461 (дата обращения: 26.12.2025).</mixed-citation></ref>
			<ref id="R9"><mixed-citation>Watkins C.J.C.H., Dayan P. Q-learning // Machine Learning. 1992. Т. 8. № 3–4. С. 279–292. DOI: 10.1007/BF00992698.</mixed-citation></ref>
			<ref id="R10"><mixed-citation>Yuwono F., Yen G.P., Christopher J. Гоночные автомобили с автопилотом: применение глубокого обучения с подкреплением // arXiv.org. 2024. С. 1–8. URL: https://arxiv.org/abs/2410.22766 (дата обращения: 26.12.2025).</mixed-citation></ref>
		</ref-list>
	</back>
</article>			</metadata>
		</record>
	</GetRecord>
</OAI-PMH>
