<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE root>
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.2" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">News of the Kabardino-Balkarian Scientific Center of the Russian Academy of Sciences</journal-id><journal-title-group><journal-title xml:lang="en">News of the Kabardino-Balkarian Scientific Center of the Russian Academy of Sciences</journal-title><trans-title-group xml:lang="ru"><trans-title>Известия Кабардино-Балкарского научного центра РАН</trans-title></trans-title-group></journal-title-group><issn publication-format="print">1991-6639</issn><issn publication-format="electronic">2949-1940</issn></journal-meta><article-meta><article-id pub-id-type="publisher-id">282117</article-id><article-id pub-id-type="doi">10.35330/1991-6639-2024-26-6-208-218</article-id><article-id pub-id-type="edn">JSKNGG</article-id><article-categories><subj-group subj-group-type="toc-heading" xml:lang="en"><subject>System analysis, management and information processing</subject></subj-group><subj-group subj-group-type="toc-heading" xml:lang="ru"><subject>Системный анализ, управление и обработка информации</subject></subj-group><subj-group subj-group-type="article-type"><subject>Research Article</subject></subj-group></article-categories><title-group><article-title xml:lang="en">Modification of a deep learning algorithm for distributing functions and tasks between a robotic complex and a person in conditions of uncertainty and variability of the environment</article-title><trans-title-group xml:lang="ru"><trans-title>Модификация алгоритма глубокого обучения для распределения функций и задач между робототехническим комплексом и человеком в условиях неопределенности и переменности окружающей среды</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-2352-992X</contrib-id><contrib-id contrib-id-type="spin">1734-9056</contrib-id><name-alternatives><name xml:lang="ru"><surname>Шереужев</surname><given-names>М. А.</given-names></name><name xml:lang="en"><surname>Shereuzhev</surname><given-names>M. А.</given-names></name></name-alternatives><address><country country="RU">Russian Federation</country></address><bio xml:lang="ru"><p>кан. тех. наук, мл. науч. сотр., Центр когнитивных технологий и систем технического зрения, старший преподаватель, кафедра «Робототехнические системы и мехатроника»</p></bio><bio xml:lang="en"><p>Candidate of Engineering Sciences, Junior Research, Center for Cognitive Technologies and Machine Vision Systems, Senior Teacher, The Department of Robotic Systems and Mechatronics</p></bio><email>m.shereuzhev@stankin.ru</email><xref ref-type="aff" rid="aff1"/><xref ref-type="aff" rid="aff2"/></contrib><contrib contrib-type="author"><name-alternatives><name xml:lang="ru"><surname>Го</surname><given-names>У</given-names></name><name xml:lang="en"><surname>Guo</surname><given-names>Wu</given-names></name></name-alternatives><address><country country="RU">Russian Federation</country></address><bio xml:lang="ru"><p>аспирант кафедры «Робототехнические системы и мехатроника»</p></bio><bio xml:lang="en"><p>Post-graduate Student at the Department of Robotic Systems and Mechatronics</p></bio><email>ug@student.bmstu.ru</email><xref ref-type="aff" rid="aff2"/></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid">https://orcid.org/0000-0003-1182-2117</contrib-id><contrib-id contrib-id-type="spin">5410-8433</contrib-id><name-alternatives><name xml:lang="en"><surname>Serebrenny</surname><given-names>V. V.</given-names></name><name xml:lang="ru"><surname>Серебренный</surname><given-names>В. В.</given-names></name></name-alternatives><address><country country="RU">Russian Federation</country></address><bio xml:lang="en"><p>Candidate of Engineering Sciences, Associate Professor, Head of the Department of Robotic Systems and Mechatronics</p></bio><bio xml:lang="ru"><p>кан. тех. наук, доцент, зав. кафедрой «Робототехнические системы и мехатроника»</p></bio><email>vsereb@bmstu.ru</email><xref ref-type="aff" rid="aff2"/></contrib></contrib-group><aff-alternatives id="aff1"><aff><institution xml:lang="en">Moscow State University of Technology STANKIN</institution></aff><aff><institution xml:lang="ru">Московский государственный технологический университет «СТАНКИН»</institution></aff></aff-alternatives><aff-alternatives id="aff2"><aff><institution xml:lang="en">Moscow State Technical University named after N. E. Bauman</institution></aff><aff><institution xml:lang="ru">Московский государственный технический университет имени Н. Э. Баумана</institution></aff></aff-alternatives><content-language>ru</content-language><pub-date date-type="pub" iso-8601-date="2024-12-15" publication-format="electronic"><day>15</day><month>12</month><year>2024</year></pub-date><pub-date date-type="collection"><year>2024</year></pub-date><volume>26</volume><issue>6</issue><issue-title xml:lang="ru"/><issue-title xml:lang="en"/><fpage>208</fpage><lpage>218</lpage><history><date date-type="received" iso-8601-date="2025-03-02"><day>02</day><month>03</month><year>2025</year></date><date date-type="accepted" iso-8601-date="2025-03-02"><day>02</day><month>03</month><year>2025</year></date></history><permissions><copyright-statement xml:lang="en">Copyright ©; 2024, Шереужев М.А., Го У., Серебренный В.V.</copyright-statement><copyright-statement xml:lang="ru">Copyright ©; 2024, Шереужев М.А., Го У., Серебренный В.В.</copyright-statement><copyright-year>2024</copyright-year><copyright-holder xml:lang="en">Шереужев М.А., Го У., Серебренный В.V.</copyright-holder><copyright-holder xml:lang="ru">Шереужев М.А., Го У., Серебренный В.В.</copyright-holder><ali:free_to_read xmlns:ali="http://www.niso.org/schemas/ali/1.0/"/><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/">https://creativecommons.org/licenses/by/4.0</ali:license_ref></license></permissions><self-uri xlink:href="https://journals.rcsi.science/1991-6639/article/view/282117">https://journals.rcsi.science/1991-6639/article/view/282117</self-uri><abstract xml:lang="en"><p>In the real world, conditions are rarely stable, which requires robotic systems to be able to adapt to uncertainty. Human-robot collaboration increases productivity, but this requires effective task allocation methods that consider the characteristics of both parties.<bold> </bold>The aim of the work is to determine optimal strategies for distributing tasks between people and collaborative robots and adaptive control of a collaborative robot under uncertainty and a changing environment.<bold> </bold>Research methods. The paper develops a graph-based approach to task allocation based on the capabilities of a human and a robot. The LSTM memory mechanism is built into the reinforcement learning algorithm to solve the problem of partial observability caused by inaccurate sensor measurements and environmental noise. The Hindsight Experience Replay method is used to overcome the problem of sparse rewards.<bold> </bold>Results.<bold> </bold>The trained model demonstrated stable convergence, achieving a high level of success rate of manipulation of objects.<bold> </bold>The integration of LSTM and HER methods into reinforcement learning allows solving the problems of distributing tasks between a human and a robot under uncertainty and a changing environment. The proposed method can be applied in various scenarios for collaborative robots in complex and changing conditions.</p></abstract><trans-abstract xml:lang="ru"><p>В реальном мире условия редко бывают стабильными, что требует от робототехнических комплексов (РТК) способности к адаптации в условиях неопределенности. Синергия человека и робота повышает производительность, однако для этого необходимы эффективные методы распределения задач, учитывающие особенности обеих сторон. Целью работы является определение оптимальных стратегий распределения задач между людьми и РТК и адаптивное управление РТК в условиях неопределенности и изменяющейся среды. Методы исследования. В работе предложен графовый подход к распределению задач, основанный на возможностях человека и робота. В алгоритм обучения с подкреплением встроен механизм памяти LSTM (Long short-term memory) для решения проблемы частичной наблюдаемости, вызванной неточностью измерений сенсоров и шумом окружающей среды. Метод HER (Hindsight Experience Replay) применен для преодоления проблемы скудных вознаграждений. Результаты. Обученная модель продемонстрировала стабильную сходимость, достигая высокого уровня успешности манипуляции объектами. Интеграция методов LSTM и HER в обучение с подкреплением позволяет успешно решать вопросы распределения задач между человеком и роботом в условиях неопределенности и изменяющейся среды. Предложенный метод можно применять в различных сценариях для РТК в сложных и изменяющихся условиях.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>взаимодействие человека и робота</kwd><kwd>адаптивный алгоритм управления</kwd><kwd>распределение задач</kwd><kwd>обучение с подкреплением</kwd></kwd-group><kwd-group xml:lang="en"><kwd>human robot interaction</kwd><kwd>adaptive control algorithm</kwd><kwd>task distribution</kwd><kwd>reinforcement learning</kwd></kwd-group><funding-group><funding-statement xml:lang="ru">Исследование выполнено на базе МГТУ «СТАНКИН» при финансовой поддержке Министерства науки и высшего образования РФ в рамках государственного задания (проект № FSFS-2024-0012).</funding-statement><funding-statement xml:lang="en">The study was carried out at MSTU STANKIN with the financial support of the Ministry of Science and Higher Education of the Russian Federation within the framework of the state assignment (project No. FSFS-2024-0012).</funding-statement></funding-group></article-meta></front><body></body><back><ref-list><ref id="B1"><label>1.</label><mixed-citation>Fiore M., Clodic A., Alami R. On planning and task achievement modalities for human-robot collaboration. In Experimental Robotics: The 14th International Symposium on Experimental Robotics. Marrakech, Morocco: Springer. 2016. Pp. 293–306.</mixed-citation></ref><ref id="B2"><label>2.</label><mixed-citation>Ghadirzadeh A., Chen X., Yin W. et al. Human-centered collaborative robots with deep reinforcement learning. IEEE Robotics and Automation Letters. 2020. Vol. 6(2). Pp. 566–571. DOI: 10.48550/arXiv.2007.01009</mixed-citation></ref><ref id="B3"><label>3.</label><mixed-citation>Qureshi A.H., Nakamura Y., Yoshikawa Y., Ishiguro H. Robot gains social intelligence through multimodal deep reinforcement learning. In IEEE-RAS. 16th International Conference on Humanoid Robots (humanoids). 2016. Pp. 745–751. DOI: 10.48550/arXiv.1702.07492</mixed-citation></ref><ref id="B4"><label>4.</label><mixed-citation>Kwok Y.K., Ahmad I. Static scheduling algorithms for allocating directed task graphs to multiprocessors. ACM Computing Surveys. 1999. Vol. 31(4). Pp. 406–471. DOI: 10.1145/344588.344618</mixed-citation></ref><ref id="B5"><label>5.</label><mixed-citation>Malik A.A., Bilberg A. Complexity-based task allocation in human-robot collaborative assembly. Industrial Robot: International Journal of Robotics Research and Application. 2019. Vol. 46(4). Pp. 471–480. DOI: 10.1108/IR-11-2018-0231</mixed-citation></ref><ref id="B6"><label>6.</label><mixed-citation>Lucignano L., Cutugno F., Rossi S., Finzi A. A dialogue system for multimodal human-robot interaction. Proceedings of the 15th ACM on International Conference on Multimodal Interaction. 2013. Pp. 197–204. DOI: 10.1145/2522848.2522873</mixed-citation></ref><ref id="B7"><label>7.</label><mixed-citation>Qiu C., Hu Y., Chen Y., Zeng B. Deep deterministic policy gradient (DDPG)-based energy harvesting wireless communications. IEEE Internet of Things Journal. 2019. Vol. 6(5). Pp. 8577–8588. DOI: 10.1109/JIOT.2019.2921159</mixed-citation></ref><ref id="B8"><label>8.</label><mixed-citation>Hochreiter S. Long Short-term Memory. Neural Computation MIT-Press. 1997.</mixed-citation></ref><ref id="B9"><label>9.</label><mixed-citation>Andrychowicz M., Wolski F., Ray A. et al. Hindsight experience replay. Advances in Neural Information Processing Systems. 2017. Vol. 30.</mixed-citation></ref><ref id="B10"><label>10.</label><mixed-citation>Towers M., Kwiatkowski A., Terry J. et al. Gymnasium: A standard interface for reinforcement learning environments. arXiv:2407.17032. 2024. DOI: 10.48550/arXiv.2407.17032</mixed-citation></ref></ref-list></back></article>
