<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">inttrans</journal-id><journal-title-group><journal-title xml:lang="ru">Интеллектуальный транспорт</journal-title><trans-title-group xml:lang="en"><trans-title>Intelligent transport</trans-title></trans-title-group></journal-title-group><issn pub-type="epub">3033-6007</issn><publisher><publisher-name>АО «НИИАС»</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.24412/3033-6007-2026-339-37-52</article-id><article-id custom-type="elpub" pub-id-type="custom">inttrans-115</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>ИСКУССТВЕННЫЙ ИНТЕЛЛЕКТ И МАШИННОЕ ОБУЧЕНИЕ</subject></subj-group></article-categories><title-group><article-title>Построение компактных графов сцены на основе топологической карты для автономной навигации мобильного робота</article-title><trans-title-group xml:lang="en"><trans-title>Building Compact Scene Graphs Based on a Topological Map for Autonomous Navigation of a Mobile Robot</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Муравьев</surname><given-names>К. Ф.</given-names></name><name name-style="western" xml:lang="en"><surname>Muravyev</surname><given-names>K. F.</given-names></name></name-alternatives><bio xml:lang="ru"><p>к.т.н., научный сотрудник</p></bio><bio xml:lang="en"><p>PhD, Research Fellow</p></bio><email xlink:type="simple">kmuravev@frccsc.ru</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Романенко</surname><given-names>В. И.</given-names></name><name name-style="western" xml:lang="en"><surname>Romanenko</surname><given-names>V. I.</given-names></name></name-alternatives><bio xml:lang="ru"><p>студент факультета компьютерных наук</p></bio><bio xml:lang="en"><p>Student of the Faculty of Computer Science</p></bio><email xlink:type="simple">viktorsamr@yandex.ru</email><xref ref-type="aff" rid="aff-2"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Федеральный исследовательский центр «Информатика и управление» Российской академии наук (ФИЦ ИУ РАН)</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Federal Research Center «Computer Science and Control» of the Russian Academy of Sciences (FRC CSC RAS)</institution><country>Russian Federation</country></aff></aff-alternatives><aff-alternatives id="aff-2"><aff xml:lang="ru"><institution>Национальный исследовательский университет «Высшая школа экономики» (НИУ ВШЭ)</institution><country>Россия</country></aff><aff xml:lang="en"><institution>National Research University Higher School of Economics (NRU HSE)</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>27</day><month>09</month><year>2026</year></pub-date><volume>10</volume><issue>3(39)</issue><fpage>37</fpage><lpage>52</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Муравьев К.Ф., Романенко В.И., 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Муравьев К.Ф., Романенко В.И.</copyright-holder><copyright-holder xml:lang="en">Muravyev K.F., Romanenko V.I.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://www.intelligent-transport.ru/jour/article/view/115">https://www.intelligent-transport.ru/jour/article/view/115</self-uri><abstract><p>Автономная навигация мобильного робота в человеко-ориентированных средах требует наличия карты, которая является не только геометрической моделью среды для планирования маршрутов, но также содержит информацию об объектах среды (двери, мебель, офисная техника и т.д.). Такое представление карты обеспечивают графы сцены, в которых вершинами являются помещения, локации и объекты, а ребра кодируют пространственные связности или отношения между объектами. Большинство современных методов построения графов сцены обладают высокой вычислительной сложностью, и построенные ими графы являются избыточными для обеспечения автономной навигации робота. В данной работе предлагается метод построения компактных графов сцены (Compact Scene Graph, CSG), в основе которого лежит вычислительно эффективный метод топологического картирования PRISM-TopoMap и привязка семантических объектов к локациям топологической карты. Построенный граф сцены позволяет строить маршруты до объектов по локациям топологической карты и осуществлять локализацию с высокой точностью. Предложенный метод был экспериментально исследован в симуляционной среде Habitat. Результаты экспериментов показали, что предложенный CSG потребляет значительно меньше памяти, чем традиционные метрические карты и графы сцен, и обеспечивает надежную локализацию по топологической карте за счет привязки к семантическим объектам.</p></abstract><trans-abstract xml:lang="en"><p>Autonomous navigation of a mobile robot in human-centered environments requires a map that contains not only a geometric model of the environment for path planning, but also information about environmental objects (doors, furniture, office equipment, etc.). Scene graphs provide such a map representation, where nodes correspond to rooms, locations, and objects, and edges encode spatial connectivity or relationships between objects. Most modern scene graph construction methods have high computational complexity, and the graphs they produce are redundant for the purposes of autonomous robot navigation. This paper proposes a method for constructing Compact Scene Graphs (CSG), which is based on the computationally efficient topological mapping method PRISM-TopoMap and the association of semantic objects with locations on the topological map. The resulting scene graph enables route planning to objects via topological map locations and achieves high-precision localization. The proposed method was experimentally evaluated in the Habitat simulation environment. The experimental results demonstrate that the proposed CSG consumes significantly less memory than traditional metric maps and scene graphs, while providing reliable localization on the topological map through association with semantic objects.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>граф сцены</kwd><kwd>топологическая карта</kwd><kwd>семантическая сегментация</kwd><kwd>автономная навигация</kwd><kwd>симуляционная среда</kwd></kwd-group><kwd-group xml:lang="en"><kwd>scene graph</kwd><kwd>topological map</kwd><kwd>semantic segmentation</kwd><kwd>autonomous navigation</kwd><kwd>simulated environment</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Labb´e, M. RTAB-Map as an open-source lidar and visual simultaneous localization and mapping library for large-scale and long-term online operation / M. Labb´e, F. Michaud // Journal of Field Robotics. – 2019. – Vol. 36, No. 2. – P. 416–446. – DOI 10.1002/rob.21831.</mixed-citation><mixed-citation xml:lang="en">Labbé, M., &amp; Michaud, F. (2019). RTAB-Map as an open-source lidar and visual simultaneous localization and mapping library for large-scale and long-term online operation. Journal of Field Robotics, 36(2), 416–446. https://doi.org/10.1002/rob.21831</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">GLIM: 3D range-inertial localization and mapping with GPU-accelerated scan matching factors / K. Koide, M. Yokozuka, S. Oishi, A. Banno // Robotics and Autonomous Systems. – 2024. – Vol. 179. – P. 104750. – DOI 10.1016/j.robot.2024.104750.</mixed-citation><mixed-citation xml:lang="en">Koide, K., Yokozuka, M., Oishi, S., &amp; Banno, A. (2024). GLIM: 3D range-inertial localization and mapping with GPU-accelerated scan matching factors. Robotics and Autonomous Systems, , Article 104750. https://doi.org/10.1016/j.robot.2024.104750</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Muravyev, K. Evaluation of RGB-D SLAM in large indoor environments / K. Muravyev, K. Yakovlev // International Conference on Interactive Collaborative Robotics. – 2022. – P. 93–104. – DOI 10.1007/978-3-031-23609-9_9.</mixed-citation><mixed-citation xml:lang="en">Muravyev, K., &amp; Yakovlev, K. (2022). Evaluation of RGB-D SLAM in large indoor environments. In International Conference on Interactive Collaborative Robotics (pp. 93–104). https://doi.org/10.1007/978-3-031-23609-9_9</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Muravyev, K. Evaluation of topological mapping methods in indoor environments / K. Muravyev, K. S. Yakovlev // IEEE Access. – 2023. – Vol. 11. – P. 132683–132698. – DOI 10.1109/ACCESS.2023.3335818.</mixed-citation><mixed-citation xml:lang="en">Muravyev, K., &amp; Yakovlev, K. S. (2023). Evaluation of topological mapping methods in indoor environments. IEEE Access, 11, 132683–132698. https://doi.org/10.1109/ACCESS.2023.</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">PRISM-TopoMap: Online topological mapping with place recognition and scan matching / K. Muravyev, A. Melekhin, D. Yudin, K. Yakovlev // IEEE Robotics and Automation Letters. – 2025. – Vol. 10, No. 4. – P. 3126–3133. – DOI 10.1109/LRA.2025.3541454.</mixed-citation><mixed-citation xml:lang="en">Muravyev, K., Melekhin, A., Yudin, D., &amp; Yakovlev, K. (2025). PRISM-TopoMap: Online topological mapping with place recognition and scan matching. IEEE Robotics and Automation Letters, 10(4), 3126–3133. https://doi.org/10.1109/LRA.2025.3541454</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Topological semantic graph memory for image-goal navigation / N. Kim, O. Kwon, H. Yoo [et al.] // Proceedings of the 6th Conference on Robot Learning (CoRL), PMLR. – 2023. – Vol. 205. – P. 393–402.</mixed-citation><mixed-citation xml:lang="en">Kim, N., Kwon, O., Yoo, H., Choi, Y., Park, J., &amp; Oh, S. (2023). Topological semantic graph memory for image-goal navigation. In Proceedings of the 6th Conference on Robot Learning (CoRL) (Vol. 205, pp. 393–402). PMLR.</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Clio: Real-time task-driven open-set 3D scene graphs / D. Maggio, Y. Chang, N. Hughes [et al.] // IEEE Robotics and Automation Letters. – 2024. – Vol. 9, No. 10. – P. 8921–8928. – DOI 10.1109/LRA.2024.3451395.</mixed-citation><mixed-citation xml:lang="en">Maggio, D., Chang, Y., Hughes, N., Trang, M., Griffith, D., Dougherty, C., Cristofalo, E., Schmid, L., &amp; Carlone, L. (2024). Clio: Real-time task-driven open-set 3D scene graphs. IEEE Robotics and Automation Letters, 9(10), 8921–8928. https://doi.org/10.1109/LRA.2024.3451395</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Hughes, N. Hydra: A real-time spatial perception system for 3D scene graph construction and optimization / N. Hughes, Y. Chang, L. Carlone // arXiv. – 2022. – arXiv:2201.13360. – DOI 10.48550/arXiv.2201.13360.</mixed-citation><mixed-citation xml:lang="en">Hughes, N., Chang, Y., &amp; Carlone, L. (2022). Hydra: A real-time spatial perception system for 3D scene graph construction and optimization (arXiv:2201.13360). arXiv. https://doi.org/10.48550/arXiv.2201.13360</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Khronos: Middleware for simplified time management in CPS / S. Peros, S. Delbruel, S. Michiels [et al.] // Proceedings of the 13th ACM International Conference on Distributed and Event-based Systems. – 2019. – P. 127–138. – DOI 10.1145/3328905.3329507.</mixed-citation><mixed-citation xml:lang="en">Peros, S., Delbruel, S., Michiels, S., Joosen, W., &amp; Hughes, D. (2019). Khronos: Middleware for simplified time management in CPS. In Proceedings of the 13th ACM International Conference on Distributed and Event-based Systems (pp. 127–138). https://doi.org/10.1145/3328905.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Habitat: A platform for embodied AI research / M. Savva, A. Kadian, O. Maksymets [et al.] // Proceedings of the IEEE/CVF International Conference on Computer Vision. – 2019. – P. 9338–9346. – DOI 10.1109/ICCV.2019.00943.</mixed-citation><mixed-citation xml:lang="en">Savva, M., Kadian, A., Maksymets, O., Zhao, Y., Wijmans, E., Jain, B., Straub, J., Liu, J., Koltun, V., Malik, J., Batra, D., &amp; Mottaghi, R. (2019). Habitat: A platform for embodied AI research. In Proceedings of the IEEE/CVF International Conference on Computer Vision (pp.–9346). https://doi.org/10.1109/ICCV.2019.00943</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">A modular robotic system for autonomous exploration and semantic updating in large-scale indoor environments / S. H. Allu, I. Kadosh, T. Summers, Y. Xiang // arXiv. – 2024. – arXiv:2409.15493. – DOI 10.48550/arXiv.2409.15493.</mixed-citation><mixed-citation xml:lang="en">Allu, S. H., Kadosh, I., Summers, T., &amp; Xiang, Y. (2024). A modular robotic system for autonomous exploration and semantic updating in large-scale indoor environments (arXiv:2409.15493). arXiv. https://doi.org/10.48550/arXiv.2409.15493</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Carpenter, G. A. Adaptive resonance theory : Tech. Rep. / G. A. Carpenter, S. Grossberg. – Boston, MA, USA : Boston University, Center for Adaptive Systems and Department of Cognitive and Neural Systems, 1993.</mixed-citation><mixed-citation xml:lang="en">Carpenter, G. A., &amp; Grossberg, S. (1993). Adaptive resonance theory (Technical Report). Boston University, Center for Adaptive Systems and Department of Cognitive and Neural Systems.</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">TopoNav: Topological navigation for efficient exploration in sparse reward environments / J. Hossain, A. Z. M. Faridee, N. Roy [et al.] // 2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS). – 2024. – P. 693–700. – DOI 10.1109/IROS58592.2024.10802380.</mixed-citation><mixed-citation xml:lang="en">Hossain, J., Faridee, A. Z. M., Roy, N., Freeman, J., Gregory, T., &amp; Trout, T. (2024). TopoNav: Topological navigation for efficient exploration in sparse reward environments. In IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS) (pp. 693–700). https://doi.org/10.1109/IROS58592.2024.10802380</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">SemanticTopoLoop: Semantic loop closure with 3D topological graph based on quadric-level object map / Z. Cao, Q. Zhang, J. Guang [et al.] // IEEE Robotics and Automation Letters. – 2024. – Vol. 9, No. 5. – P. 4257–4264. – DOI 10.1109/LRA.2024.3374169.</mixed-citation><mixed-citation xml:lang="en">Cao, Z., Zhang, Q., Guang, J., Wu, S., Hu, Z., &amp; Liu, J. (2024). SemanticTopoLoop: Semantic loop closure with 3D topological graph based on quadric-level object map. IEEE Robotics and Automation Letters, 9(5), 4257–4264. https://doi.org/10.1109/LRA.2024.3374169</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Learning 3D semantic scene graphs from 3D indoor reconstructions / J. Wald, H. Dhamo, N. Navab, F. Tombari // Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition. – 2020. – P. 3961–3970. – DOI 10.1109/CVPR42600.2020.00402.</mixed-citation><mixed-citation xml:lang="en">Wald, J., Dhamo, H., Navab, N., &amp; Tombari, F. (2020). Learning 3D semantic scene graphs from D indoor reconstructions. In Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (pp. 3961–3970). https://doi.org/10.1109/CVPR42600.2020.00402</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Exploiting edge-oriented reasoning for 3D point-based scene graph analysis / C. Zhang, J. Yu, Y. Song, W. Cai // Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition. – 2021. – P. 9705–9715. – DOI 10.1109/CVPR46437.2021.00958.</mixed-citation><mixed-citation xml:lang="en">Zhang, C., Yu, J., Song, Y., &amp; Cai, W. (2021). Exploiting edge-oriented reasoning for 3D pointbased scene graph analysis. In Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (pp. 9705–9715). https://doi.org/10.1109/CVPR46437.2021.00958</mixed-citation></citation-alternatives></ref><ref id="cit17"><label>17</label><citation-alternatives><mixed-citation xml:lang="ru">SGFormer++: Semantic graph transformer for incremental 3D scene graph generation / M. Qi, C. Lv, Z. Fu [et al.] // arXiv. – 2026. – arXiv:2606.15328. – DOI 10.48550/arXiv.2606.15328.</mixed-citation><mixed-citation xml:lang="en">Qi, M., Lv, C., Fu, Z., Zhang, X., &amp; Ma, H. (2026). SGFormer++: Semantic graph transformer for incremental 3D scene graph generation (arXiv:2606.15328). arXiv. https://doi.org/10.48550/arXiv.2606.15328</mixed-citation></citation-alternatives></ref><ref id="cit18"><label>18</label><citation-alternatives><mixed-citation xml:lang="ru">Learning transferable visual models from natural language supervision / A. Radford, J. W. Kim, C. Hallacy [et al.] // Proceedings of the 38th International Conference on Machine Learning, PMLR. – 2021. – Vol. 139. – P. 8748–8763.</mixed-citation><mixed-citation xml:lang="en">Radford, A., Kim, J. W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., Krueger, G., &amp; Sutskever, I. (2021). Learning transferable visual models from natural language supervision. In Proceedings of the 38th International Conference on Machine Learning (Vol. 139, pp. 8748–8763). PMLR.</mixed-citation></citation-alternatives></ref><ref id="cit19"><label>19</label><citation-alternatives><mixed-citation xml:lang="ru">SGRec3D: Self-supervised 3D scene graph learning via object-level scene reconstruction / S. Koch, N. Vaskevicius, M. Colosi [et al.] // Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision. – 2024. – P. 3392–3402. – DOI 10.1109/WACV57701.2024.00337.</mixed-citation><mixed-citation xml:lang="en">Koch, S., Vaskevicius, N., Colosi, M., Hermosilla, P., &amp; Ropinski, T. (2024). SGRec3D: Self-supervised 3D scene graph learning via object-level scene reconstruction. In Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision (pp. 3392–3402). https://doi.org/10.1109/WACV57701.2024.00337</mixed-citation></citation-alternatives></ref><ref id="cit20"><label>20</label><citation-alternatives><mixed-citation xml:lang="ru">vS-Graphs: Integrating visual SLAM and situational graphs through multi-level scene understanding / A. Tourani, S. Ejaz, H. Bavle [et al.] // arXiv. – 2025. – arXiv:2503.01783. – DOI 10.48550/arXiv.2503.01783.</mixed-citation><mixed-citation xml:lang="en">Tourani, A., Ejaz, S., Bavle, H., Morilla-Cabello, D., Sanchez-Lopez, J. L., &amp; Voos, H. (2025). vS-Graphs: Integrating visual SLAM and situational graphs through multi-level scene understanding (arXiv:2503.01783). arXiv. https://doi.org/10.48550/arXiv.2503.01783</mixed-citation></citation-alternatives></ref><ref id="cit21"><label>21</label><citation-alternatives><mixed-citation xml:lang="ru">S-Graphs+: Real-time localization and mapping leveraging hierarchical representations / H. Bavle, J. L. S´anchez-L´opez, M. Shaheer [et al.] // IEEE Robotics and Automation Letters. – 2023. – Vol. 8, No. 8. – P. 4927–4934. – DOI 10.1109/LRA.2023.3290512.</mixed-citation><mixed-citation xml:lang="en">Bavle, H., Sánchez-López, J. L., Shaheer, M., Civera, J., &amp; Voos, H. (2023). S-Graphs+: Real-time localization and mapping leveraging hierarchical representations. IEEE Robotics and Automation Letters, 8(8), 4927–4934. https://doi.org/10.1109/LRA.2023.3290512</mixed-citation></citation-alternatives></ref><ref id="cit22"><label>22</label><citation-alternatives><mixed-citation xml:lang="ru">ConceptGraphs: Open-vocabulary 3D scene graphs for perception and planning / Q. Gu, A. Kuwajerwala, S. Morin [et al.] // 2024 IEEE International Conference on Robotics and Automation (ICRA). – 2024. – P. 5021–5028. – DOI 10.1109/ICRA57147.2024.10610243.</mixed-citation><mixed-citation xml:lang="en">Gu, Q., Kuwajerwala, A., Morin, S., Jatavallabhula, K. M., Sen, B., Agarwal, A., Rivera, C., Paul, W., Ellis, K., Chellappa, R., Gan, C., de Melo, C. M., Tenenbaum, J. B., Torralba, A., Shkurti, F., &amp; Paull, L. (2024). ConceptGraphs: Open-vocabulary 3D scene graphs for perception and planning. In 2024 IEEE International Conference on Robotics and Automation (ICRA) (pp. 5021–5028). https://doi.org/10.1109/ICRA57147.2024.10610243</mixed-citation></citation-alternatives></ref><ref id="cit23"><label>23</label><citation-alternatives><mixed-citation xml:lang="ru">Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection / S. Liu, Z. Zeng, T. Ren [et al.] // Computer Vision – ECCV 2024, Springer. – 2024. – Vol. 47. – P. 38–55. – DOI 10.1007/978-3-031-72970-6_3.</mixed-citation><mixed-citation xml:lang="en">Liu, S., Zeng, Z., Ren, T., Li, F., Zhang, H., Yang, J., Jiang, Q., Li, C., Yang, J., Su, H., Zhu, J., &amp; Zhang, L. (2024). Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection. In Computer Vision – ECCV 2024 (Vol. 47, pp. 38–55). Springer. https://doi.org/10.1007/978-3-031-72970-6_3</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
