<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">pribor</journal-id><journal-title-group><journal-title xml:lang="ru">Известия высших учебных заведений. Приборостроение</journal-title><trans-title-group xml:lang="en"><trans-title>Journal of Instrument Engineering</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">0021-3454</issn><issn pub-type="epub">2500-0381</issn><publisher><publisher-name>Национальный исследовательский университет ИТМО</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.17586/0021-3454-2026-69-6-485-496</article-id><article-id custom-type="elpub" pub-id-type="custom">pribor-546</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>СИСТЕМНЫЙ АНАЛИЗ, УПРАВЛЕНИЕ И ОБРАБОТКА ИНФОРМАЦИИ</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>SYSTEM ANALYSIS, MANAGEMENT AND INFORMATION PROCESSING</subject></subj-group></article-categories><title-group><article-title>Объектно-центрическое картографирование динамического окружения с использованием фундаментальных моделей</article-title><trans-title-group xml:lang="en"><trans-title>SAM3R: Object-Centric 3D Mapping via Foundation-Model-Guided Data Association in Changing Scenes</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Мохрат</surname><given-names>М.</given-names></name><name name-style="western" xml:lang="en"><surname>Mohrat</surname><given-names>M.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Малик Мохрат — аспирант, факультет систем управления и робототехники; инженер-исследователь, ведущий инженер-разработчик</p><p>Санкт-Петербург</p></bio><bio xml:lang="en"><p>Malik Mohrat — PhD student, Faculty of Control Systems and Robotics; Leading Engineer-Developer</p><p>Saint Petersburg; Moscow</p></bio><email xlink:type="simple">mmohrat@itmo.ru</email><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Деревянка</surname><given-names>Е. Е.</given-names></name><name name-style="western" xml:lang="en"><surname>Derevyanka</surname><given-names>E. E.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Екатерина Евгеньевна Деревянка — канд. физ.-мат. наук,  исполнительный директор</p><p>Москва</p></bio><bio xml:lang="en"><p>Ekaterina Е. Derevyanka — PhD, Executive Director</p><p>Moscow</p></bio><email xlink:type="simple">eevderevyanka@sberbank.ru</email><xref ref-type="aff" rid="aff-2"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Обрубов</surname><given-names>И. И.</given-names></name><name name-style="western" xml:lang="en"><surname>Obrubov</surname><given-names>I. I.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Илья Игоревич Обрубов —  ведущий инженер</p><p>Москва</p></bio><bio xml:lang="en"><p>Ilya I. Obrubov — Leading Engineer</p><p>Moscow</p></bio><email xlink:type="simple">iiobrubov@sberbank.ru</email><xref ref-type="aff" rid="aff-2"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Сосин</surname><given-names>И. А.</given-names></name><name name-style="western" xml:lang="en"><surname>Sosin</surname><given-names>I. А.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Иван Александрович Сосин — исполнительный директор</p><p>Москва</p></bio><bio xml:lang="en"><p>Ivan А. Sosin — Executive Director</p><p>Moscow</p></bio><email xlink:type="simple">iasosin@sberbank.ru</email><xref ref-type="aff" rid="aff-2"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Колюбин</surname><given-names>С. А.</given-names></name><name name-style="western" xml:lang="en"><surname>Kolyubin</surname><given-names>S. A.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Сергей Алексеевич Колюбин — д-р техн. наук, профессор, факультет систем управления и робототехники</p><p>Санкт-Петербург</p></bio><bio xml:lang="en"><p>Sergey A. Kolyubin — Dr. Sci., Professor, Faculty of Control Systems and Robotics</p><p>Saint Petersburg</p></bio><email xlink:type="simple">s.kolyubin@itmo.ru</email><xref ref-type="aff" rid="aff-3"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Университет ИТМО; Центр робототехники Сбера</institution><country>Россия</country></aff><aff xml:lang="en"><institution>ITMO University; Sber Robotics Center</institution><country>Russian Federation</country></aff></aff-alternatives><aff-alternatives id="aff-2"><aff xml:lang="ru"><institution>Центр робототехники Сбера</institution><country>Россия</country></aff><aff xml:lang="en"><institution>Sber Robotics Center</institution><country>Russian Federation</country></aff></aff-alternatives><aff-alternatives id="aff-3"><aff xml:lang="ru"><institution>Университет ИТМО</institution><country>Россия</country></aff><aff xml:lang="en"><institution>ITMO University</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>18</day><month>07</month><year>2026</year></pub-date><volume>69</volume><issue>6</issue><fpage>485</fpage><lpage>496</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Национальный исследовательский университет ИТМО, 2026</copyright-statement><copyright-year>2026</copyright-year><copyright-holder xml:lang="ru">Национальный исследовательский университет ИТМО</copyright-holder><copyright-holder xml:lang="en">Национальный исследовательский университет ИТМО</copyright-holder><license xlink:href="https://pribor.ifmo.ru/jour/about/submissions#copyrightNotice" xlink:type="simple"><license-p>https://pribor.ifmo.ru/jour/about/submissions#copyrightNotice</license-p></license></permissions><self-uri xlink:href="https://pribor.ifmo.ru/jour/article/view/546">https://pribor.ifmo.ru/jour/article/view/546</self-uri><abstract><p>Для того чтобы агенты могли ориентироваться в пространстве, необходимо использовать их объектно-ориентированные трехмерные представления, сохраняющие согласованность во времени по мере поступления новых кадров с бортовых камер. Традиционные методы либо используют откалиброванные RGBD-данные с эталонной глубиной, либо характеризуются недостаточной модульностью из-за тесной интеграции с внутренними представлениями фундаментальных моделей. Предложен метод онлайн-картографирования динамического окружения, позволяющий преодолеть отмеченные недостатки и не требующий дополнительного обучения. Метод объединяет показатели пространственного перекрытия, смещения 3D-центроидов и визуальносемантического сходства в единую целевую функцию. Показано, что эти компоненты целевой функции дополняют друг друга, так как визуальная идентификация неоднозначна при наличии пространственно разнесенных объектах одного класса, в то время как геометрические данные ненадежны для различных объектов, находящихся в смежных локациях. Отслеживание объектов в динамичной сцене осуществляется путем агрегирования оценок во времени, модулируемого наличием объекта в поле зрения. Испытания на наборах данных ScanNet200 и Replica подтвердили конкурентоспособность разработанного метода. Качественная оценка на данных Aria Digital Twin продемонстрировала устойчивость идентификации объектов в условиях значительных смещений и оптических перекрытий.</p></abstract><trans-abstract xml:lang="en"><p>For embodied agents to navigate and reason indoor spaces, they need object-level 3D representations that stay consistent over time as new frames arrive from a monocular camera. Current online 3D instance segmentation methods either depend on posed RGB-D input with ground-truth depth or couple tightly to the internal representations of specific foundation models, sacrificing modularity. We observe that appearance-based and geometry-based object matching exhibit complementary failure modes: appearance is ambiguous among spatially separated duplicates, while geometry is unreliable for visually distinct objects at similar locations. This motivates SAM3R, a training-free pipeline that fuses spatial overlap, 3D centroid displacement, and visual-semantic similarity into a single assignment cost solved via bipartite matching. The cost is constructed entirely from the outputs of frozen foundation models without accessing internal representations. Object tracks are classified through a cascaded decision tree that detects scene changes via field-of-view gated temporal voting. On ScanNet200 and Replica, SAM3R performs competitively with methods that require architecture-specific features or additional training, despite operating in a fully online, monocular setting. Qualitative evaluation on the Aria Digital Twin dataset further demonstrates that the pipeline maintains correct object identities through physical object manipulation, including hand occlusion and large spatial displacement.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>семантическое картографирование</kwd><kwd>динамическое окружение</kwd><kwd>пространственная сегментация</kwd><kwd>фундаментальные модели</kwd><kwd>отслеживание объектов</kwd><kwd>понимание сцен</kwd></kwd-group><kwd-group xml:lang="en"><kwd>3D Instance Segmentation</kwd><kwd>Foundation Models</kwd><kwd>Object Tracking</kwd><kwd>Bipartite Matching</kwd><kwd>Scene Understanding</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Zhang J., Dai L., Meng F., Fan Q. et al. CVPR, 2023, рр. 6672–6682, DOI:10.1109/CVPR52729.2023.00645.</mixed-citation><mixed-citation xml:lang="en">Zhang J., Dai L., Meng F., Fan Q. et al. 3D-aware object goal navigation via simultaneous exploration and identification // CVPR. 2023. P. 6672–6682. DOI:10.1109/CVPR52729.2023.00645.</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Mur-Artal R.L, Tard’os J.D. IEEE Transactions on Robotics, 2017, no. 5(33), pp. 1255–1262.</mixed-citation><mixed-citation xml:lang="en">Mur-Artal R., Tardos J. D. ORB-SLAM2: An Open-Source SLAM System for Monocular, Stereo, and RGB-D Cameras // IEEE Transactions on Robotics. 2017. Vol. 33, N 5. P. 1255–1262.</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Kirillov A., Mintun E., Ravi N., Mao H. et al. ICCV, 2023, рр. 4015–4026, DOI: 10.1109/ICCV51070.2023.00371.</mixed-citation><mixed-citation xml:lang="en">Kirillov A., Mintun E., Ravi N., Mao H. et al. Segment Anything // ICCV. 2023. P. 4015–4026. DOI: 10.1109/ICCV51070.2023.00371.</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Oquab M., Darcet T., Moutakanni T., Vo H.V. et al. Transactions on Machine Learning Research, 2024, vol. 2024, https://openreview.net/pdf?id=a68SUt6zFt.</mixed-citation><mixed-citation xml:lang="en">Oquab M., Darcet T., Moutakanni Th., Vo H. V. et al. DINOv2: Learning Robust Visual Features without Supervision // Transactions on Machine Learning Research. 2024. Vol. 2024. https://openreview.net/pdf?id=a68SUt6zFt.</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Radford A., Kim J.W., Hallacy Ch., Ramesh A. et al. ICML, 2021, рр. 8748–8763, https://dblp.org/rec/conf/icml/RadfordKHRGASAM21.html.</mixed-citation><mixed-citation xml:lang="en">Radford A., Kim J. W., Hallacy Ch., Ramesh A. et al. Learning Transferable Visual Models from Natural Language Supervision // ICML. 2021. P. 8748–8763. https://dblp.org/rec/conf/icml/RadfordKHRGASAM21.html.</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Wang Sh., Leroy V., Cabon Y., Chidlovskii B. et al. CVPR, 2024, рр. 20697–20709, DOI: 10.1109/CVPR52733.2024.01956.</mixed-citation><mixed-citation xml:lang="en">Wang Sh., Leroy V., Cabon Y., Chidlovskii B. et al. DUSt3R: Geometric 3D Vision Made Easy // CVPR. 2024. P. 20697–20709. DOI: 10.1109/CVPR52733.2024.01956.</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Zhang J., Herrmann Ch., Hur J., Jampani V. et al. ICLR, 2025, https://openreview.net/pdf?id=lJpqxFgWCM.</mixed-citation><mixed-citation xml:lang="en">Zhang J., Herrmann Ch., Hur J., Jampani V. et al. MonST3R: A Simple Approach for Estimating Geometry in the Presence of Motion // ICLR. 2025. https://openreview.net/pdf?id=lJpqxFgWCM.</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Wojke N., Bewley A., Paulus D. ICIP, 2017, рр. 3645–3649, DOI: 10.1109/ICIP.2017.8296962.</mixed-citation><mixed-citation xml:lang="en">Wojke N., Bewley A., Paulus D. Simple Online and Realtime Tracking with a Deep Association Metric // ICIP. 2017. P. 3645–3649. DOI: 10.1109/ICIP.2017.8296962.</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Du Zh., Danier D., Lenssen J.E, Bilen H. arXiv preprint arXiv:2512.15577, 2025.</mixed-citation><mixed-citation xml:lang="en">Du Zh., Danier D., Lenssen J. E., Bilen H. MoonSeg3R: Monocular Online Zero-Shot Segment Anything in 3D with Reconstructive Foundation Priors // arXiv preprint arXiv:2512.15577. 2025.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Tang Y., Zhang J., Lan Y., Guo Y. et al. arXiv:2503.01309v3 [cs.CV], 2025, DOI:10.48550/arXiv.2503.01309.</mixed-citation><mixed-citation xml:lang="en">Tang Y., Zhang J., Lan Y., Guo Y. et al. Onlineanyseg: Online zero-shot 3d segmentation by visual foundation model guided 2d mask merging // arXiv:2503.01309v3 [cs.CV]. 2025. DOI:10.48550/arXiv.2503.01309.</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Sun Y.-C., Tseng Y.-H., Ho Y.-H., Liu Y.-L. arXiv preprint arXiv:2601.08831, 2025.</mixed-citation><mixed-citation xml:lang="en">Sun Y.-Ch., Tseng Y.-H., Ho Y.-H., Liu Y.-L. 3AM: Segment Anything with Geometric Consistency in Videos // arXiv preprint arXiv:2601.08831. 2025.</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Kuhn H.W. Naval Research Logistics Quarterly, 1955, no. 1–2(2), pp. 83–97.</mixed-citation><mixed-citation xml:lang="en">Kuhn H. W. The Hungarian Method for the Assignment Problem // Naval Research Logistics Quarterly. 1955. Vol. 2, N 1–2. P. 83–97.</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Ravi N., Gabeur V., Hu Y.-T., Hu R. et al. arXiv preprint arXiv:2408.00714, 2024.</mixed-citation><mixed-citation xml:lang="en">Ravi N., Gabeur V., Hu Y.-T., Hu R. et al. SAM~2: Segment Anything in Images and Videos // arXiv preprint arXiv:2408.00714. 2024.</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">Cheng H.K., Cho S.W., Schwing A.G. arXiv preprint arXiv:2410.16268, 2024.</mixed-citation><mixed-citation xml:lang="en">Cheng H. K., Cho S. W., Schwing A. G. SAM2Long: Enhancing SAM\,2 for Long Video Segmentation with a TrainingFree Memory Tree // arXiv preprint arXiv:2410.16268. 2024.</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Videnović J., Kristan M. &amp; Lukežič A. International Journal of Computer Vision, 2026, vol. 134, art. no. 211, https://doi.org/10.1007/s11263-026-02790-7.</mixed-citation><mixed-citation xml:lang="en">Videnović J., Kristan M. &amp; Lukežič A. Distractor-Aware Memory-Based Visual Object Tracking // International Journal of Computer Vision. 2026. Vol. 134. Art. no. 211. https://doi.org/10.1007/s11263-026-02790-7.</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Zhang Ch., Han D., Zheng Sh., Choi J. et al. arXiv preprint arXiv:2312.09579, 2023.</mixed-citation><mixed-citation xml:lang="en">Zhang Ch., Han D., Zheng Sh., Choi J. et al. Mobilesamv2: Faster segment anything to everything // arXiv preprint arXiv:2312.09579. 2023.</mixed-citation></citation-alternatives></ref><ref id="cit17"><label>17</label><citation-alternatives><mixed-citation xml:lang="ru">Yang Y., Wu X., He T., Zhao H. et al. arXiv:2306.03908v1 [cs.CV], 2023, https://doi.org/10.48550/arXiv.2306.03908.</mixed-citation><mixed-citation xml:lang="en">Yang Y., Wu X., He T., Zhao H. et al. SAM3D: Segment Anything in 3D Scenes // ICCVarXiv:2306.03908v1 [cs.CV]. 2023. https://doi.org/10.48550/arXiv.2306.03908.</mixed-citation></citation-alternatives></ref><ref id="cit18"><label>18</label><citation-alternatives><mixed-citation xml:lang="ru">Johnson J., Douze M., Jegou H. IEEE Transactions on Big Data, 2021, no. 3(7), pp. 535–547.</mixed-citation><mixed-citation xml:lang="en">Johnson J., Douze M., Jegou H. DINOv3 // IEEE Transactions on Big Data. 2021. Vol. 7, N 3. P. 535–547.</mixed-citation></citation-alternatives></ref><ref id="cit19"><label>19</label><citation-alternatives><mixed-citation xml:lang="ru">Bewley A., Ge Z., Ott L., Ramos F. and Upcroft B. 2016 IEEE International Conference on Image Processing (ICIP), Phoenix, AZ, USA, 2016, pp. 3464–3468, DOI: 10.1109/ICIP.2016.7533003.</mixed-citation><mixed-citation xml:lang="en">Bewley A., Ge Z., Ott L., Ramos F., and Upcroft B. Simple online and realtime tracking // 2016 IEEE Intern. Conf. on Image Processing (ICIP). Phoenix, AZ, USA, 2016. P. 3464–3468. DOI: 10.1109/ICIP.2016.7533003.</mixed-citation></citation-alternatives></ref><ref id="cit20"><label>20</label><citation-alternatives><mixed-citation xml:lang="ru">Wen B., Yang W., Kautz J. and Birchfield S. 2024 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR), Seattle, WA, USA, 2024, pp. 17868–17879, DOI: 10.1109/CVPR52733.2024.01692.</mixed-citation><mixed-citation xml:lang="en">Wen B., Yang W., Kautz J., and Birchfield S. FoundationPose: Unified 6D Pose Estimation and Tracking of Novel Objects // 2024 IEEE/CVF Conf. on Computer Vision and Pattern Recognition (CVPR). Seattle, WA, USA, 2024. Р. 17868–17879. DOI: 10.1109/CVPR52733.2024.01692.</mixed-citation></citation-alternatives></ref><ref id="cit21"><label>21</label><citation-alternatives><mixed-citation xml:lang="ru">Lu Sh., Chang H., Jing E.P., Boularias A. et al. CoRL, 2023, рр. 1610–1620, https://proceedings.mlr.press/v229/lu23a/lu23a.pdf.</mixed-citation><mixed-citation xml:lang="en">Lu Sh., Chang H., Jing E. P., Boularias A. et al. OVIR-3D: Open-Vocabulary 3D Instance Retrieval Without Training on 3D Data // CoRL. 2023. P. 1610–1620. https://proceedings.mlr.press/v229/lu23a/lu23a.pdf.</mixed-citation></citation-alternatives></ref><ref id="cit22"><label>22</label><citation-alternatives><mixed-citation xml:lang="ru">Xu X., Chen H., Zhao L., Wang Zh. et al. arXiv:2408.11811v3 [cs.CV], 2025, https://arxiv.org/html/2408.11811v3.</mixed-citation><mixed-citation xml:lang="en">Xu X., Chen H., Zhao L., Wang Zh. et al. EmbodiedSAM: Online Segment Any 3D Thing in Real Time // arXiv:2408.11811v3 [cs.CV]. 2025. https://arxiv.org/html/2408.11811v3.</mixed-citation></citation-alternatives></ref><ref id="cit23"><label>23</label><citation-alternatives><mixed-citation xml:lang="ru">Ester M., Kriegel H.-P., Sander J., Xu X. KDD, 1996, рр. 226–231, https://www.cs.sfu.ca/~ester/papers/kdd_96.pdf.</mixed-citation><mixed-citation xml:lang="en">Ester M., Kriegel H.-P., Sander J., Xu X. A Density-Based Algorithm for Discovering Clusters in Large Spatial Databases with Noise // KDD. 1996. P. 226–231. https://www.cs.sfu.ca/~ester/papers/kdd_96.pdf.</mixed-citation></citation-alternatives></ref><ref id="cit24"><label>24</label><citation-alternatives><mixed-citation xml:lang="ru">Rozenberszki D., Litany O., Dai A. Computer Vision – ECCV 2022, Lecture Notes in Computer Science, Springer, Cham, 2022, vol. 13693, https://doi.org/10.1007/978-3-031-19827-4_8.</mixed-citation><mixed-citation xml:lang="en">Rozenberszki D., Litany O., Dai A. Language-Grounded Indoor 3D Semantic Segmentation in the Wild // Computer Vision – ECCV 2022. Lecture Notes in Computer Science. 2022. Vol. 13693. Springer, Cham. https://doi.org/10.1007/978-3-031-19827-4_8.</mixed-citation></citation-alternatives></ref><ref id="cit25"><label>25</label><citation-alternatives><mixed-citation xml:lang="ru">Yan M., Zhang J., Zhu Y., Wang H. CVPR, 2024, рр. 28274–28284, DOI: 10.1109/CVPR52733.2024.02671.</mixed-citation><mixed-citation xml:lang="en">Yan M., Zhang J., Zhu Y., Wang H. MaskClustering: View Consensus Based Mask Graph Clustering for OpenVocabulary 3D Instance Segmentation // CVPR. 2024. P. 28274–28284. DOI: 10.1109/CVPR52733.2024.02671.</mixed-citation></citation-alternatives></ref><ref id="cit26"><label>26</label><citation-alternatives><mixed-citation xml:lang="ru">Straub J., Whelan Th., Ma L., Chen Y. et al. arXiv preprint arXiv:1906.05797, 2019.</mixed-citation><mixed-citation xml:lang="en">Straub J., Whelan Th., Ma L., Chen Y. et al. The Replica Dataset: A Digital Replica of Indoor Spaces // arXiv preprint arXiv:1906.05797. 2019.</mixed-citation></citation-alternatives></ref><ref id="cit27"><label>27</label><citation-alternatives><mixed-citation xml:lang="ru">Pan X., Charron N., Yang Y., Peters S. et al. ICCV, 2023, рр. 20133–20143, DOI:10.1109/ICCV51070.2023.01842.</mixed-citation><mixed-citation xml:lang="en">Pan X., Charron N., Yang Y., Peters S. et al. Aria digital twin: A new benchmark dataset for egocentric 3d machine perception // ICCV. 2023. P. 20133–20143. DOI:10.1109/ICCV51070.2023.01842.</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
