[{"data":1,"prerenderedAt":437},["ShallowReactive",2],{"method-kinectfusion2011":3},{"method":4,"reference":46,"equipment":73,"figures":92,"results":93},{"id":5,"label":6,"shortName":7,"title":8,"year":9,"era":10,"cluster":11,"scope":12,"keyIdeaZh":13,"keyIdeaEn":14,"fulltextStatus":15,"publicationStatus":16,"recommendation":17,"constructionRelevance":18,"validationEnvironment":19,"strengths":21,"limitations":24,"sensors":29,"platform":31,"estimator":34,"association":35,"timeModel":36,"deskew":37,"loopClosure":38,"globalOptimization":39,"mapRepresentation":40,"prior":39,"outputGeometry":41,"compute":42,"codeUrl":43,"codeLicense":44,"relatedVersions":45},"kinectfusion2011","Newcombe et al., 2011b","KinectFusion","KinectFusion: Real-time dense surface mapping and tracking",2011,"classic","C08","odometry_with_local_mapping","KinectFusion 將 Kinect 深度串流即時融合到單一全域截斷符號距離函數（Truncated Signed Distance Function, TSDF）體素模型中，並以光線投射（raycasting）產生的模型表面預測，用由粗到細的 ICP（point-to-plane、投影式資料關聯）追蹤感測器位姿。追蹤對象是累積模型而非前一影格，因而在房間尺度內漂移有限。所有步驟皆可在 GPU 上平行化。","KinectFusion fuses depth frames into a global TSDF volume on the GPU and tracks each frame by coarse-to-fine point-to-plane ICP against a raycast prediction of the fused model.","full_text_reviewed","peer_reviewed_published","main_body","論文未於營建現場測試。作者指出大面積平面會造成 ICP 三自由度不受約束，並將整棟建築重建列為尚未解決的記憶體與漂移問題；這兩點直接對應室內牆面、樓板與大空間掃描情境（推論連結）。另外 Kinect 深度在不反射紅外光的材料、細長結構與掠射角表面會產生空洞（Sec. 2.1），工地的鋼筋、玻璃與斜視牆面可能受影響（推論）；論文的迴圈實驗為轉盤定性展示，未提供數值精度。",[20],"controlled_experiment",[22,23],"Constant-time tracking and mapping in room-sized scenes with limited drift (abstract)","Uses only depth, so robust to indoor lighting conditions (Sec. 4.3)",[25,26,27,28],"Main failure case: a large planar scene filling most of the view leaves three DOF unconstrained, causing drift or failure (Sec. 4.3)","Works well for medium-sized rooms with volumes of \u003C=7 m3 (as printed); reconstructing the interior of a whole building would raise memory problems and drift appearing as misalignment upon loop closures (Sec. 6 Conclusions)","Relocalisation is interactive; efficient automatic relocalisation in large models is an open challenge (Sec. 3.5; Sec. 6)","Kinect depth has holes where materials or scene structures do not reflect IR light, on very thin structures and at glancing incidence angles, and suffers motion blur under fast motion (Sec. 2.1)",[30],"RGB-D",[32,33],"hand-held Kinect (Fig. 1; Sec. 1; experiment 5 with 560 x 4 free-moving frames)","Kinect fixed in place observing a tabletop scene on a turntable rotated through a full turn in about 19 s (560 frames)","coarse-to-fine point-to-plane ICP against the raycast model prediction over the bottom 3 pyramid levels with at most 4, 5 and 10 iterations (coarse to fine); small-angle linearisation gives per-correspondence 6x6 systems summed on the GPU by tree reduction and solved by Cholesky on the CPU; a null-space check and an increment-magnitude check switch the system into relocalisation mode","projective data association on bilateral-filtered depth (vertex and normal map pyramid, L = 3), rejecting pairs by vertex distance and normal-angle thresholds; raw depth, not the filtered depth, is fused into the TSDF","discrete poses (every frame)","not_reported","none explicit; loops are closed only implicitly by frame-to-model tracking (turntable loop-closing frames nearly overlap after one pass and more tightly after four passes); on tracking failure an interactive relocalisation asks the user to align the live depth frame with the prediction from the last known pose","none","single fixed-extent dense TSDF volume in GPU memory storing truncated distance and weight (16 bits per component); projective TSDF with nearest-neighbour depth lookup and weighted running average, optional weight cap for moving-average reconstruction of dynamic scenes; 256^3 voxels in the turntable experiments, 64^3 to 512^3 evaluated","dense TSDF surface rendered by raycasting; mesh export not reported in sections read","commodity GPU (model not reported); TSDF update above 65 gigavoxels per second (about 2 ms per full 512^3 volume); tracking at the 30 Hz Kinect frame rate; constant-time tracking and mapping for a given voxel resolution; 64^3 volume with every 6th frame shows graceful degradation",null,"not_verified",[],{"id":5,"kind":47,"shortName":7,"title":8,"authors":48,"year":9,"venue":59,"venueType":60,"publisher":61,"volumeIssuePages":62,"doi":63,"arxivId":43,"url":64,"firstPublicDate":65,"publicationStatus":16,"metadataStatus":66,"fulltextStatus":15,"era":10,"classicReason":67,"codeUrl":43,"cluster":11,"topics":68,"mdpi":69,"verification":70,"label":6,"fulltextRoute":71,"versionRead":72,"addedByCensus":69},"method",[49,50,51,52,53,54,55,56,57,58],"Richard A. Newcombe","Shahram Izadi","Otmar Hilliges","David Molyneaux","David Kim","Andrew J. Davison","Pushmeet Kohli","Jamie Shotton","Steve Hodges","Andrew Fitzgibbon","2011 10th IEEE International Symposium on Mixed and Augmented Reality (ISMAR)","conference","IEEE","pp. 127-136","10.1109\u002Fismar.2011.6092378","https:\u002F\u002Fwww.microsoft.com\u002Fen-us\u002Fresearch\u002Fwp-content\u002Fuploads\u002F2016\u002F02\u002Fismar2011.pdf","2011-10","metadata_verified","principle reused: TSDF volumetric fusion with frame-to-model ICP against a raycast model prediction underlies Kintinuous, voxel hashing, BundleFusion and many later dense mapping systems.",[11],false,"corrected","author copy","Microsoft Research hosted PDF of the ISMAR 2011 paper (10 pages, no IEEE header, copyright line or proceedings page numbers; PDF created 2011-09-13); IEEE Xplore version of record not compared",[74,80,86],{"category":75,"model":76,"canonical":76,"role":77,"dataset":43,"specs":78,"locator":79},"rgbd","Kinect","method input","structured-light depth sensor; on-board ASIC produces an 11-bit 640x480 depth map at 30 Hz; conservative range about 0.4 to 8 m used for raycasting; only depth used","Sec. 2.1; Sec. 3.4",{"category":81,"model":82,"canonical":82,"role":83,"dataset":43,"specs":84,"locator":85},"other","turntable","reference or ground truth","tabletop scene rotated through a full rotation in about 19 s (560 frames) with the Kinect fixed, equivalent to a precise circular sensor path","Sec. 4.1",{"category":87,"model":88,"canonical":88,"role":89,"dataset":43,"specs":90,"locator":91},"compute","commodity GPU (model not reported)","compute for runtime","all tracking and mapping on GPU; TSDF update above 65 gigavoxels per second","Abstract; Sec. 3.3",[],{"totalRows":94,"groupCount":95,"groups":96,"others":430},17,5,[97,229,279,406],{"slug":98,"group":99,"sourceId":100,"sourceLabel":101,"table":102,"selfRows":103,"metrics":104,"seqs":109,"entrants":132,"cells":142,"outcomes":223,"locators":224,"hardware":225,"wordings":226,"notes":227},"dvoslam2013-table-iii","dvoslam2013:Table III","dvoslam2013","Kerl et al., 2013","Table III",10,[105],{"label":106,"unit":107,"statistic":108,"alignment":37},"RMSE of absolute trajectory error","m","RMSE",[110,114,116,118,120,122,124,126,128,130],{"dataset":111,"sequence":112,"environment":113},"TUM RGB-D","fr1\u002Fxyz","indoor TUM RGB-D benchmark sequences (carrying mode and scene type not described in the paper)",{"dataset":111,"sequence":115,"environment":113},"fr1\u002Frpy",{"dataset":111,"sequence":117,"environment":113},"fr1\u002Fdesk",{"dataset":111,"sequence":119,"environment":113},"fr1\u002Fdesk2",{"dataset":111,"sequence":121,"environment":113},"fr1\u002Froom",{"dataset":111,"sequence":123,"environment":113},"fr1\u002F360",{"dataset":111,"sequence":125,"environment":113},"fr1\u002Fteddy",{"dataset":111,"sequence":127,"environment":113},"fr1\u002Fplant",{"dataset":111,"sequence":129,"environment":113},"fr3\u002Foffice",{"dataset":111,"sequence":131,"environment":113},"average (as printed)",[133,136,138,140],{"name":134,"methodId":43,"linkable":69,"proposed":135,"self":69},"Ours (DVO-SLAM)",true,{"name":137,"methodId":43,"linkable":69,"proposed":69,"self":69},"RGB-D SLAM [2], [31]",{"name":139,"methodId":43,"linkable":69,"proposed":69,"self":69},"MRSMap [11]",{"name":141,"methodId":5,"linkable":135,"proposed":69,"self":135},"KinFu (PCL KinectFusion) [5]",[143,147,150,153,156,158,159,161,163,165,167,169,171,173,174,176,178,181,183,185,187,189,191,192,194,197,199,201,203,206,208,209,211,214,216,218,220,221],[144,144,144,145,146,144,146,146,144],0,0.011,-1,[148,144,144,149,146,144,146,146,144],1,0.014,[151,144,144,152,146,144,146,146,144],2,0.013,[154,144,144,155,146,144,146,146,144],3,0.026,[144,144,148,157,146,144,146,146,144],0.02,[148,144,148,155,146,144,146,146,144],[151,144,148,160,146,144,146,146,144],0.027,[154,144,148,162,146,144,146,146,144],0.133,[144,144,151,164,146,144,146,146,144],0.021,[148,144,151,166,146,144,146,146,144],0.023,[151,144,151,168,146,144,146,146,144],0.043,[154,144,151,170,146,144,146,146,144],0.057,[144,144,154,172,146,144,146,146,144],0.046,[148,144,154,168,146,144,146,146,144],[151,144,154,175,146,144,146,146,144],0.049,[154,144,154,177,146,144,146,146,144],0.42,[144,144,179,180,146,144,146,146,144],4,0.053,[148,144,179,182,146,144,146,146,144],0.084,[151,144,179,184,146,144,146,146,144],0.069,[154,144,179,186,146,144,146,146,144],0.313,[144,144,95,188,146,144,146,146,144],0.083,[148,144,95,190,146,144,146,146,144],0.079,[151,144,95,184,146,144,146,146,144],[154,144,95,193,146,144,146,146,144],0.913,[144,144,195,196,146,144,146,146,144],6,0.034,[148,144,195,198,146,144,146,146,144],0.076,[151,144,195,200,146,144,146,146,144],0.039,[154,144,195,202,146,144,146,146,144],0.154,[144,144,204,205,146,144,146,146,144],7,0.028,[148,144,204,207,146,144,146,146,144],0.091,[151,144,204,155,146,144,146,146,144],[154,144,204,210,146,144,146,146,144],0.598,[144,144,212,213,146,144,146,146,144],8,0.035,[154,144,212,215,146,144,146,146,144],0.064,[144,144,217,196,146,144,146,146,144],9,[148,144,217,219,146,144,146,146,144],0.054,[151,144,217,168,146,144,146,146,144],[154,144,217,222,146,144,146,146,144],0.297,[],[102],[],[],[228],"RMSE of absolute trajectory error (m) on TUM RGB-D sequences versus RGB-D SLAM (Engelhard, Endres et al.), MRSMap and the PCL KinectFusion implementation (KinFu); dashes = not available",{"slug":230,"group":231,"sourceId":232,"sourceLabel":233,"table":234,"selfRows":179,"metrics":235,"seqs":240,"entrants":249,"cells":255,"outcomes":268,"locators":269,"hardware":272,"wordings":275,"notes":276},"infinitam2015-table-1","infinitam2015:Table 1","infinitam2015","Kähler et al., 2015","Table 1",[236],{"label":237,"unit":238,"statistic":239,"alignment":37},"average computation time per frame","ms","mean",[241,245],{"dataset":242,"sequence":243,"environment":244},"authors' teddy sequence","teddy","indoor desk scene",{"dataset":246,"sequence":247,"environment":248},"authors' couch sequence","couch","indoor living room, Structure Sensor with iPad Air 2 IMU (carrying mode not stated)",[250,252],{"name":251,"methodId":5,"linkable":135,"proposed":69,"self":135},"KinectFusion implementation [14]",{"name":253,"methodId":254,"linkable":135,"proposed":69,"self":69},"Voxel hashing implementation [16]","voxelhashing2013",[256,258,260,262,264,266],[144,144,144,257,146,144,144,146,144],26.15,[148,144,144,259,146,144,144,146,144],25.87,[144,144,144,261,146,144,148,146,144],502.69,[144,144,148,263,146,148,144,146,148],19.34,[148,144,148,265,146,148,144,146,148],15.18,[144,144,148,267,146,148,148,146,148],312.86,[],[270,271],"Table 1(a)","Table 1(b)",[273,274],"Nvidia Titan X","Intel Core i7-5960X",[],[277,278],"Average computation time per frame over the teddy sequence (Kinect for XBOX 360, 640x480 colour and disparity, no IMU) for three visualisation strategies of InfiniTAM and for the KinectFusion [14] and voxel hashing [16] implementations","Average computation time per frame over the couch sequence (Structure Sensor 320x240 depth with iPad Air 2 IMU) for three visualisation strategies of InfiniTAM and for KinectFusion [14] and voxel hashing [16]; the KinectFusion fixed 512^3 volume could not cover the whole couch scene",{"slug":280,"group":281,"sourceId":282,"sourceLabel":283,"table":284,"selfRows":148,"metrics":285,"seqs":288,"entrants":292,"cells":353,"outcomes":390,"locators":401,"hardware":402,"wordings":403,"notes":404},"ghadimzadeh2025slamnde-table-2","ghadimzadeh2025slamnde:Table 2","ghadimzadeh2025slamnde","Ghadimzadeh Alamdari et al., 2025","Table 2",[286],{"label":287,"unit":39,"statistic":37,"alignment":39},"Result (run outcome)",[289],{"dataset":290,"sequence":37,"environment":291},"Luleå SubT tunnel dataset (Koval et al. 2022)","underground tunnel",[293,296,299,301,303,306,309,312,315,318,320,322,325,327,329,331,334,336,339,341,343,346,348,351],{"name":294,"methodId":295,"linkable":135,"proposed":69,"self":69},"Mono-SLAM","monoslam2007",{"name":297,"methodId":298,"linkable":135,"proposed":69,"self":69},"PTAM","ptam2007",{"name":300,"methodId":43,"linkable":69,"proposed":69,"self":69},"S-PTAM",{"name":302,"methodId":43,"linkable":69,"proposed":69,"self":69},"OV2SLAM",{"name":304,"methodId":305,"linkable":135,"proposed":69,"self":69},"ORB-SLAM (footnote 1)","orbslam2015",{"name":307,"methodId":308,"linkable":135,"proposed":69,"self":69},"DTAM","dtam2011",{"name":310,"methodId":311,"linkable":135,"proposed":69,"self":69},"LSD-SLAM","lsdslam2014",{"name":313,"methodId":314,"linkable":135,"proposed":69,"self":69},"SVO","svo2017",{"name":316,"methodId":317,"linkable":135,"proposed":69,"self":69},"DSO","dso2018",{"name":319,"methodId":5,"linkable":135,"proposed":69,"self":135},"Kinetic Fusion",{"name":321,"methodId":43,"linkable":69,"proposed":69,"self":69},"Dense visual SLAM",{"name":323,"methodId":324,"linkable":135,"proposed":69,"self":69},"Elastic Fusion SLAM","elasticfusion2015",{"name":326,"methodId":43,"linkable":69,"proposed":69,"self":69},"Realtime onboard VI estimation",{"name":328,"methodId":43,"linkable":69,"proposed":69,"self":69},"Multi-sensor fusion",{"name":330,"methodId":43,"linkable":69,"proposed":69,"self":69},"SOFT-SLAM",{"name":332,"methodId":333,"linkable":135,"proposed":69,"self":69},"MSCKF","mourikis2007msckf",{"name":335,"methodId":43,"linkable":69,"proposed":69,"self":69},"ROVIO",{"name":337,"methodId":338,"linkable":135,"proposed":69,"self":69},"OKVIS","okvis2015",{"name":340,"methodId":43,"linkable":69,"proposed":69,"self":69},"VIORB",{"name":342,"methodId":43,"linkable":69,"proposed":69,"self":69},"S-MSCKF",{"name":344,"methodId":345,"linkable":135,"proposed":69,"self":69},"VINS-Mono","vinsmono2018",{"name":347,"methodId":43,"linkable":69,"proposed":69,"self":69},"STCM-SLAM",{"name":349,"methodId":350,"linkable":135,"proposed":69,"self":69},"Kimera","kimera2020",{"name":352,"methodId":43,"linkable":69,"proposed":69,"self":69},"Yolo-SLAM",[354,355,356,357,358,359,360,361,362,363,364,365,367,369,371,373,375,377,378,380,382,384,386,388],[144,144,144,43,144,144,146,146,144],[148,144,144,43,148,144,146,146,144],[151,144,144,43,151,144,146,146,144],[154,144,144,43,151,144,146,146,144],[179,144,144,43,154,144,146,146,144],[95,144,144,43,179,144,146,146,144],[195,144,144,43,95,144,146,146,144],[204,144,144,43,195,144,146,146,144],[212,144,144,43,151,144,146,146,144],[217,144,144,43,179,144,146,146,144],[103,144,144,43,144,144,146,146,144],[366,144,144,43,204,144,146,146,144],11,[368,144,144,43,179,144,146,146,144],12,[370,144,144,43,179,144,146,146,144],13,[372,144,144,43,179,144,146,146,144],14,[374,144,144,43,195,144,146,146,144],15,[376,144,144,43,151,144,146,146,144],16,[94,144,144,43,195,144,146,146,144],[379,144,144,43,204,144,146,146,144],18,[381,144,144,43,179,144,146,146,144],19,[383,144,144,43,212,144,146,146,144],20,[385,144,144,43,179,144,146,146,144],21,[387,144,144,43,144,144,146,146,144],22,[389,144,144,43,217,144,146,146,144],23,[391,392,393,394,395,396,397,398,399,400],"failed (feature detection and tracking)","failed (initialization for ground floor)","not_run (authors could not run the code)","success (footnote 1: authors could not run ORB-SLAM 3, so the original ORB-SLAM was used)","not_run (no publicly available repository)","failed (feature tracking)","failed (tracking)","not_run (inconsistent repository)","success","other: Result cell reads 'SLAM for dynamic environments'; no run outcome stated",[284],[],[],[405],"Run outcome ('Result' column) of each reviewed vision-based method on the Luleå tunnel test dataset; '+' marks methods not integrated with ROS; the '*' (incompatible with VLP-16) symbol is printed on almost every row",{"slug":407,"group":408,"sourceId":5,"sourceLabel":6,"table":409,"selfRows":148,"metrics":410,"seqs":414,"entrants":418,"cells":420,"outcomes":423,"locators":424,"hardware":426,"wordings":427,"notes":428},"kinectfusion2011-text-sec-1","kinectfusion2011:Text Sec. 1","Text Sec. 1",[411],{"label":412,"unit":413,"statistic":37,"alignment":39},"tracking frame rate","Hz",[415],{"dataset":416,"sequence":37,"environment":417},"live Kinect input","indoor room-sized scenes",[419],{"name":7,"methodId":5,"linkable":135,"proposed":135,"self":135},[421],[144,144,144,422,146,144,144,146,144],30,[],[425],"Sec. 1; Sec. 4.2",[88],[],[429],"Tracking rate stated in text; tracking and mapping run at the Kinect frame rate in constant time for a given voxel resolution",[431],{"group":432,"slug":433,"sourceLabel":6,"table":434,"selfRows":148,"datasets":435},"kinectfusion2011:Text Sec. 3.3","kinectfusion2011-text-sec-3-3","Text Sec. 3.3",[436],"not_applicable",1790510655375]