[{"data":1,"prerenderedAt":441},["ShallowReactive",2],{"method-vggt2025":3},{"method":4,"reference":49,"equipment":72,"figures":80,"results":81},{"id":5,"label":6,"shortName":7,"title":8,"year":9,"era":10,"cluster":11,"scope":12,"keyIdeaZh":13,"keyIdeaEn":14,"fulltextStatus":15,"publicationStatus":16,"recommendation":17,"constructionRelevance":18,"validationEnvironment":19,"strengths":21,"limitations":26,"sensors":31,"platform":33,"estimator":34,"association":35,"timeModel":36,"deskew":36,"loopClosure":36,"globalOptimization":37,"mapRepresentation":38,"prior":39,"outputGeometry":40,"compute":41,"codeUrl":42,"codeLicense":43,"relatedVersions":44},"vggt2025","Wang et al., 2025b","VGGT","VGGT: Visual Geometry Grounded Transformer",2025,"recent","C09","map_representation_or_reconstruction","VGGT 是前饋式 Transformer，可由一張到數百張影像直接推論相機參數、深度圖、點圖與點軌跡，不需後續幾何最佳化；若再加上選用的 BA 後處理，位姿精度還能提升。訓練時以平均點距離正規化真值，模型學到的是固定的標準尺度而非公制尺度。ETH3D 點雲評估前以 Umeyama 演算法對齊真值，但論文未說明是否同時估計尺度，DTU 與 ETH3D 的精度指標也未註明單位，因此這些數值不能直接視為公制幾何誤差。","Feed-forward multi-view geometry Transformer predicting cameras, depth and point maps in normalized scale; evaluated after similarity alignment.","full_text_reviewed","peer_reviewed_published","background","論文未涉及營建場域。",[20],"public_benchmark",[22,23,24,25],"Outperforms DUSt3R and MASt3R global alignment on ETH3D point maps in a feed-forward regime (Sec. 4.3; Table 3)","Depth plus camera unprojection more accurate than the direct point head (Sec. 4.3; Table 3)","On DTU without GT cameras, Overall 0.382 versus 1.741 for DUSt3R, close to methods that use GT cameras (Sec. 4.2; Table 2)","Optional BA refinement raises pose AUC@30 from 85.3 to 93.5 on RealEstate10K in about 1.8 s (Table 1)",[27,28,29,30],"No fisheye or panoramic support; degrades with extreme rotations; fails under substantial non-rigid deformation (Limitations)","GPU memory grows with frame count, 40.63 GB peak for 200 frames on an H100 (Table 9); [vggtslam2025] reports about 60 images on a 24 GB RTX 4090","Principal point assumed at the image centre (Sec. 3.1)","Outputs normalized, non-metric geometry; ETH3D metrics computed after Umeyama alignment and units for DTU and ETH3D are not stated (Sec. 3.4; Sec. 4.3)",[32],"monocular camera (one to hundreds of views)",[],"feed-forward Transformer with alternating frame\u002Fglobal attention predicting cameras, depth, point maps and tracks","implicit (learned); 3D point tracks","not_applicable","none in the feed-forward model; the paper also reports an optional post-hoc bundle-adjustment variant ('Ours (with BA)', Table 1)","per-view depth and point maps in the first-camera frame","learned prior trained on a large mixture of 3D datasets","point clouds (point head or depth+camera unprojection), camera parameters, depth maps; normalized, non-metric scale","Feature backbone on one H100 (flash attention v3, 336x518 images): 0.04 s and 1.88 GB for 1 frame, 1.04 s and 11.41 GB for 50 frames, 8.75 s and 40.63 GB for 200 frames (arXiv v1 Table 9); about 0.2 s feed-forward and about 1.8 s with BA for 10 frames (Table 1); trained on 64 A100 GPUs for nine days (Sec. 3.4); [vggtslam2025] reports about 60 images on a 24 GB RTX 4090","https:\u002F\u002Fgithub.com\u002Ffacebookresearch\u002Fvggt","VGGT License v1 (custom Meta licence incorporating an Acceptable Use Policy; grant clause read, no explicit non-commercial clause; full terms not legally reviewed)",[45],{"relation":46,"title":47,"doi_or_url":48},"preprint","arXiv:2503.11651","https:\u002F\u002Farxiv.org\u002Fabs\u002F2503.11651",{"id":5,"kind":50,"shortName":7,"title":8,"authors":51,"year":9,"venue":58,"venueType":59,"publisher":60,"volumeIssuePages":61,"doi":62,"arxivId":63,"url":64,"firstPublicDate":65,"publicationStatus":16,"metadataStatus":66,"fulltextStatus":15,"era":10,"classicReason":36,"codeUrl":42,"cluster":11,"topics":67,"mdpi":68,"verification":69,"label":6,"fulltextRoute":70,"versionRead":71,"addedByCensus":68},"component",[52,53,54,55,56,57],"Jianyuan Wang","Minghao Chen","Nikita Karaev","Andrea Vedaldi","Christian Rupprecht","David Novotny","2025 IEEE\u002FCVF Conference on Computer Vision and Pattern Recognition (CVPR)","conference","IEEE","pp. 5294-5306","10.1109\u002Fcvpr52734.2025.00499","2503.11651","https:\u002F\u002Fapi.crossref.org\u002Fworks\u002F10.1109\u002Fcvpr52734.2025.00499","2025-03-14","metadata_verified",[11],false,"corrected","arXiv","arXiv v1 (14 Mar 2025) including appendix A to E, read in full; CVF open-access CVPR 2025 main paper also compared: Tables 2 to 8 numerically identical, but Table 1 differs (FLARE CO3Dv2 AUC@30 83.4 in CVF vs 83.3 in arXiv v1, and the CVF caption marks MV-DUSt3R's CO3Dv2 value 69.5 as not trained on CO3D); the CVF main paper lacks the runtime and memory Table 9 and the appendix IMC Table 10",[73],{"category":74,"model":75,"canonical":75,"role":76,"dataset":77,"specs":78,"locator":79},"compute","NVIDIA H100","compute for runtime",null,"Single GPU with flash attention v3 for runtime and memory measurements; also used for Table 1 timings","Table 1 caption; Table 9",[],{"totalRows":82,"groupCount":83,"groups":84,"others":440},35,4,[85,176,245,361],{"slug":86,"group":87,"sourceId":5,"sourceLabel":6,"table":88,"selfRows":89,"metrics":90,"seqs":100,"entrants":119,"cells":122,"outcomes":168,"locators":169,"hardware":171,"wordings":173,"notes":174},"vggt2025-table-9","vggt2025:Table 9","Table 9",18,[91,96],{"label":92,"unit":93,"statistic":94,"alignment":95},"Backbone inference time for N frames","s","not_reported","none",{"label":97,"unit":98,"statistic":99,"alignment":95},"Peak GPU memory","GB","max",[101,103,105,107,109,111,113,115,117],{"dataset":36,"sequence":102,"environment":36},"1 input frames",{"dataset":36,"sequence":104,"environment":36},"2 input frames",{"dataset":36,"sequence":106,"environment":36},"4 input frames",{"dataset":36,"sequence":108,"environment":36},"8 input frames",{"dataset":36,"sequence":110,"environment":36},"10 input frames",{"dataset":36,"sequence":112,"environment":36},"20 input frames",{"dataset":36,"sequence":114,"environment":36},"50 input frames",{"dataset":36,"sequence":116,"environment":36},"100 input frames",{"dataset":36,"sequence":118,"environment":36},"200 input frames",[120],{"name":7,"methodId":5,"linkable":121,"proposed":121,"self":121},true,[123,127,130,132,134,137,139,142,144,146,148,151,153,156,158,161,163,166],[124,124,124,125,126,124,124,126,124],0,0.04,-1,[124,128,124,129,126,124,124,126,124],1,1.88,[124,124,128,131,126,124,124,126,124],0.05,[124,128,128,133,126,124,124,126,124],2.07,[124,124,135,136,126,124,124,126,124],2,0.07,[124,128,135,138,126,124,124,126,124],2.45,[124,124,140,141,126,124,124,126,124],3,0.11,[124,128,140,143,126,124,124,126,124],3.23,[124,124,83,145,126,124,124,126,124],0.14,[124,128,83,147,126,124,124,126,124],3.63,[124,124,149,150,126,124,124,126,124],5,0.31,[124,128,149,152,126,124,124,126,124],5.58,[124,124,154,155,126,124,124,126,124],6,1.04,[124,128,154,157,126,124,124,126,124],11.41,[124,124,159,160,126,124,124,126,124],7,3.12,[124,128,159,162,126,124,124,126,124],21.15,[124,124,164,165,126,124,124,126,124],8,8.75,[124,128,164,167,126,124,124,126,124],40.63,[],[170],"Table 9 (arXiv v1, Sec. 5)",[172],"single NVIDIA H100 with flash attention v3; 336x518 images",[],[175],"Feature-backbone inference time and peak GPU memory versus number of input frames (arXiv v1 only; not in the CVF main paper)",{"slug":177,"group":178,"sourceId":5,"sourceLabel":6,"table":179,"selfRows":164,"metrics":180,"seqs":189,"entrants":194,"cells":205,"outcomes":236,"locators":237,"hardware":238,"wordings":240,"notes":241},"vggt2025-table-3","vggt2025:Table 3","Table 3",[181,183,185,187],{"label":182,"unit":94,"statistic":94,"alignment":94},"Acc.",{"label":184,"unit":94,"statistic":94,"alignment":94},"Comp.",{"label":186,"unit":94,"statistic":94,"alignment":94},"Overall (Chamfer)",{"label":188,"unit":93,"statistic":94,"alignment":94},"Time (approximate, per reconstruction)",[190],{"dataset":191,"sequence":192,"environment":193},"ETH3D","10 random frames per scene","not described in the paper",[195,198,201,203],{"name":196,"methodId":197,"linkable":121,"proposed":68,"self":68},"DUSt3R","dust3r2024",{"name":199,"methodId":200,"linkable":121,"proposed":68,"self":68},"MASt3R","mast3r2024",{"name":202,"methodId":5,"linkable":121,"proposed":121,"self":121},"Ours (Point)",{"name":204,"methodId":5,"linkable":121,"proposed":121,"self":121},"Ours (Depth + Cam)",[206,208,210,212,213,215,217,219,221,223,225,227,229,231,233,235],[124,124,124,207,126,124,126,126,124],1.167,[124,128,124,209,126,124,126,126,124],0.842,[124,135,124,211,126,124,126,126,124],1.005,[124,140,124,159,126,124,124,126,124],[128,124,124,214,126,124,126,126,124],0.968,[128,128,124,216,126,124,126,126,124],0.684,[128,135,124,218,126,124,126,126,124],0.826,[128,140,124,220,126,124,124,126,124],9,[135,124,124,222,126,124,126,126,128],0.901,[135,128,124,224,126,124,126,126,128],0.518,[135,135,124,226,126,124,126,126,128],0.709,[135,140,124,228,126,124,124,126,128],0.2,[140,124,124,230,126,124,126,126,135],0.873,[140,128,124,232,126,124,126,126,135],0.482,[140,135,124,234,126,124,126,126,135],0.677,[140,140,124,228,126,124,124,126,135],[],[179],[239],"not stated for Table 3 (Table 1 timings used one NVIDIA H100)",[],[242,243,244],"Point map estimation on ETH3D, 10 random frames per scene, predicted cloud aligned to GT with the Umeyama algorithm (similarity or rigid not stated), invalid points filtered with official masks; global alignment; units not stated","Point map estimation on ETH3D, 10 random frames per scene, predicted cloud aligned to GT with the Umeyama algorithm (similarity or rigid not stated), invalid points filtered with official masks; point map head, feed-forward; units not stated","Point map estimation on ETH3D, 10 random frames per scene, predicted cloud aligned to GT with the Umeyama algorithm (similarity or rigid not stated), invalid points filtered with official masks; depth head unprojected with camera head, feed-forward; units not stated",{"slug":246,"group":247,"sourceId":5,"sourceLabel":6,"table":248,"selfRows":154,"metrics":249,"seqs":255,"entrants":262,"cells":285,"outcomes":352,"locators":354,"hardware":355,"wordings":357,"notes":358},"vggt2025-table-1","vggt2025:Table 1","Table 1",[250,253],{"label":251,"unit":252,"statistic":94,"alignment":95},"AUC@30 (RRA and RTA)","AUC score (0 to 100)",{"label":254,"unit":93,"statistic":94,"alignment":95},"Time (approximate)",[256,258,260],{"dataset":257,"sequence":192,"environment":193},"RealEstate10K (unseen)",{"dataset":259,"sequence":192,"environment":193},"CO3Dv2",{"dataset":261,"sequence":192,"environment":193},"RealEstate10K and CO3Dv2 (single time column)",[263,265,267,269,270,271,273,275,277,279,281,283],{"name":264,"methodId":77,"linkable":68,"proposed":68,"self":68},"Colmap+SPSG",{"name":266,"methodId":77,"linkable":68,"proposed":68,"self":68},"PixSfM",{"name":268,"methodId":77,"linkable":68,"proposed":68,"self":68},"PoseDiff",{"name":196,"methodId":197,"linkable":121,"proposed":68,"self":68},{"name":199,"methodId":200,"linkable":121,"proposed":68,"self":68},{"name":272,"methodId":77,"linkable":68,"proposed":68,"self":68},"VGGSfM v2",{"name":274,"methodId":77,"linkable":68,"proposed":68,"self":68},"MV-DUSt3R",{"name":276,"methodId":77,"linkable":68,"proposed":68,"self":68},"CUT3R",{"name":278,"methodId":77,"linkable":68,"proposed":68,"self":68},"FLARE",{"name":280,"methodId":77,"linkable":68,"proposed":68,"self":68},"Fast3R",{"name":282,"methodId":5,"linkable":121,"proposed":121,"self":121},"Ours (Feed-Forward)",{"name":284,"methodId":5,"linkable":121,"proposed":121,"self":121},"Ours (with BA)",[286,288,290,292,294,296,298,300,302,304,306,308,310,312,314,316,318,320,322,324,326,329,331,334,336,338,339,340,341,342,343,345,346,348,349,350],[124,124,124,287,126,124,126,126,124],45.2,[124,124,128,289,126,124,126,126,124],25.3,[128,124,124,291,126,124,126,126,124],49.4,[128,124,128,293,126,124,126,126,124],30.1,[135,124,124,295,126,124,126,126,124],48,[135,124,128,297,126,124,126,126,124],66.5,[140,124,124,299,126,124,126,126,124],67.7,[140,124,128,301,126,124,126,126,124],76.7,[83,124,124,303,126,124,126,126,124],76.4,[83,124,128,305,126,124,126,126,124],81.8,[149,124,124,307,126,124,126,126,124],78.9,[149,124,128,309,126,124,126,126,124],83.4,[154,124,124,311,126,124,126,126,124],71.3,[154,124,128,313,126,124,126,126,124],69.5,[159,124,124,315,126,124,126,126,124],75.3,[159,124,128,317,126,124,126,126,124],82.8,[164,124,124,319,126,124,126,126,124],78.8,[164,124,128,321,126,124,126,126,124],83.3,[220,124,124,323,126,124,126,126,124],72.7,[220,124,128,325,126,124,126,126,124],82.5,[327,124,124,328,126,124,126,126,124],10,85.3,[327,124,128,330,126,124,126,126,124],88.2,[332,124,124,333,126,124,126,126,124],11,93.5,[332,124,128,335,126,124,126,126,124],91.8,[124,128,135,337,126,124,124,126,128],15,[128,128,135,77,124,124,124,126,128],[135,128,135,159,126,124,124,126,128],[140,128,135,159,126,124,124,126,128],[83,128,135,220,126,124,124,126,128],[149,128,135,327,126,124,124,126,128],[154,128,135,344,126,124,124,126,128],0.6,[159,128,135,344,126,124,124,126,128],[164,128,135,347,126,124,124,126,128],0.5,[220,128,135,228,126,124,124,126,128],[327,128,135,228,126,124,124,126,128],[332,128,135,351,126,124,124,126,128],1.8,[353],"other: reported only as > 20 s",[248],[356],"one NVIDIA H100 GPU",[],[359,360],"Camera pose estimation with 10 random frames per scene, AUC@30 combining relative rotation and translation accuracy; no method trained on RealEstate10K; per-method time column not extracted; values from arXiv v1; the CVF version prints 83.4 for FLARE on CO3Dv2 and marks MV-DUSt3R's CO3Dv2 value as not trained on CO3D","Camera pose estimation with 10 random frames per scene; approximate per-method time column (values printed with ~), measured on one H100 GPU; one time per method, not split by dataset",{"slug":362,"group":363,"sourceId":5,"sourceLabel":6,"table":364,"selfRows":140,"metrics":365,"seqs":369,"entrants":373,"cells":388,"outcomes":433,"locators":434,"hardware":435,"wordings":436,"notes":437},"vggt2025-table-2","vggt2025:Table 2","Table 2",[366,367,368],{"label":182,"unit":94,"statistic":94,"alignment":94},{"label":184,"unit":94,"statistic":94,"alignment":94},{"label":186,"unit":94,"statistic":94,"alignment":94},[370],{"dataset":371,"sequence":372,"environment":193},"DTU","evaluation scenes",[374,376,378,380,382,383,385,386],{"name":375,"methodId":77,"linkable":68,"proposed":68,"self":68},"Gipuma",{"name":377,"methodId":77,"linkable":68,"proposed":68,"self":68},"MVSNet",{"name":379,"methodId":77,"linkable":68,"proposed":68,"self":68},"CIDER",{"name":381,"methodId":77,"linkable":68,"proposed":68,"self":68},"PatchmatchNet",{"name":199,"methodId":200,"linkable":121,"proposed":68,"self":68},{"name":384,"methodId":77,"linkable":68,"proposed":68,"self":68},"GeoMVSNet",{"name":196,"methodId":197,"linkable":121,"proposed":68,"self":68},{"name":387,"methodId":5,"linkable":121,"proposed":121,"self":121},"Ours",[389,391,392,394,396,398,400,402,404,406,407,409,410,412,414,416,418,420,422,424,426,428,430,431],[124,124,124,390,126,124,126,126,124],0.283,[124,128,124,230,126,124,126,126,124],[124,135,124,393,126,124,126,126,124],0.578,[128,124,124,395,126,124,126,126,124],0.396,[128,128,124,397,126,124,126,126,124],0.527,[128,135,124,399,126,124,126,126,124],0.462,[135,124,124,401,126,124,126,126,124],0.417,[135,128,124,403,126,124,126,126,124],0.437,[135,135,124,405,126,124,126,126,124],0.427,[140,124,124,405,126,124,126,126,124],[140,128,124,408,126,124,126,126,124],0.377,[140,135,124,401,126,124,126,126,124],[83,124,124,411,126,124,126,126,124],0.403,[83,128,124,413,126,124,126,126,124],0.344,[83,135,124,415,126,124,126,126,124],0.374,[149,124,124,417,126,124,126,126,124],0.331,[149,128,124,419,126,124,126,126,124],0.259,[149,135,124,421,126,124,126,126,124],0.295,[154,124,124,423,126,124,126,126,128],2.677,[154,128,124,425,126,124,126,126,128],0.805,[154,135,124,427,126,124,126,126,128],1.741,[159,124,124,429,126,124,126,126,128],0.389,[159,128,124,415,126,124,126,126,128],[159,135,124,432,126,124,126,126,128],0.382,[],[364],[],[],[438,439],"Dense MVS estimation on DTU; method uses known GT cameras; units not stated in the paper","Dense MVS estimation on DTU; method does not know GT cameras; units not stated in the paper",[],1790510666045]