1<!DOCTYPE html> 2<html> 3<head> 4 <meta charset="utf-8"> 5 <!-- Meta tags for social media banners, these should be filled in appropriatly as they are your "business card" --> 6 <!-- Replace the content tag with appropriate information --> 7 <meta name="description" content="DESCRIPTION META TAG"> 8 <meta property="og:title" content="SOCIAL MEDIA TITLE TAG"/> 9 <meta property="og:description" content="SOCIAL MEDIA DESCRIPTION TAG TAG"/> 10 <meta property="og:url" content="URL OF THE WEBSITE"/> 11 <!-- Path to banner image, should be in the path listed below. Optimal dimenssions are 1200X630--> 12 <meta property="og:image" content="static/image/your_banner_image.png" /> 13 <meta property="og:image:width" content="1200"/> 14 <meta property="og:image:height" content="630"/> 15 16 17 <meta name="twitter:title" content="TWITTER BANNER TITLE META TAG"> 18 <meta name="twitter:description" content="TWITTER BANNER DESCRIPTION META TAG"> 19 <!-- Path to banner image, should be in the path listed below. Optimal dimenssions are 1200X600--> 20 <meta name="twitter:image" content="static/images/your_twitter_banner_image.png"> 21 <meta name="twitter:card" content="summary_large_image"> 22 <!-- Keywords for your paper to be indexed by--> 23 <meta name="keywords" content="KEYWORDS SHOULD BE PLACED HERE"> 24 <meta name="viewport" content="width=device-width, initial-scale=1"> 25 26 27 <title>Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation</title> 28 <!-- <link rel="icon" type="image/x-icon" href="static/images/favicon.ico"> --> 29 <link href="https://fonts.googleapis.com/css?family=Google+Sans|Noto+Sans|Castoro" 30 rel="stylesheet"> 31 32 <link rel="stylesheet" href="static/css/bulma.min.css"> 33 <link rel="stylesheet" href="static/css/bulma-carousel.min.css"> 34 <link rel="stylesheet" href="static/css/bulma-slider.min.css"> 35 <link rel="stylesheet" href="static/css/fontawesome.all.min.css"> 36 <link rel="stylesheet" 37 href="https://cdn.jsdelivr.net/gh/jpswalsh/academicons@1/css/academicons.min.css"> 38 <link rel="stylesheet" href="static/css/index.css"> 39 40
40<script src="https://ajax.googleapis.com/ajax/libs/jquery/3.5.1/jquery.min.js"></script>
40 41
41<script src="https://documentcloud.adobe.com/view-sdk/main.js"></script>
41 42
42<script defer src="static/js/fontawesome.all.min.js"></script>
42 43
43<script src="static/js/bulma-carousel.min.js"></script>
43 44
44<script src="static/js/bulma-slider.min.js"></script>
44 45
45<script src="static/js/index.js"></script>
45 46</head> 47<body> 48 49 50 <section class="hero"> 51 <div class="hero-body"> 52 <div class="container is-max-desktop"> 53 <div class="columns is-centered"> 54 <div class="column has-text-centered"> 55 <h1 class="title is-1 publication-title">Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation</h1> 56 <div class="is-size-5 publication-authors"> 57 <!-- Paper authors --> 58 <span class="author-block"> 59 <a href="https://dominikvincent.github.io/" target="_blank">Dominik Hollidt</a><sup>*â </sup>,</span> 60 <span class="author-block"> 61 <a href="https://clintonjwang.github.io/" target="_blank">Clinton Wang</a><sup>*</sup>,</span> 62 <a href="https://people.csail.mit.edu/polina/" target="_blank">Polina Golland</a><sup>*</sup>,</span> 63 <span class="author-block"> 64 <a href="https://people.inf.ethz.ch/pomarc/" target="_blank">Marc Pollefeys</a><sup>â </sup> 65 </span> 66 </div> 67 68 <div class="is-size-5 publication-authors"> 69 <span class="author-block"><sup>*</sup>MIT</span> 70 <span class="author-block"><sup>â </sup>ETH Zürich</span><br> 71 <!-- <span>Conferance name and year</span> --> 72 <!-- <span class="eql-cntrb"><small><br><sup>*</sup>Indicates Equal Contribution</small></span> --> 73 </div> 74 75 <div class="column has-text-centered"> 76 <div class="publication-links"> 77 <!-- Arxiv PDF link --> 78 <span class="link-block"> 79 <a href="https://arxiv.org/pdf/2310.05133.pdf" target="_blank" 80 class="external-link button is-normal is-rounded is-dark"> 81 <span class="icon"> 82 <i class="fas fa-file-pdf"></i> 83 </span> 84 <span>Paper</span> 85 </a> 86 </span> 87 88 <!-- Supplementary PDF link --> 89 <!-- <span class="link-block"> 90 <a href="static/pdfs/supplementary_material.pdf" target="_blank" 91 class="external-link button is-normal is-rounded is-dark"> 92 <span class="icon"> 93 <i class="fas fa-file-pdf"></i> 94 </span> 95 <span>Supplementary</span> --> 96 </a> 97 </span> 98 99 <!-- Github link --> 100 <span class="link-block"> 101 <a href="https://github.com/DominikVincent/GeometryAwareField2FieldTransformations" target="_blank" 102 class="external-link button is-normal is-rounded is-dark"> 103 <span class="icon"> 104 <i class="fab fa-github"></i> 105 </span> 106 <span>Code</span> 107 </a> 108 </span> 109 110 <!-- ArXiv abstract Link --> 111 <span class="link-block"> 112 <a href="https://arxiv.org/abs/2310.05133" target="_blank" 113 class="external-link button is-normal is-rounded is-dark"> 114 <span class="icon"> 115 <i class="ai ai-arxiv"></i> 116 </span> 117 <span>arXiv</span> 118 </a> 119 </span> 120 </div> 121 </div> 122 </div> 123 </div> 124 </div> 125 </div> 126</section> 127 128 129<!-- Teaser video--> 130<!-- <section class="hero teaser"> 131 <div class="container is-max-desktop"> 132 <div class="hero-body"> 133 <video poster="" id="tree" autoplay controls muted loop height="100%"> 134 Your video here 135 <source src="static/videos/banner_video.mp4" 136 type="video/mp4"> 137 </video> 138 <h2 class="subtitle has-text-centered"> 139 Aliquam vitae elit ullamcorper tellus egestas pellentesque. Ut lacus tellus, maximus vel lectus at, placerat pretium mi. Maecenas dignissim tincidunt vestibulum. Sed consequat hendrerit nisl ut maximus. 140 </h2> 141 </div> 142 </div> 143</section> --> 144<!-- End teaser video --> 145 146<!-- Paper abstract --> 147<section class="section hero is-light"> 148 <div class="container is-max-desktop"> 149 <div class="columns is-centered has-text-centered"> 150 <div class="column is-four-fifths"> 151 <h2 class="title is-3">Abstract</h2> 152 <div class="content has-text-justified"> 153 <p> 154 We present a novel approach to perform 3D semantic segmentation solely from 2D supervision by leveraging Neural Radiance Fields (NeRFs). By extracting features along a surface point cloud, we achieve a compact representation of the scene which is sample-efficient and conducive to 3D reasoning. Learning this feature space in an unsupervised manner via masked autoencoding enables few-shot segmentation. Our method is agnostic to the scene parameterization, working on scenes fit with any type of NeRF. 155 </p> 156 </div> 157 </div> 158 </div> 159 </div> 160</section> 161<!-- End paper abstract --> 162 163 164<!-- Image carousel --> 165<section class="hero is-small"> 166 <div class="hero-body"> 167 <div class="container"> 168 <div id="results-carousel" class="carousel results-carousel"> 169 <div class="item"> 170 <!-- Your image here --> 171 <img src="static/images/method.png" alt="MY ALT TEXT"/> 172 <h2 class="subtitle has-text-centered"> 173 A visualization of our method. (1) We shoot rays from several poses into the scene and sample points according to the proposal network of the NeRF. (2) We obtain a point cloud capturing the surface geometry of the scene by applying scene boundary filtering, density filtering, surface sampling and ground removal. (3) The point cloud gets segmented via a semantic segmentation network and integrated via volumetric rendering (4) The network is trained with the 2D semantic maps and cross-entropy loss. 174 </h2> 175 </div> 176 <div class="item"> 177 <!-- Your image here --> 178 <img src="static/images/klevr_res.png" alt="MY ALT TEXT"/> 179 <h2 class="subtitle has-text-centered"> 180 181 </h2> 182 </div> 183 <div class="item" style="text-align: center;"> 184 <img src="static/images/klevr_quant.png" alt="MY ALT TEXT" class="bar-chart"> 185 <h2 class="subtitle has-text-centered"> 186 The SPT outperforms all other methods on KLEVR. PointNet++ and Custom are significantly inferior. * indicates the use of S3DIS pretrained weights. 187 </h2> 188 </div> 189 190 <div class="item"> 191 <!-- Your image here --> 192 <img src="static/images/toybox5_res.png" alt="MY ALT TEXT"/> 193 <h2 class="subtitle has-text-centered"> 194 KLEVR qualitative predictions: PointNet++ and the plain transformer 195 produces visually less good results. The SPT performs the best but the Field Head yields smoother predictions. 196 </h2>
197 </div> 198 <div class="item" style="text-align: center;"> 199 <!-- Your image here --> 200 <img src="static/images/toybox_quant.png" class="bar-chart" alt="MY ALT TEXT"/> 201 <h2 class="subtitle has-text-centered"> 202 The SPT outperforms all other methods on ToyBox5. PointNet++ and Custom are significantly inferior. * indicates the use of S3DIS pretrained weights. 203 </h2> 204 </div> 205 </div> 206</div> 207</div> 208</section> 209<!-- End image carousel --> 210 211<section class="section hero is-light"> 212 <div class="container is-max-desktop"> 213 <div class="columns is-centered has-text-centered"> 214 <div class="column is-four-fifths"> 215 <h2 class="title is-3">Pretraining</h2> 216 <div class="content has-text-justified"> 217 <p> 218 We utilize pretraining in data scarce scenarios to reduce the number of training data required. Within the pretraining an auto encoder structure is used to recover the rgb values or normals of masked points. Pretraining on normals can bootstrap the accuracy on the downstream task of semantic segmentation. 219 </p> 220 </div> 221 </div> 222 </div> 223 </div> 224</section> 225 226<section class="hero is-small"> 227 <div class="hero-body"> 228 <div class="container"> 229 <div id="results-carousel" class="carousel results-carousel"> 230 <div class="item"> 231 Your image here 232 <img src="static/images/masked_auto_encoding.png" alt="MY ALT TEXT"/> 233 <h2 class="subtitle has-text-centered"> 234 In the pretraining masked autoencoding stage part of the 235 point cloud gets masked and properties of it have to be recovered 236 from features extracted from the non-masked point cloud and just 237 the position of the masked point cloud. 238 </h2> 239 </div> 240 <div class="item" style="text-align: center;"> 241 Your image here 242 <img src="static/images/klevr_pretraining.png" class="bar-chart" alt="MY ALT TEXT"/> 243 <h2 class="subtitle has-text-centered"> 244 Pretraining has a positive effect for the SPT on KLEVR 245 where only 10% of the data is used. Normal pretraining particularly boosts the accuracy. 246 </h2> 247 </div> 248 <div class="item" style="text-align: center;"> 249 Your image here 250 <img src="static/images/toybox_100_10_pretrain.png" class="bar-chart" alt="MY ALT TEXT"/> 251 <h2 class="subtitle has-text-centered"> 252 Using only 20% of the training scenes and 10 images per scene shows that RGB pretraining harms the downstream task's performance significantly. Normal pretraining has the same effect as 253 using S3DIS pretrained weights. 254 </h2> 255 </div> 256 <div class="item" style="text-align: center;"> 257 Your image here 258 <img src="static/images/toybox_100_270_pretrain.png" class="bar-chart" alt="MY ALT TEXT"/> 259 <h2 class="subtitle has-text-centered"> 260 Using only 20% of the training scenes shows that RGB pretraining harms the downstream task's performance significantly. Normal pretraining has the same effect as 261 using S3DIS pretrained weights. 262 </h2> 263 </div> 264 </div> 265</div> 266</div> 267</section> 268 269<section class="section hero is-light"> 270 <div class="container is-max-desktop"> 271 <div class="columns is-centered has-text-centered"> 272 <div class="column is-four-fifths"> 273 <h2 class="title is-3">Ablation Studies</h2> 274 <div class="content has-text-justified"> 275 <p> 276 Within ablation studies we investigate the impact of the design decisions (ground removal, proximity loss, surface sampling) and show that pretraining on normals provides a more accurate normal estimation. 277 </p> 278 </div> 279 </div> 280 </div> 281 </div> 282</section> 283 284 285 286<section class="hero is-small"> 287 <div class="hero-body"> 288 <div class="container"> 289 <div id="results-carousel" class="carousel results-carousel"> 290 <div class="item" style="text-align: center;"> 291 <!-- Your image here --> 292 <img src="static/images/ground_removal.png" class="bar-chart" alt="MY ALT TEXT"/> 293 <h2 class="subtitle has-text-centered"> 294 Ground removal is benefitial in accuracy and reduces the number of points significantly. 295 </h2> 296 </div> 297 <div class="item" style="text-align: center;"> 298 <!-- Your image here --> 299 <img src="static/images/surface_sampling.png" class="bar-chart" alt="MY ALT TEXT"/> 300 <h2 class="subtitle has-text-centered">
301 Surface sampling can be benefitial in accuracy and reduces the number of points significantly. 302 </h2> 303 </div> 304 <div class="item" style="text-align: center;"> 305 <!-- Your image here --> 306 <img src="static/images/proximity_loss.png" class="bar-chart" alt="MY ALT TEXT"/> 307 <h2 class="subtitle has-text-centered"> 308 The proximity loss improves the accuracy. 309 </h2> 310 </div> 311 <div class="item" style="text-align: center;"> 312 <!-- Your image here --> 313 <img src="static/images/normals.png" class="bar-chart" alt="MY ALT TEXT"/> 314 <h2 class="subtitle has-text-centered"> 315 The model pretrained on normals predicts more accurate normals than the underlying NeRF. 316 </h2> 317 </div> 318 </div> 319</div> 320</div> 321</section> 322 323<!-- Small text --> 324 325 326 327<!-- Youtube video --> 328<!-- <section class="hero is-small is-light"> 329 <div class="hero-body"> 330 <div class="container"> 331 Paper video. 332 <h2 class="title is-3">Video Presentation</h2> 333 <div class="columns is-centered has-text-centered"> 334 <div class="column is-four-fifths"> 335 336 <div class="publication-video"> 337 Youtube embed code here 338 <iframe src="https://www.youtube.com/embed/JkaxUblCGz0" frameborder="0" allow="autoplay; encrypted-media" allowfullscreen></iframe> 339 </div> 340 </div> 341 </div> 342 </div> 343 </div> 344</section> --> 345<!-- End youtube video --> 346 347 348<!-- Video carousel --> 349<!-- <section class="hero is-small"> 350 <div class="hero-body"> 351 <div class="container"> 352 <h2 class="title is-3">Another Carousel</h2> 353 <div id="results-carousel" class="carousel results-carousel"> 354 <div class="item item-video1"> 355 <video poster="" id="video1" autoplay controls muted loop height="100%"> 356 Your video file here 357 <source src="static/videos/carousel1.mp4" 358 type="video/mp4"> 359 </video> 360 </div> 361 <div class="item item-video2"> 362 <video poster="" id="video2" autoplay controls muted loop height="100%"> 363 Your video file here 364 <source src="static/videos/carousel2.mp4" 365 type="video/mp4"> 366 </video> 367 </div> 368 <div class="item item-video3"> 369 <video poster="" id="video3" autoplay controls muted loop height="100%">\ 370 Your video file here 371 <source src="static/videos/carousel3.mp4" 372 type="video/mp4"> 373 </video> 374 </div> 375 </div> 376 </div> 377 </div> 378</section> --> 379<!-- End video carousel --> 380 381 382 383 384 385 386<!-- Paper poster --> 387<!-- <section class="hero is-small is-light"> 388 <div class="hero-body"> 389 <div class="container"> 390 <h2 class="title">Poster</h2> 391 392 <iframe src="static/pdfs/sample.pdf" width="100%" height="550"> 393 </iframe> 394 395 </div> 396 </div> 397 </section> --> 398<!--End paper poster --> 399 400 401<!--BibTex citation --> 402 <section class="section" id="BibTeX"> 403 <div class="container is-max-desktop content"> 404 <h2 class="title">BibTeX</h2> 405 <pre><code>@misc{hollidt2023geometry, 406 title={Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation}, 407 author={Dominik Hollidt and Clinton Wang and Polina Golland and Marc Pollefeys}, 408 year={2023}, 409 eprint={2310.05133}, 410 archivePrefix={arXiv}, 411 primaryClass={cs.CV} 412}</code></pre> 413 </div> 414</section> 415<!--End BibTex citation --> 416 417 418 <footer class="footer"> 419 <div class="container"> 420 <div class="columns is-centered"> 421 <div class="column is-8"> 422 <div class="content"> 423 424 <p> 425 This page was built using the <a href="https://github.com/eliahuhorwitz/Academic-project-page-template" target="_blank">Academic Project Page Template</a> which was adopted from the <a href="https://nerfies.github.io" target="_blank">Nerfies</a> project page. 426 You are free to borrow the of this website, we just ask that you link back to this page in the footer. <br> This website is licensed under a <a rel="license" href="http://creativecommons.org/licenses/by-sa/4.0/" target="_blank">Creative 427 Commons Attribution-ShareAlike 4.0 International License</a>. 428 </p> 429 430 </div> 431 </div> 432 </div> 433 </div> 434</footer> 435 436<!-- Statcounter tracking code --> 437 438<!-- You can add a tracker to track page visits by creating an account at statcounter.com --> 439 440 <!-- End of Statcounter Code --> 441 442 </body> 443 </html>
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.