PageSourceSearch

https://dominikvincent.github.io/projects/nesf/

html dominikvincent.github.io collected 2026-10-03 09:02:16 UTC 18,551 bytes, 443 lines download raw bytes

1<!DOCTYPE html>
2<html>
3<head>
4  <meta charset="utf-8">
5  <!-- Meta tags for social media banners, these should be filled in appropriatly as they are your "business card" -->
6  <!-- Replace the content tag with appropriate information -->
7  <meta name="description" content="DESCRIPTION META TAG">
8  <meta property="og:title" content="SOCIAL MEDIA TITLE TAG"/>
9  <meta property="og:description" content="SOCIAL MEDIA DESCRIPTION TAG TAG"/>
10  <meta property="og:url" content="URL OF THE WEBSITE"/>
11  <!-- Path to banner image, should be in the path listed below. Optimal dimenssions are 1200X630-->
12  <meta property="og:image" content="static/image/your_banner_image.png" />
13  <meta property="og:image:width" content="1200"/>
14  <meta property="og:image:height" content="630"/>
15
16
17  <meta name="twitter:title" content="TWITTER BANNER TITLE META TAG">
18  <meta name="twitter:description" content="TWITTER BANNER DESCRIPTION META TAG">
19  <!-- Path to banner image, should be in the path listed below. Optimal dimenssions are 1200X600-->
20  <meta name="twitter:image" content="static/images/your_twitter_banner_image.png">
21  <meta name="twitter:card" content="summary_large_image">
22  <!-- Keywords for your paper to be indexed by-->
23  <meta name="keywords" content="KEYWORDS SHOULD BE PLACED HERE">
24  <meta name="viewport" content="width=device-width, initial-scale=1">
25
26
27  <title>Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation</title>
28  <!-- <link rel="icon" type="image/x-icon" href="static/images/favicon.ico"> -->
29  <link href="https://fonts.googleapis.com/css?family=Google+Sans|Noto+Sans|Castoro"
30  rel="stylesheet">
31
32  <link rel="stylesheet" href="static/css/bulma.min.css">
33  <link rel="stylesheet" href="static/css/bulma-carousel.min.css">
34  <link rel="stylesheet" href="static/css/bulma-slider.min.css">
35  <link rel="stylesheet" href="static/css/fontawesome.all.min.css">
36  <link rel="stylesheet"
37  href="https://cdn.jsdelivr.net/gh/jpswalsh/academicons@1/css/academicons.min.css">
38  <link rel="stylesheet" href="static/css/index.css">
39
40  
40<script src="https://ajax.googleapis.com/ajax/libs/jquery/3.5.1/jquery.min.js"></script>
40
41  
41<script src="https://documentcloud.adobe.com/view-sdk/main.js"></script>
41
42  
42<script defer src="static/js/fontawesome.all.min.js"></script>
42
43  
43<script src="static/js/bulma-carousel.min.js"></script>
43
44  
44<script src="static/js/bulma-slider.min.js"></script>
44
45  
45<script src="static/js/index.js"></script>
45
46</head>
47<body>
48
49
50  <section class="hero">
51    <div class="hero-body">
52      <div class="container is-max-desktop">
53        <div class="columns is-centered">
54          <div class="column has-text-centered">
55            <h1 class="title is-1 publication-title">Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation</h1>
56            <div class="is-size-5 publication-authors">
57              <!-- Paper authors -->
58              <span class="author-block">
59                <a href="https://dominikvincent.github.io/" target="_blank">Dominik Hollidt</a><sup>*†</sup>,</span>
60                <span class="author-block">
61                  <a href="https://clintonjwang.github.io/" target="_blank">Clinton Wang</a><sup>*</sup>,</span>
62                  <a href="https://people.csail.mit.edu/polina/" target="_blank">Polina Golland</a><sup>*</sup>,</span>
63                  <span class="author-block">
64                    <a href="https://people.inf.ethz.ch/pomarc/" target="_blank">Marc Pollefeys</a><sup>†</sup>
65                  </span>
66                  </div>
67
68                  <div class="is-size-5 publication-authors">
69                    <span class="author-block"><sup>*</sup>MIT</span>
70                    <span class="author-block"><sup>†</sup>ETH Zürich</span><br>
71                      <!-- <span>Conferance name and year</span> -->
72                    <!-- <span class="eql-cntrb"><small><br><sup>*</sup>Indicates Equal Contribution</small></span> -->
73                  </div>
74
75                  <div class="column has-text-centered">
76                    <div class="publication-links">
77                         <!-- Arxiv PDF link -->
78                      <span class="link-block">
79                        <a href="https://arxiv.org/pdf/2310.05133.pdf" target="_blank"
80                        class="external-link button is-normal is-rounded is-dark">
81                        <span class="icon">
82                          <i class="fas fa-file-pdf"></i>
83                        </span>
84                        <span>Paper</span>
85                      </a>
86                    </span>
87
88                    <!-- Supplementary PDF link -->
89                    <!-- <span class="link-block">
90                      <a href="static/pdfs/supplementary_material.pdf" target="_blank"
91                      class="external-link button is-normal is-rounded is-dark">
92                      <span class="icon">
93                        <i class="fas fa-file-pdf"></i>
94                      </span>
95                      <span>Supplementary</span> -->
96                    </a>
97                  </span>
98
99                  <!-- Github link -->
100                  <span class="link-block">
101                    <a href="https://github.com/DominikVincent/GeometryAwareField2FieldTransformations" target="_blank"
102                    class="external-link button is-normal is-rounded is-dark">
103                    <span class="icon">
104                      <i class="fab fa-github"></i>
105                    </span>
106                    <span>Code</span>
107                  </a>
108                </span>
109
110                <!-- ArXiv abstract Link -->
111                <span class="link-block">
112                  <a href="https://arxiv.org/abs/2310.05133" target="_blank"
113                  class="external-link button is-normal is-rounded is-dark">
114                  <span class="icon">
115                    <i class="ai ai-arxiv"></i>
116                  </span>
117                  <span>arXiv</span>
118                </a>
119              </span>
120            </div>
121          </div>
122        </div>
123      </div>
124    </div>
125  </div>
126</section>
127
128
129<!-- Teaser video-->
130<!-- <section class="hero teaser">
131  <div class="container is-max-desktop">
132    <div class="hero-body">
133      <video poster="" id="tree" autoplay controls muted loop height="100%">
134        Your video here
135        <source src="static/videos/banner_video.mp4"
136        type="video/mp4">
137      </video>
138      <h2 class="subtitle has-text-centered">
139        Aliquam vitae elit ullamcorper tellus egestas pellentesque. Ut lacus tellus, maximus vel lectus at, placerat pretium mi. Maecenas dignissim tincidunt vestibulum. Sed consequat hendrerit nisl ut maximus. 
140      </h2>
141    </div>
142  </div>
143</section> -->
144<!-- End teaser video -->
145
146<!-- Paper abstract -->
147<section class="section hero is-light">
148  <div class="container is-max-desktop">
149    <div class="columns is-centered has-text-centered">
150      <div class="column is-four-fifths">
151        <h2 class="title is-3">Abstract</h2>
152        <div class="content has-text-justified">
153          <p>
154            We present a novel approach to perform 3D semantic segmentation solely from 2D supervision by leveraging Neural Radiance Fields (NeRFs). By extracting features along a surface point cloud, we achieve a compact representation of the scene which is sample-efficient and conducive to 3D reasoning. Learning this feature space in an unsupervised manner via masked autoencoding enables few-shot segmentation. Our method is agnostic to the scene parameterization, working on scenes fit with any type of NeRF.
155          </p>
156        </div>
157      </div>
158    </div>
159  </div>
160</section>
161<!-- End paper abstract -->
162
163
164<!-- Image carousel -->
165<section class="hero is-small">
166  <div class="hero-body">
167    <div class="container">
168      <div id="results-carousel" class="carousel results-carousel">
169      <div class="item">
170        <!-- Your image here -->
171        <img src="static/images/method.png" alt="MY ALT TEXT"/>
172        <h2 class="subtitle has-text-centered">
173          A visualization of our method. (1) We shoot rays from several poses into the scene and sample points according to the proposal network of the NeRF. (2) We obtain a point cloud capturing the surface geometry of the scene by applying scene boundary filtering, density filtering, surface sampling and ground removal. (3) The point cloud gets segmented via a semantic segmentation network and integrated via volumetric rendering (4) The network is trained with the 2D semantic maps and cross-entropy loss.
174        </h2>
175      </div>
176       <div class="item">
177        <!-- Your image here -->
178        <img src="static/images/klevr_res.png" alt="MY ALT TEXT"/>
179        <h2 class="subtitle has-text-centered">
180          
181        </h2>
182      </div>
183      <div class="item" style="text-align: center;">
184        <img src="static/images/klevr_quant.png" alt="MY ALT TEXT"  class="bar-chart">
185        <h2 class="subtitle has-text-centered">
186            The SPT outperforms all other methods on KLEVR. PointNet++ and Custom are significantly inferior. * indicates the use of S3DIS pretrained weights.
187        </h2>
188    </div>
189    
190      <div class="item">
191        <!-- Your image here -->
192        <img src="static/images/toybox5_res.png" alt="MY ALT TEXT"/>
193        <h2 class="subtitle has-text-centered">
194          KLEVR qualitative predictions: PointNet++ and the plain transformer
195          produces visually less good results. The SPT performs the best but the Field Head yields smoother predictions.
196       </h2>
197     </div>
198     <div class="item" style="text-align: center;">
199      <!-- Your image here -->
200      <img src="static/images/toybox_quant.png" class="bar-chart" alt="MY ALT TEXT"/>
201      <h2 class="subtitle has-text-centered">
202        The SPT outperforms all other methods on ToyBox5. PointNet++ and Custom are significantly inferior. * indicates the use of S3DIS pretrained weights.
203      </h2>
204    </div>
205  </div>
206</div>
207</div>
208</section>
209<!-- End image carousel -->
210
211<section class="section hero is-light">
212  <div class="container is-max-desktop">
213    <div class="columns is-centered has-text-centered">
214      <div class="column is-four-fifths">
215        <h2 class="title is-3">Pretraining</h2>
216        <div class="content has-text-justified">
217          <p>
218            We utilize pretraining in data scarce scenarios to reduce the number of training data required. Within the pretraining an auto encoder structure is used to recover the rgb values or normals of masked points. Pretraining on normals can bootstrap the accuracy on the downstream task of semantic segmentation. 
219          </p>
220        </div>
221      </div>
222    </div>
223  </div>
224</section>
225
226<section class="hero is-small">
227  <div class="hero-body">
228    <div class="container">
229      <div id="results-carousel" class="carousel results-carousel">
230      <div class="item">
231        Your image here
232        <img src="static/images/masked_auto_encoding.png" alt="MY ALT TEXT"/>
233        <h2 class="subtitle has-text-centered">
234          In the pretraining masked autoencoding stage part of the
235          point cloud gets masked and properties of it have to be recovered
236          from features extracted from the non-masked point cloud and just
237          the position of the masked point cloud.
238        </h2>
239      </div>
240       <div class="item" style="text-align: center;">
241        Your image here
242        <img src="static/images/klevr_pretraining.png" class="bar-chart" alt="MY ALT TEXT"/>
243        <h2 class="subtitle has-text-centered">
244          Pretraining has a positive effect for the SPT on KLEVR
245          where only 10% of the data is used. Normal pretraining particularly boosts the accuracy.
246        </h2>
247      </div>
248      <div class="item" style="text-align: center;">
249        Your image here
250        <img src="static/images/toybox_100_10_pretrain.png" class="bar-chart" alt="MY ALT TEXT"/>
251        <h2 class="subtitle has-text-centered">
252          Using only 20% of the training scenes and 10 images per scene shows that RGB pretraining harms the downstream task's performance significantly. Normal pretraining has the same effect as
253          using S3DIS pretrained weights.
254        </h2>
255      </div>
256      <div class="item" style="text-align: center;">
257        Your image here
258        <img src="static/images/toybox_100_270_pretrain.png" class="bar-chart" alt="MY ALT TEXT"/>
259        <h2 class="subtitle has-text-centered">
260          Using only 20% of the training scenes shows that RGB pretraining harms the downstream task's performance significantly. Normal pretraining has the same effect as
261          using S3DIS pretrained weights.
262       </h2>
263     </div>
264  </div>
265</div>
266</div>
267</section> 
268
269<section class="section hero is-light">
270  <div class="container is-max-desktop">
271    <div class="columns is-centered has-text-centered">
272      <div class="column is-four-fifths">
273        <h2 class="title is-3">Ablation Studies</h2>
274        <div class="content has-text-justified">
275          <p>
276            Within ablation studies we investigate the impact of the design decisions (ground removal, proximity loss, surface sampling) and show that pretraining on normals provides a more accurate normal estimation.
277          </p>
278        </div>
279      </div>
280    </div>
281  </div>
282</section>
283
284
285
286<section class="hero is-small">
287  <div class="hero-body">
288    <div class="container">
289      <div id="results-carousel" class="carousel results-carousel">
290      <div class="item" style="text-align: center;">
291        <!-- Your image here -->
292        <img src="static/images/ground_removal.png" class="bar-chart" alt="MY ALT TEXT"/>
293        <h2 class="subtitle has-text-centered">
294          Ground removal is benefitial in accuracy and reduces the number of points significantly.
295        </h2>
296      </div>
297       <div class="item" style="text-align: center;">
298        <!-- Your image here -->
299        <img src="static/images/surface_sampling.png" class="bar-chart" alt="MY ALT TEXT"/>
300        <h2 class="subtitle has-text-centered">
301          Surface sampling can be benefitial in accuracy and reduces the number of points significantly.
302        </h2>
303      </div>
304      <div class="item" style="text-align: center;">
305        <!-- Your image here -->
306        <img src="static/images/proximity_loss.png" class="bar-chart" alt="MY ALT TEXT"/>
307        <h2 class="subtitle has-text-centered">
308          The proximity loss improves the accuracy.
309        </h2>
310      </div>
311      <div class="item" style="text-align: center;">
312        <!-- Your image here -->
313        <img src="static/images/normals.png" class="bar-chart" alt="MY ALT TEXT"/>
314        <h2 class="subtitle has-text-centered">
315          The model pretrained on normals predicts more accurate normals than the underlying NeRF.
316       </h2>
317     </div>
318  </div>
319</div>
320</div>
321</section>
322
323<!-- Small text -->
324
325
326
327<!-- Youtube video -->
328<!-- <section class="hero is-small is-light">
329  <div class="hero-body">
330    <div class="container">
331      Paper video.
332      <h2 class="title is-3">Video Presentation</h2>
333      <div class="columns is-centered has-text-centered">
334        <div class="column is-four-fifths">
335          
336          <div class="publication-video">
337            Youtube embed code here
338            <iframe src="https://www.youtube.com/embed/JkaxUblCGz0" frameborder="0" allow="autoplay; encrypted-media" allowfullscreen></iframe>
339          </div>
340        </div>
341      </div>
342    </div>
343  </div>
344</section> -->
345<!-- End youtube video -->
346
347
348<!-- Video carousel -->
349<!-- <section class="hero is-small">
350  <div class="hero-body">
351    <div class="container">
352      <h2 class="title is-3">Another Carousel</h2>
353      <div id="results-carousel" class="carousel results-carousel">
354        <div class="item item-video1">
355          <video poster="" id="video1" autoplay controls muted loop height="100%">
356            Your video file here
357            <source src="static/videos/carousel1.mp4"
358            type="video/mp4">
359          </video>
360        </div>
361        <div class="item item-video2">
362          <video poster="" id="video2" autoplay controls muted loop height="100%">
363            Your video file here
364            <source src="static/videos/carousel2.mp4"
365            type="video/mp4">
366          </video>
367        </div>
368        <div class="item item-video3">
369          <video poster="" id="video3" autoplay controls muted loop height="100%">\
370            Your video file here
371            <source src="static/videos/carousel3.mp4"
372            type="video/mp4">
373          </video>
374        </div>
375      </div>
376    </div>
377  </div>
378</section> -->
379<!-- End video carousel -->
380
381
382
383
384
385
386<!-- Paper poster -->
387<!-- <section class="hero is-small is-light">
388  <div class="hero-body">
389    <div class="container">
390      <h2 class="title">Poster</h2>
391
392      <iframe  src="static/pdfs/sample.pdf" width="100%" height="550">
393          </iframe>
394        
395      </div>
396    </div>
397  </section> -->
398<!--End paper poster -->
399
400
401<!--BibTex citation -->
402  <section class="section" id="BibTeX">
403    <div class="container is-max-desktop content">
404      <h2 class="title">BibTeX</h2>
405      <pre><code>@misc{hollidt2023geometry,
406      title={Geometry Aware Field-to-field Transformations for 3D Semantic Segmentation}, 
407      author={Dominik Hollidt and Clinton Wang and Polina Golland and Marc Pollefeys},
408      year={2023},
409      eprint={2310.05133},
410      archivePrefix={arXiv},
411      primaryClass={cs.CV}
412}</code></pre>
413    </div>
414</section>
415<!--End BibTex citation -->
416
417
418  <footer class="footer">
419  <div class="container">
420    <div class="columns is-centered">
421      <div class="column is-8">
422        <div class="content">
423
424          <p>
425            This page was built using the <a href="https://github.com/eliahuhorwitz/Academic-project-page-template" target="_blank">Academic Project Page Template</a> which was adopted from the <a href="https://nerfies.github.io" target="_blank">Nerfies</a> project page.
426            You are free to borrow the of this website, we just ask that you link back to this page in the footer. <br> This website is licensed under a <a rel="license"  href="http://creativecommons.org/licenses/by-sa/4.0/" target="_blank">Creative
427            Commons Attribution-ShareAlike 4.0 International License</a>.
428          </p>
429
430        </div>
431      </div>
432    </div>
433  </div>
434</footer>
435
436<!-- Statcounter tracking code -->
437  
438<!-- You can add a tracker to track page visits by creating an account at statcounter.com -->
439
440    <!-- End of Statcounter Code -->
441
442  </body>
443  </html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.