PageSourceSearch

https://egoallo.github.io/

html egoallo.github.io collected 2026-10-03 09:08:15 UTC 18,936 bytes, 559 lines download raw bytes

1<!doctype html>
2<html lang="en">
3  <head>
4    <title>EgoAllo</title>
5    <meta
6      content="We use egocentric SLAM poses and images to estimate 3D human body pose, height, and hands in the world."
7      name="description"
8    />
9
10    <!-- Usual metadata. -->
11    <meta charset="UTF-8" />
12    <link href="./favicon.png" rel="shortcut icon" type="image/x-icon" />
13    <meta
14      name="viewport"
15      content="width=device-width, initial-scale=1, minimum-scale=1"
16    />
17
18    <!-- Open Graph / Facebook. -->
19    <meta property="og:title" content="EgoAllo" />
20    <meta
21      property="og:description"
22      content="We use egocentric SLAM poses and images to estimate 3D human body pose, height, and hands in the world."
23    />
24    <meta property="og:type" content="website" />
25
26    <!-- Twitter. -->
27    <meta property="twitter:title" content="EgoAllo" />
28    <meta
29      property="twitter:description"
30      content="We use egocentric SLAM poses and images to estimate 3D human body pose, height, and hands in the world."
31    />
32
33    <!-- Webfont. -->
34    <link rel="preconnect" href="https://fonts.googleapis.com" />
35    <link rel="preconnect" href="https://fonts.gstatic.com" crossorigin />
36    <link
37      href="https://fonts.googleapis.com/css2?family=Inter:ital,opsz,wght@0,14..32,100..900;1,14..32,100..900&display=swap"
38      rel="stylesheet"
39    />
40
41    <!-- Styles. -->
42    <link
43      href="https://cdn.jsdelivr.net/npm/[email protected]/modern-normalize.min.css"
44      rel="stylesheet"
45    />
46    <link
47      href="https://cdnjs.cloudflare.com/ajax/libs/tabler-icons/3.19.0/tabler-icons-outline.min.css"
48      rel="stylesheet"
49    />
50    <link href="style.css" rel="stylesheet" type="text/css" />
51
52    <!-- KaTeX -->
53    <link
54      rel="stylesheet"
55      href="https://cdn.jsdelivr.net/npm/[email protected]/dist/katex.min.css"
56      integrity="sha384-nB0miv6/jRmo5UMMR1wu3Gz6NLsoTkbqJghGIsx//Rlm+ZU03BU6SQNC66uf4l5+"
57      crossorigin="anonymous"
58    />
59    
59<script
60      defer
61      src="https://cdn.jsdelivr.net/npm/[email protected]/dist/katex.min.js"
62      integrity="sha384-7zkQWkzuo3B5mTepMUcHkMB5jZaolc2xDwL6VFqjFALcbeS9Ggm/Yr2r3Dy4lfFg"
63      crossorigin="anonymous"
64    ></script>
64
65    
65<script
66      defer
67      src="https://cdn.jsdelivr.net/npm/[email protected]/dist/contrib/auto-render.min.js"
68      integrity="sha384-43gviWU0YVjaDtb/GhzOouOXtZMP/7XUzwPTstBeZFe/+rCMvRwr4yROQP43s0Xk"
69      crossorigin="anonymous"
70    ></script>
70
71    
71<script>
72      document.addEventListener("DOMContentLoaded", function () {
73        renderMathInElement(document.body, {
74          // customised options
75          // • auto-render specific keys, e.g.:
76          delimiters: [
77            { left: "$$", right: "$$", display: true },
78            { left: "$", right: "$", display: false },
79            { left: "\\(", right: "\\)", display: false },
80            { left: "\\[", right: "\\]", display: true },
81          ],
82          // • rendering keys, e.g.:
83          throwOnError: false,
84        });
85      });
86    </script>
86
87
88    <!-- Google tag (gtag.js) -->
vendor: 80 bytes, lines 88-91
88
89    <script
90      async
91      src="https://www.googletagmanager.com/gtag/js?id=
91G-E49FXX81PL
vendor: 21 bytes, lines 91-93
91"
92    ></script>
93    
93<script>
94      
vendor: 163 bytes, lines 94-100
94window.dataLayer = window.dataLayer || [];
95      function gtag() {
96        dataLayer.push(arguments);
97      }
98      gtag("js", new Date());
99
100      gtag("config", "
100G-E49FXX81PL
vendor: 8 bytes, lines 100-101
100");
101    
101</script>
101
102  </head>
103  <body>
104    <!-- Title. We tweak the wrapping behaviors a bit. -->
105    <div style="height: 1em"></div>
106    <h1 style="padding: 0 1em">
107      <!-- &#8209; is a non-breaking hyphen. -->
108      Estimating Body and Hand&nbsp;Motion in an Ego&#8209;sensed&nbsp;World
109    </h1>
110
111    <!-- Authors. -->
112    <div id="author-list">
113      <a href="https://brentyi.github.io"> Brent Yi<sup>1</sup> </a>
114      <a href="https://www.linkedin.com/in/vickie-ye-0b4b6888/" target="_blank">
115        Vickie Ye<sup>1</sup>
116      </a>
117      <a href="https://www.linkedin.com/in/maya-zheng/" target="_blank">
118        Maya Zheng<sup>1</sup>
119      </a>
120      <a href="https://annie-liyunqi.github.io" target="_blank">
121        Yunqi Li<sup>2</sup>
122      </a>
123      <a href="https://muelea.github.io/" target="_blank">
124        Lea M&uuml;ller<sup>1</sup>
125      </a>
126      <a href="https://geopavlakos.github.io/" target="_blank">
127        Georgios Pavlakos<sup>3</sup>
128      </a>
129      <a href="https://people.eecs.berkeley.edu/~yima/" target="_blank"
130        >Yi Ma<sup>1</sup></a
131      >
132      <a href="https://people.eecs.berkeley.edu/~malik/" target="_blank">
133        Jitendra Malik<sup>1</sup>
134      </a>
135      <a href="https://people.eecs.berkeley.edu/~kanazawa/" target="_blank">
136        Angjoo Kanazawa<sup>1</sup>
137      </a>
138    </div>
139    <div style="height: 0.5em"></div>
140
141    <!-- Affiliations -->
142    <div id="affiliations" style="color: #777">
143      <div><sup>1</sup>&nbsp;UC Berkeley</div>
144      <div><sup>2</sup>&nbsp;ShanghaiTech</div>
145      <div><sup>3</sup>&nbsp;UT Austin</div>
146    </div>
147    <div style="height: 0.75em"></div>
148
149    <!-- Conference -->
150    <div id="" style="text-align: center; font-weight: 500">
151      CVPR 2025 (Highlight)
152    </div>
153    <div style="height: 1.25em"></div>
154
155    <!-- Links -->
156    <div
157      style="
158        display: flex;
159        justify-content: center;
160        gap: 0.75em;
161        flex-wrap: wrap;
162      "
163    >
164      <a href="https://arxiv.org/abs/2410.03665" target="_blank">
165        <button>
166          <i class="ti ti-article"></i>
167          arXiv
168        </button>
169      </a>
170      <a href="./EgoAllo_December2024.pdf">
171        <button>
172          <i class="ti ti-file-type-pdf"></i>
173          Paper
174        </button>
175      </a>
176      <a href="https://github.com/brentyi/egoallo" target="_blank">
177        <button>
178          <i class="ti ti-brand-github"></i>
179          Code
180        </button>
181      </a>
182      <a href="./results.html" target="_blank">
183        <button
184          style="
185            background: #8e2de2;
186            /* background: linear-gradient(to right, #8e2de2, #4a00e0); */
187            box-shadow: 0 0 0.75em 0 rgba(255, 255, 50, 1);
188          "
189        >
190          <i class="ti ti-hand-click"></i>
191          Interactive Results
192        </button>
193      </a>
194    </div>
195    <div style="height: 0.5em"></div>
196
197    <!-- tldr -->
198    <section class="wide">
199      <p style="text-align: center; max-width: 30em">
200        <strong>TLDR;</strong>
201        We use egocentric (
202        <i class="ti ti-eyeglass" style="font-weight: 600"></i>
203        ) SLAM poses and images to estimate the wearer's body pose, height, and
204        hands.
205      </p>
206      <div style="height: 0.5em"></div>
207      <video
208        src="./egoallo_overview.mp4"
209        width="100%"
210        controls
211        muted
212        style="box-shadow: 0 0 1em rgba(0, 0, 0, 0.07)"
213      ></video>
214    </section>
215    <section>
216      <h2>Highlights</h2>
217
218      <p>
219        We cast estimation as sampling from a conditional diffusion model. Paper
220        highlights include:
221      </p>
222      <ol>
223        <li>formulating desirable invariance properties for conditioning,</li>
224        <li>
225          validating a new conditioning parameterization that achieves these
226          properties,
227        </li>
228        <li>
229          a Levenberg-Marquardt guidance optimizer for incorporating visual hand
230          observations.
231        </li>
232      </ol>
233      <p>Interactive results:</p>
234      <a
235        href="./results.html"
236        target="_blank"
237        style="position: relative; display: block; overflow: hidden"
238      >
239        <video
240          src="./interactive_demo.mp4"
241          style="opacity: 50%; display: block"
242          width="100%"
243          loop
244          autoplay
245          playsinline
246          muted
247          id="interactive-demo-video"
248        ></video>
249        <button
250          style="
251            position: absolute;
252            top: 50%;
253            left: 50%;
254            transform: translate(-50%, -50%);
255            background: #8e2de2;
256            /* background: linear-gradient(to right, #8e2de2, #4a00e0); */
257            box-shadow: 0 0 2em 0 rgba(255, 255, 50, 1);
258            whitespace: nowrap;
259          "
260        >
261          <i class="ti ti-hand-click"></i>
262          Open Interactive Results
263        </button>
264      </a>
265      
265<script>
266        // Make this video play slower.
267        document.addEventListener("DOMContentLoaded", function () {
268          var video = document.getElementById("interactive-demo-video");
269          video.playbackRate = 0.6;
270        });
271      </script>
271
272      <p style="opacity; 0.8">
273        Visualized scenes are outputs from
274        <a href="https://nerf.studio">Nerfstudio</a>
275        and
276        <a
277          href="https://facebookresearch.github.io/projectaria_tools/docs/ARK/mps"
278        >
279          Project Aria MPS</a
280        >.
281      </p>
282    </section>
283
284    <section>
285      <h2>Method</h2>
286      <p>
287        Our system, EgoAllo, uses <em>ego</em>centric observations to estimate
288        the wearer of a head-mounted device's actions in the
289        <em>allo</em>centric scene coordinate frame. To summarize, we:
290      </p>
291      <ol>
292        <li>
293          Train a human motion and height prior conditioned on head motion.
294        </li>
295        <li>
296          Guide sampling from the prior to align with visual hand observations.
297        </li>
298      </ol>
299      <img src="./images/method.svg" style="width: 100%; margin: 1em 0" />
300    </section>
301
302    <section>
303      <h2>Invariant Conditioning for Learning</h2>
304      <p>
305        Our main insight for improving estimation is in conditioning
306        representation.
307      </p>
308      <p>
309        We aim to condition a human motion prior on head motion. Naively
310        conditioning on absolute poses, however, would introduce sensitivity to
311        arbitrary world frame choices. Consider these trajectories, which have
312        identical local body motion but completely different absolute head
313        poses:
314      </p>
315      <div style="margin: 2em 0; text-align: center">
316        <img
317          src="./images/conditioning/top0.png"
318          style="width: 50%; vertical-align: top"
319          class="full-width-on-mobile"
320        />
321        <img
322          src="./images/conditioning/top2.png"
323          style="width: 35%; vertical-align: top"
324          class="full-width-on-mobile"
325        />
326      </div>
327      <p>
328        Training a model using these poses as conditioning would result in poor
329        generalization, as inputs become susceptible to infinite possible world
330        frame shifts.
331      </p>
332      <p>
333        Prior works have solved this by aligning trajectories with their first
334        frame. We observe, however, that canonicalizing sequences this way leads
335        to sensitivity to
336        <em>time</em>. Consider two slices of the same motion:
337      </p>
338      <div style="margin: 2em 0; text-align: center">
339        <!-- I lower opacity here because when I rendered these I made the
340          opacity of the humans too high... need to re-render. -->
341        <img
342          src="./images/conditioning/slice0.png"
343          style="width: 40%; vertical-align: top; opacity: 0.8"
344          class="full-width-on-mobile"
345        />
346        <img
347          src="./images/conditioning/slice1.png"
348          style="width: 40%; vertical-align: top; opacity: 0.8"
349          class="full-width-on-mobile"
350        />
351      </div>
352      <p>
353        Head poses from canonicalized sequences can still differ significantly,
354        even for the same body motion <span style="color: red">(circled)</span>:
355      </p>
356      <div style="margin: 2em 0; text-align: center">
357        <img
358          src="./images/conditioning/canonical_top0_circled.png"
359          style="width: 36%; vertical-align: top"
360          class="full-width-on-mobile"
361        />
362        <img
363          src="./images/conditioning/canonical_top2_circled.png"
364          style="width: 50%; vertical-align: top"
365          class="full-width-on-mobile"
366        />
367      </div>
368      <p>
369        This also hinders generalization: networks must "re-learn" outputs for
370        each slice of the input.
371      </p>
372      <p>
373        Motivated by this, our paper proposes <strong>(1)</strong>&nbsp;spatial
374        and temporal invariance properties that are desirable for head pose
375        conditioning, and <strong>(2)</strong>&nbsp;an alternative
376        parameterization that achieves them.
377      </p>
378      <p>
379        Using the central pupil frame (CPF) to measure head motion, our
380        invariant parameterization couples relative CPF motion
381        $\Delta\mathbf{T}_\text{cpf}^t$ with <em>per-timestep</em> canonicalized
382        pose $\mathbf{T}_{\text{canonical},\text{cpf}}^t$.
383      </p>
384      <img
385        src="./images/conditioning/conditioning_annotated.svg"
386        style="width: 100%; margin: 1em 0"
387      />
388      <p>
389        These transformations have improved invariance properties over prior
390        methods, while fully defining head pose trajectories relative to the
391        floor plane.
392      </p>
393      <p>
394        Quantitatively, this explains joint position estimation error
395        differences between 5% and 18%.
396      </p>
397      <p>
398        Qualitatively, we observe better estimated motion. For example, see the
399        improvements in foot motion for this dynamic sequence:
400      </p>
401      <video
402        src="./results_videos/cond_side_by_side_smaller.mp4"
403        style="display: block"
404        width="100%"
405        controls
406        loop
407        autoplay
408        muted
409      ></video>
410      <span style="opacity: 0.5; margin: 0.75em 0; font-size: 0.8em">
411        Trajectory source: EgoExo4D, unc_soccer_09-22-23_01_27.
412      </span>
413      <p>
414        We also present qualitative results compared to ground-truth, compared
415        to baselines, and when we draw multiple samples from the diffusion
416        model:
417      </p>
418      <video
419        src="./egoallo_body_results.mp4"
420        width="100%"
421        controls
422        muted
423        style="box-shadow: 0 0 1em rgba(0, 0, 0, 0.07)"
424      ></video>
425    </section>
426
427    <section>
428      <h2>Hand Guidance</h2>
429      <p>
430        At test time, we incorporate visual hand observations via diffusion
431        guidance.
432      </p>
433      <p>
434        First, we extract hand observations using
435        <a href="https://geopavlakos.github.io/hamer/">HaMeR</a> and optionally
436        Aria's
437        <a
438          href="https://facebookresearch.github.io/projectaria_tools/docs/data_formats/mps/hand_tracking"
439          >wrist and palm estimator</a
440        >.
441      </p>
442      <p>
443        In each denoising step, we then guide sampling using both 3D and
444        reprojection-based costs. Our Levenberg-Marquardt optimizer for doing so
445        is much faster than gradient-based approaches:
446      </p>
447
448      <div style="margin: 2em 0">
449        <strong style="text-align: center; display: block; margin: 0 0 0.5em 0">
450          Guidance Optimizer Runtime, RTX 4090
451        </strong>
452        <img
453          src="./images/guidance_optimizer_timing.svg"
454          style="width: 30em; max-width: 100%; margin: 0 auto; display: block"
455        />
456      </div>
457
458      <p>
459        The optimizer is implemented in Python, with autodiffed sparse
460        Jacobians. If you're interested, see our
461        <a href="https://github.com/brentyi/egoallo">code release</a> and
462        library <a href="https://github.com/brentyi/jaxls">jaxls</a>!
463      </p>
464
465      <p>
466        With our guided sampling approach, we observe that jointly estimating
467        human hands with bodies (<span style="color: purple">purple</span>)
468        reduces ambiguities when compared to single-frame monocular estimates
469        (<span style="color: teal">blue</span>):
470      </p>
471      <img
472        src="./images/hands_vs_hamer.png"
473        style="
474          width: 100%;
475          max-width: 30em;
476          margin: 1.5em auto 2em auto;
477          display: block;
478        "
479      />
480      <p>
481        Compared to naive HaMeR, EgoAllo with head pose + HaMeR input drops
482        world-frame hand joint errors by as much as 40%.
483      </p>
484    </section>
485
486    <section>
487      <h2>Related links</h2>
488      If this problem is interesting to you, here are some papers that you might
489      enjoy!
490      <ul>
491        <li>
492          <a href="https://siplab.org/projects/AvatarPoser">
493            AvatarPoser: Articulated Full-Body Pose Tracking from Sparse Motion
494            Sensing
495          </a>
496        </li>
497        <li>
498          <a href="https://lijiaman.github.io/projects/egoego/">
499            Ego-Body Pose Estimation via Ego-Head Pose Estimation
500          </a>
501        </li>
502        <li>
503          <a href="https://arxiv.org/abs/2304.11118">
504            BoDiffusion: Diffusing Sparse Observations for Full-Body Human
505            Motion Synthesis
506          </a>
507        </li>
508        <li>
509          <a href="https://arxiv.org/abs/2308.06493">
510            EgoPoser: Robust Real-Time Egocentric Pose Estimation from Sparse
511            and Intermittent Observations Everywhere
512          </a>
513        </li>
514        <li>
515          <a href="https://arxiv.org/abs/2409.13426">
516            HMD<sup>2</sup>: Environment-aware Motion Generation from Single
517            Egocentric Head-Mounted Device
518          </a>
519        </li>
520      </ul>
521    </section>
522    <section>
523      <h2>Acknowledgements</h2>
524      <p>
525        We would like to thank Hongsuk Choi, Michael Taylor, Tyler Bonnen,
526        Songwei Ge, Chung Min Kim, and Justin Kerr for insightful technical
527        discussion and suggestions, as well as Jiaman Li for helpful answers to
528        questions about EgoEgo.
529      </p>
530      <p>
531        This project was funded in part by NSF:CNS-2235013 and IARPA DOI/IBC No.
532        140D0423C0035. YM acknowledges support from the joint Simons
533        Foundation-NSF DMS grant #2031899, the ONR grant N00014-22-1-2102, the
534        NSF grant #2402951, and partial support from TBSI, InnoHK, and the
535        University of Hong Kong. JM was supported by ONR MURI N00014-21-1-2801.
536        BY is supported by the National Science Foundation Graduate Research
537        Fellowship Program under Grant DGE 2146752.
538      </p>
539    </section>
540
541    <section>
542      <h2>Citation</h2>
543      <code style="overflow-x: scroll; white-space: nowrap">
544        @inproceedings{yi2025egoallo,<br />
545        &nbsp;&nbsp;&nbsp;&nbsp;title={Estimating body and hand motion in an
546        ego-sensed world},<br />
547        &nbsp;&nbsp;&nbsp;&nbsp;author={Yi, Brent and Ye, Vickie and Zheng, Maya
548        and Li, Yunqi and M{\"u}ller, Lea and Pavlakos, Georgios and Ma, Yi and
549        Malik, Jitendra and Kanazawa, Angjoo},<br />
550        &nbsp;&nbsp;&nbsp;&nbsp;booktitle={Proceedings of the Computer Vision
551        and Pattern Recognition Conference},<br />
552        &nbsp;&nbsp;&nbsp;&nbsp;pages={7072--7084},<br />
553        &nbsp;&nbsp;&nbsp;&nbsp;year={2025}<br />
554        }
555      </code>
556    </section>
557    <div style="height: 4em"></div>
558  </body>
559</html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.