PageSourceSearch

https://chihyaoma.github.io/

html chihyaoma.github.io collected 2026-10-03 08:47:19 UTC 55,132 bytes, 177 lines download raw bytes

1<!DOCTYPE html> <html lang="en"> <head> <meta name="google-site-verification" content="7jvmJDQ2vKfquua9_gEyNGt0vVblbDKJgWlT4WrxPb8" /> <meta charset="utf-8"> <meta http-equiv="X-UA-Compatible" content="IE=edge"> <meta name="viewport" content="width=device-width, initial-scale=1 maximum-scale=1; minimum-scale=1; user-scalable=no;"> <meta content="Kevin Chih-Yao Ma's Personal Website." name="description"> <meta name="keywords" content="Kevin Chih-Yao Ma"> <meta name="author" content="Kevin Chih-Yao Ma"> <title>Kevin Chih-Yao Ma</title> <!-- favicon --> <link rel="shortcut icon" href="/static/assets/img/landing/20171124-cropped-circle.png"> <!-- Main CSS --> <link href="/static/assets/app-20180125.min.css" rel="stylesheet"> <link href="/static/css/custom.css" rel="stylesheet"> <!-- Main scripts --> 
1<script src="/static/assets/app-20180125.min.js"></script>
1 
1<script async defer src="/static/js/github_api.min.js"></script>
1 <!-- Google AdSense --> 
1<script async src="//pagead2.googlesyndication.com/pagead/js/adsbygoogle.js"></script>
1 
1<script> (adsbygoogle = window.adsbygoogle || []).push({ google_ad_client: "ca-pub-6196184668650108", enable_page_level_ads: true }); </script>
1 <!-- Begin Jekyll SEO tag v2.8.0 --> <title>index | Kevin Chih-Yao Ma</title> <meta name="generator" content="Jekyll v3.10.0" /> <meta property="og:title" content="index" /> <meta name="author" content="Kevin Chih-Yao Ma" /> <meta property="og:locale" content="en_US" /> <meta name="description" content="Kevin Chih-Yao Ma’s Personal Website." /> <meta property="og:description" content="Kevin Chih-Yao Ma’s Personal Website." /> <link rel="canonical" href="https://chihyaoma.github.io/" /> <meta property="og:url" content="https://chihyaoma.github.io/" /> <meta property="og:site_name" content="Kevin Chih-Yao Ma" /> <meta property="og:type" content="website" /> <meta name="twitter:card" content="summary" /> <meta property="twitter:title" content="index" /> 
1<script type="application/ld+json"> {"@context":"https://schema.org","@type":"WebSite","author":{"@type":"Person","name":"Kevin Chih-Yao Ma"},"description":"Kevin Chih-Yao Ma’s Personal Website.","headline":"index","name":"Kevin Chih-Yao Ma","publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chihyaoma.github.io/static/assets/img/landing/20171124-cropped-circle.png"},"name":"Kevin Chih-Yao Ma"},"url":"https://chihyaoma.github.io/"}</script>
1 <!-- End Jekyll SEO tag --> </head> <body id="page-top" class="landing-page"> <div class="navbar-wrapper"> <nav class="navbar navbar-default navbar-fixed-top" role="navigation"> <div class="container"> <div class="navbar-header page-scroll"> <button type="button" class="navbar-toggle collapsed" data-toggle="collapse" data-target="#navbar" aria-expanded="false" aria-controls="navbar"> <span class="sr-only">Toggle navigation</span> <span class="icon-bar"></span> <span class="icon-bar"></span> <span class="icon-bar"></span> </button> <a class="navbar-brand" href="#page-top" id="i18_title"><span data-i18n="website.title">Kevin Chih-Yao Ma</span></a> </div> <div id="navbar" class="navbar-collapse collapse"> <ul class="nav navbar-nav navbar-right" id="i18_navbar"> <li> <a class="page-scroll" href="#about-me "> <span data-i18n="nav.about_me">About</span> </a> </li> <li> <a class="page-scroll" href="#career "> <span data-i18n="nav.career">Career</span> </a> </li> <li> <a class="page-scroll" href="#publications "> <span data-i18n="">Publications</span> </a> </li> <li> <a class="page-scroll" href="#skills "> <span data-i18n="nav.skills">Research Interests</span> </a> </li> <li> <a class="page-scroll" href="#services "> <span data-i18n="service.service">Services</span> </a> </li> <li> <a class="page-scroll" href="project/ "> <span data-i18n="nav.project">Projects</span> </a> </li> </ul> </div> </div> </nav> </div> <div id="inSlider" class="carousel carousel-fade" data-ride="carousel"> <ol class="carousel-indicators"> <li data-target="#inSlider" data-slide-to="0" class="active"></li> <li data-target="#inSlider" data-slide-to="1"></li> </ol> <div class="carousel-inner" role="listbox"> <div class="item active"> <div class="container"> <div class="carousel-caption"> </div> <div class="carousel-image wow zoomIn"> <!-- <img src="static/img/landing/laptop.png" alt="laptop"/> --> </div> </div> <!-- Set background for slide in css --> <div class="header-back" style="background-image: url(/static/assets/img/landing/header-Mount_Rainier-2.jpg);" title="banner image"> </div> </div> <div class="item"> <div class="container"> <div class="carousel-caption"> </div> <div class="carousel-image wow zoomIn"> <!-- <img src="static/img/landing/laptop.png" alt="laptop"/> --> </div> </div> <!-- Set background for slide in css --> <div class="header-back" style="background-image: url(/static/assets/img/landing/header-Mount_Rainier-1.jpg);" title="banner image"></div> </div> </div> <a class="left carousel-control" href="#inSlider" role="button" data-slide="prev"> <span class="glyphicon glyphicon-chevron-left" aria-hidden="true"></span> <span class="sr-only">Previous</span> </a> <a class="right carousel-control" href="#inSlider" role="button" data-slide="next"> <span class="glyphicon glyphicon-chevron-right" aria-hidden="true"></span> <span class="sr-only">Next</span> </a> </div> <section id="about-me" class="features " style="margin-top: 0;"> <div class="container" id="i18_about_me"> <div class="row m-b-lg"> <div class="col-lg-12 text-center"> <div class="navy-line"></div> <h1><span data-i18n="about_me.about_me">About Me</span></h1> <!-- <p>Donec sed odio dui. Etiam porta sem malesuada magna mollis euismod.</p> --> </div> </div> <div class="row"> <!-- first two icons --> <div class="col-xs-2 wow fadeInLeft"> <div class="team-member"> <div class="vote-icon" style="text-align: center"> <i class="fa icon-python"></i> </div> </div> </div> <div class="col-xs-2 wow fadeInLeft"> <div class="team-member"> <div class="vote-icon" style="text-align: center"> <i class="fa fa-linux"></i> </div> </div> </div> <!-- avatar --> <div class="col-xs-4"> <div class="team-member wow zoomIn"> <img src="/static/assets/img/landing/20171124_cropped.jpg" height="110" width="110" class="img-responsive img-circle" alt=""> <h4><span class="navy">Kevin Chih-Yao</span> Ma</h4> <ul class="list-inline social-icon"> <li><a href="https://github.com/chihyaoma" target="blank"><i class="fa fa-github"></i></a></li> <li><a href="https://www.linkedin.com/in/kevin-chih-yao-ma-9b5b3063" target="blank"><i class="fa fa-linkedin"></i></a></li> <li><a href="https://www.facebook.com/chihyaoma" target="blank"><i class="fa fa-facebook"></i></a></li> <li><a href="https://twitter.com/chihyaoma" target="blank"><i class="fa fa-twitter"></i></a></li> <li><a href="mailto:[email protected]"><i class="fa fa-envelope-o"></i></a></li> <!-- <li><a href="/feed.xml" target="blank"><i class="fa fa-rss"></i></a></li> --> </ul> </div> </div> <!-- second two icons --> <div class="col-xs-2 wow fadeInRight"> <div class="team-member"> <div class="vote-icon" style="text-align: center"> <i class="fa fa-globe"></i> </div> </div> </div> <div class="col-xs-2 wow fadeInRight"> <div class="team-member"> <div class="vote-icon" style="text-align: center"> <i class="fa icon-ubuntu"></i> </div> </div> </div> </div> <div class="row">
1 <div class="col-lg-8 col-lg-offset-2 text-justify m-t-lg m-b-lg wow zoomIn"> <p><font size="3"><span data-i18n="about_me.des">My name is Kevin Chih-Yao Ma (馬志堯). I am a Member of Technical Staff at Microsoft AI, working on native multimodal foundation models. Previously, I worked at Meta GenAI, where I built Meta's media generation models. I am notably a builder for codebase and take lead on various modeling parts in pre-training, model scaling, and post-training. My work has produced Emu, MovieGen, and various Meta AI's media generation products based on these models, including Imagine, Animation, Editing, Personalization, etc.</span></font></p> <!-- <p align="center" id="hiring_msg"> Meta's Gen AI Media Foundation team will be hiring research interns for 2024 Summer. --> <!-- <br> --> <!-- Please drop me an email if you are interested in. </p> --> </div> </div> </div> </section> <section id="career" class="features gray-section timeline" style="margin-top: 0;"> <div class="container" id="i18_career"> <div class="row"> <div class="col-lg-12 text-center"> <div class="navy-line"></div> <h1><span data-i18n="career.my_career">Career</span></h1> </div> </div> <div class="row features-block"> <div class="col-lg-12"> <div id="vertical-timeline" class="vertical-container light-timeline center-orientation"> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpRight "> <h2><span data-i18n="career.company_h">Microsoft AI</span></h2> <p><span data-i18n="career.company_h_desc">Growing the team, built training codebase, babysit model runs, work on data, work on eval, then ... <br/> release <a href="https://microsoft.ai/news/introducing-mai-image-2/">MAI-Image-2</a> that placed MAI among the top 3 labs worldwide<br /> <br /> with <a href="https://scholar.google.com/citations?user=nzEluBwAAAAJ&hl=en">Nando de Freitas</a>, <a href="https://scholar.google.com/citations?user=L7lMQkQAAAAJ&hl=en">Karén Simonyan</a>, and <a href="https://en.wikipedia.org/wiki/Mustafa_Suleyman">Mustafa Suleyman</a> </span></p> <span class="vertical-date"><span data-i18n="career.company_h_date"> June. 2025 - Present </span> <br/> <small><span data-i18n="career.company_h_job">Member of Technical Staff</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpLeft "> <h2><span data-i18n="career.company_h">Meta</span></h2> <p><span data-i18n="career.company_h_desc">A lead IC in Meta's <a href="https://ai.meta.com/research/movie-gen/">Movie Gen</a>.<br/> Co-lead in building training codebase and designing model arch, led model scaling, and co-led post-training. <br/><br/> Core contributor of <a href="https://arxiv.org/abs/2309.15807">Emu</a>
1 that powers <a href="https://ai.meta.com/blog/emu-text-to-video-generation-image-editing-research/">Emu Video/Edit</a> and <a href="https://imagine.meta.com/">Imagine</a>.<br /> <br /> with <a href="https://sites.google.com/site/vajdap/what-we-do">Peter Vajda</a> and <a href="https://research.fb.com/people/he-zijian/">Zijian He</a> (GenAI Media Foundation team) </span></p> <span class="vertical-date"><span data-i18n="career.company_h_date"> Aug. 2023 - June 2025 </span> <br/> <small><span data-i18n="career.company_h_job">Staff Research Scientist</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpRight "> <h2><span data-i18n="career.company_h">Meta</span></h2> <p><span data-i18n="career.company_h_desc">Generative Models, Federated Learning, Semi-Supervised Learning<br /><br /> with <a href="https://sites.google.com/site/vajdap/what-we-do">Peter Vajda</a> and <a href="https://research.fb.com/people/he-zijian/">Zijian He</a> (Mobile Vision & GenAI Media Foundation team) </span></p> <span class="vertical-date"><span data-i18n="career.company_h_date"> Feb. 2022 - July. 2023 </span> <br/> <small><span data-i18n="career.company_h_job">Senior Research Scientist</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpLeft "> <h2><span data-i18n="career.company_g">Meta</span></h2> <p><span data-i18n="career.company_g_desc">Data-efficient learning<br /><br /> with <a href="https://sites.google.com/site/vajdap/what-we-do">Peter Vajda</a> and <a href="https://research.fb.com/people/he-zijian/">Zijian He</a> (Mobile Vision team) </span></p> <span class="vertical-date"><span data-i18n="career.company_g_date"> Aug. 2020 - Jan. 2022 </span> <br/> <small><span data-i18n="career.company_g_job">Research Scientist</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpRight "> <h2><span data-i18n="career.company_f">Meta</span></h2> <p><span data-i18n="career.company_f_desc">Self-Supervised Learning <br /><br /> with <a href="http://rohrbach.vision/">Marcus Rohrbach</a> (FAIR), <a href="http://www.skamalas.com/">Yannis Kalantidis</a> (AML), <a href="https://wind09.github.io/">Kan Chen</a> (Mobile Vision), and <a href="https://sites.google.com/site/vajdap/what-we-do">Peter Vajda</a> (Mobile Vision) </span></p> <span class="vertical-date"><span data-i18n="career.company_f_date"> Summer & Fall 2019 </span> <br/> <small><span data-i18n="career.company_f_job">Research Intern</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpLeft "> <h2><span data-i18n="career.company_a">Salesforce Research</span></h2> <p><span data-i18n="career.company_a_desc">Vision-and-Language Navigation <br /><br /> with <a href="http://www.stat.ucla.edu/~caiming/">Caiming Xiong</a> and <a href="https://www.socher.org/">Richard Socher</a> </span></p> <span class="vertical-date"><span data-i18n="career.company_a_date"> Summer 2018 </span> <br/> <small><span data-i18n="career.company_a_job">Research Intern</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpRight "> <h2><span data-i18n="career.company_b">NEC-Labs Machine Learning</span></h2> <p><span data-i18n="career.company_b_desc">Relational reasoning for human action recognition and video captioning <br /><br /> with <a href="http://asim.ai/">Asim Kadav</a> </span></p> <span class="vertical-date"><span data-i18n="career.company_b_date">
1 Summer & Fall 2017 </span> <br/> <small><span data-i18n="career.company_b_job">Research Intern</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpLeft "> <h2><span data-i18n="career.company_c">Georgia Tech</span></h2> <p><span data-i18n="career.company_c_desc">Electrical and Computer Engineering <br /><br /> with <a href="https://ghassanalregib.info/">Ghassan AlRegib</a> (advisor) and <a href="https://www.cc.gatech.edu/~zk15/">Zsolt Kira</a> </span></p> <span class="vertical-date"><span data-i18n="career.company_c_date"> 2014 Fall - 2020 Spring </span> <br/> <small><span data-i18n="career.company_c_job">Ph.D. student</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpRight "> <h2><span data-i18n="career.company_d">National Chiao Tung University</span></h2> <p><span data-i18n="career.company_d_desc">Electrical and Computer Engineering <br /><br /> with <a href="https://www.hmhang.com/">Hsueh-Ming Hang</a> </span></p> <span class="vertical-date"><span data-i18n="career.company_d_date"> Aug. 2012 - May 2014 </span> <br/> <small><span data-i18n="career.company_d_job">Research Assistant</span></small> </span> </div> </div> <div class="vertical-timeline-block"> <div class="vertical-timeline-icon navy-bg wow rotateIn"> <i class="fa fa-plus-square"></i> </div> <div class="vertical-timeline-content wow rotateInUpLeft "> <h2><span data-i18n="career.company_e">National Chiao Tung University</span></h2> <p><span data-i18n="career.company_e_desc">Electrical and Computer Engineering </span></p> <span class="vertical-date"><span data-i18n="career.company_e_date"> Aug. 2006 - May 2011 </span> <br/> <small><span data-i18n="career.company_e_job">B.S./M.S.</span></small> </span> </div> </div> </div> </div> </div> </div> </section> <section id="publications" class="features " style="margin-top: 0;"> 
1<script type="text/javascript" src="build/hidebib.js"></script>
1 
1<script type="text/javascript"> function toggle_visibility(id) { var e = document.getElementById(id); if(e.style.display == 'block') e.style.display = 'none'; else e.style.display = 'block'; } </script>
1 <div class="container"> <div class="row m-b-lg"> <div class="col-lg-12 text-center"> <div class="navy-line"></div> <h1><span data-i18n="publication.publication">Selected Publications</span></h1> </div> </div> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="75%" valign="top" align="center"> <a href="https://scholar.google.com/citations?hl=en&user=HrrtgKkAAAAJ&view_op=list_works&sortby=pubdate"><img src="static/assets/img/landing/google-scholar-icon.png" alt="game" width="4%" style="border-style: none"></a> &emsp; Please see my <a href="https://scholar.google.com/citations?hl=en&user=HrrtgKkAAAAJ&view_op=list_works&sortby=pubdate"><heading>Google Scholar</heading></a> for complete publication list. </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <video id="vid" muted="muted" width="50%" controls="autoPlay" onended="this.currentTime = 0; this.play();" autoplay=""> <source src="static/assets/video/moviegen.mp4" type="video/mp4"> </video> </td> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Movie Gen: A Cast of Media Foundation Models </heading></font><br> Meta's Movie Gen team<br> <a href="https://ai.meta.com/research/movie-gen/">[Webpage]</a> / <a href="https://arxiv.org/abs/2410.13720">[arXiv]</a> / <a href="https://github.com/facebookresearch/MovieGenBench">[MovieGenBench (GitHub)]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('moviegen_bibtex');">bibtex</a>] <div id='moviegen_bibtex' style="display:none; font-size:small;"><pre><code>
2  @misc{polyak2024moviegencastmedia,
3    title={Movie Gen: A Cast of Media Foundation Models}, 
4    author={Adam Polyak and Amit Zohar and Andrew Brown and Andros Tjandra and Animesh Sinha and Ann Lee and Apoorv Vyas and Bowen Shi and Chih-Yao Ma and Ching-Yao Chuang and David Yan and Dhruv Choudhary and Dingkang Wang and Geet Sethi and Guan Pang and Haoyu Ma and Ishan Misra and Ji Hou and Jialiang Wang and Kiran Jagadeesh and Kunpeng Li and Luxin Zhang and Mannat Singh and Mary Williamson and Matt Le and Matthew Yu and Mitesh Kumar Singh and Peizhao Zhang and Peter Vajda and Quentin Duval and Rohit Girdhar and Roshan Sumbaly and Sai Saketh Rambhatla and Sam Tsai and Samaneh Azadi and Samyak Datta and Sanyuan Chen and Sean Bell and Sharadh Ramaswamy and Shelly Sheynin and Siddharth Bhattacharya and Simran Motwani and Tao Xu and Tianhe Li and Tingbo Hou and Wei-Ning Hsu and Xi Yin and Xiaoliang Dai and Yaniv Taigman and Yaqiao Luo and Yen-Cheng Liu and Yi-Chiao Wu and Yue Zhao and Yuval Kirstain and Zecheng He and Zijian He and Albert Pumarola and Ali Thabet and Artsiom Sanakoyeu and Arun Mallya and Baishan Guo and Boris Araya and Breena Kerr and Carleigh Wood and Ce Liu and Cen Peng and Dimitry Vengertsev and Edgar Schonfeld and Elliot Blanchard and Felix Juefei-Xu and Fraylie Nord and Jeff Liang and John Hoffman and Jonas Kohler and Kaolin Fire and Karthik Sivakumar and Lawrence Chen and Licheng Yu and Luya Gao and Markos Georgopoulos and Rashel Moritz and Sara K. Sampson and Shikai Li and Simone Parmeggiani and Steve Fine and Tara Fowler and Vladan Petrovic and Yuming Du},
5    year={2024},
6    eprint={2410.13720},
7    archivePrefix={arXiv},
8    primaryClass={cs.CV},
9    url={https://arxiv.org/abs/2410.13720}, 
10}
11</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/emu.jpg" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Emu: Enhancing Image Generation Models Using Photogenic Needles in a Haystack </heading></font><br> Xiaoliang Dai<sup>*</sup>, Ji Hou<sup>*</sup>, <strong> Chih-Yao Ma<sup>*</sup></strong>, Sam Tsai<sup>*</sup>, Jialiang Wang<sup>*</sup>, Rui Wang<sup>*</sup>, Peizhao Zhang<sup>*</sup>, Simon Vandenhende, Xiaofang Wang, Abhimanyu Dubey, Matthew Yu, Abhishek Kadian, Filip Radenovic, Dhruv Mahajan, Kunpeng Li, Yue Zhao, Vladan Petrovic, Mitesh Kumar Singh, Simran Motwani, Yi Wen, Yiwen Song, Roshan Sumbaly<sup>+</sup>, Vignesh Ramanathan<sup>+</sup>, Zijian He<sup>+</sup>, Peter Vajda<sup>+</sup>, Devi Parikh<sup>+</sup><br> (<sup>*</sup>: Core contributors: equal contribution, alphabetical order.)<br> (<sup>+</sup>: Equal last authors.)<br> <a href="https://arxiv.org/abs/2309.15807">[arXiv]</a> / <!-- <a href="https://github.com/facebookresearch/suncet">[GitHub]</a> / --> <!-- <a href="https://sites.google.com/view/chiawen-kuo/home/sea?authuser=0">[Project]</a> / --> [<a href="javascript:void(0)" onclick="toggle_visibility('emu_bibtex');">bibtex</a>] <div id='emu_bibtex' style="display:none; font-size:small;"><pre><code>
12  @article{dai2023emu,
13    title={Emu: Enhancing image generation models using photogenic needles in a haystack},
14    author={Dai, Xiaoliang and Hou, Ji and Ma, Chih-Yao and Tsai, Sam and Wang, Jialiang and Wang, Rui and Zhang, Peizhao and Vandenhende, Simon and Wang, Xiaofang and Dubey, Abhimanyu and others},
15    journal={arXiv preprint arXiv:2309.15807},
16    year={2023}
17  }
18</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication">
18 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/ropaws.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>RoPAWS: Robust Semi-supervised Representation Learning from Uncurated Data </heading></font><br> Sangwoo Mo, Jong-Chyi Su, <strong> Chih-Yao Ma</strong>, Mido Assran, Ishan Misra, Licheng Yu, Sean Bell<br> <em>International Conference on Learning Representations (ICLR)</em>, 2023 <br> <a href="https://arxiv.org/abs/2302.14483">[arXiv]</a> / <a href="https://github.com/facebookresearch/suncet">[GitHub]</a> / <!-- <a href="https://sites.google.com/view/chiawen-kuo/home/sea?authuser=0">[Project]</a> / --> [<a href="javascript:void(0)" onclick="toggle_visibility('ropaws_bibtex');">bibtex</a>] <div id='ropaws_bibtex' style="display:none; font-size:small;"><pre><code>
19  @inproceedings{mo2023ropaws,
20    title={RoPAWS: Robust Semi-supervised Representation Learning from Uncurated Data},
21    author={Mo, Sangwoo and Su, Jong-Chyi and Ma, Chih-Yao and Assran, Mido and Misra, Ishan and Yu, Licheng and Bell, Sean},
22    booktitle={Proceedings of the International Conference on Learning Representations (ICLR)},
23    year={2023}
24  }
25</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/trainable.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Trainable Projected Gradient Method for Robust Fine-tuning </heading></font><br> Junjiao Tian, Zecheng He, Xiaoliang Dai, <strong> Chih-Yao Ma</strong>, Yen-Cheng Liu, Zsolt Kira<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2023 <br> <a href="https://arxiv.org/abs/2303.10720">[arXiv]</a> / <a href="https://github.com/PotatoTian/TPGM">[GitHub]</a> / <!-- <a href="https://sites.google.com/view/chiawen-kuo/home/sea?authuser=0">[Project]</a> / --> [<a href="javascript:void(0)" onclick="toggle_visibility('trainable_bibtex');">bibtex</a>] <div id='trainable_bibtex' style="display:none; font-size:small;"><pre><code>
26  @inproceedings{tian2023trainable,
27    title={Trainable Projected Gradient Method for Robust Fine-tuning},
28    author={Tian, Junjiao and He, Zecheng and Dai, Xiaoliang and Ma, Chih-Yao and Liu, Yen-Cheng and Kira, Zsolt},
29    booktitle={Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition},
30    pages={7836--7845},
31    year={2023}
32  }
33</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/structure.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Structure-Encoding Auxiliary Tasks for Improved Visual Representation in Vision-and-Language Navigation </heading></font><br> Chia-Wen Kuo, <strong> Chih-Yao Ma</strong>, Judy Hoffman, Zsolt Kira<br> <em>Winter Conference on Applications of Computer Vision (WACV)</em>, 2022<br> <a href="https://arxiv.org/abs/2211.11116">[arXiv]</a> / <!-- <a href="https://github.com/facebookresearch/adaptive_teacher">[GitHub]</a> / --> <a href="https://sites.google.com/view/chiawen-kuo/home/sea?authuser=0">[Project]</a> / <!-- [GitHub] (coming soon) --> [<a href="javascript:void(0)" onclick="toggle_visibility('structure_bibtex');">bibtex</a>] <div id='structure_bibtex' style="display:none; font-size:small;"><pre><code>
34  @inproceedings{kuo2023structure,
35    title={Structure-Encoding Auxiliary Tasks for Improved Visual Representation in Vision-and-Language Navigation},
36    author={Chia-Wen Kuo and Chih-Yao Ma and Judy Hoffman and Zsolt Kira},
37    booktitle={Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision},
38    pages={1104--1113},
39    year={2023}
40  }
41</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication">
41 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/polyhistor.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Polyhistor: Parameter-Efficient Multi-Task Adaptation for Dense Vision Tasks </heading></font><br> Yen-Cheng Liu, <strong> Chih-Yao Ma</strong>, Junjiao Tian, Zijian He, Zsolt Kira<br> <em>Conference on Neural Information Processing Systems (NeurIPS)</em>, 2022<br> <a href="https://arxiv.org/abs/2210.03265">[arXiv]</a> / <!-- <a href="https://github.com/facebookresearch/adaptive_teacher">[GitHub]</a> / --> <a href="https://ycliu93.github.io/projects/polyhistor.html">[Project]</a> / [GitHub] (coming soon) [<a href="javascript:void(0)" onclick="toggle_visibility('polyhistor_bibtex');">bibtex</a>] <div id='polyhistor_bibtex' style="display:none; font-size:small;"><pre><code>
42@article{liu2022polyhistor,
43  title={Polyhistor: Parameter-Efficient Multi-Task Adaptation for Dense Vision Tasks},
44  author={Liu, Yen-Cheng and Ma, Chih-Yao and Tian, Junjiao and He, Zijian and Kira, Zsolt},
45  journal={Advances in neural information processing systems},
46  year={2022}
47} 
48</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/ossod.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Open-Set Semi-Supervised Object Detection </heading></font><br> Yen-Cheng Liu, <strong> Chih-Yao Ma</strong>, Xiaoliang Dai, Junjiao Tian, Peter Vadja, Zijian He, Zsolt Kira<br> <em>European Conference on Computer Vision (ECCV)</em>, 2022 <strong>(Oral)</strong><br> <a href="https://arxiv.org/abs/2208.13722">[arXiv]</a> / <!-- <a href="https://github.com/facebookresearch/adaptive_teacher">[GitHub]</a> / --> <a href="https://ycliu93.github.io/projects/ossod.html">[Project]</a> / [GitHub] (coming soon) [<a href="javascript:void(0)" onclick="toggle_visibility('ossod_bibtex');">bibtex</a>] <div id='ossod_bibtex' style="display:none; font-size:small;"><pre><code>
49@inproceedings{liu2022open,
50  title={Open-Set Semi-Supervised Object Detection},
51  author={Liu, Yen-Cheng and Ma, Chih-Yao and Dai, Xiaoliang and Tian, Junjiao and Vajda, Peter and He, Zijian and Kira, Zsolt},
52  booktitle={Proceedings of the European Conference on Computer Vision (ECCV)},
53  year={2022}
54} 
55</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/aut.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Cross-Domain Adaptive Teacher for Object Detection </heading></font><br> Yu-Jhe Li, Xiaoliang Dai, <strong> Chih-Yao Ma</strong>, Yen-Cheng Liu, Kan Chen, Bichen Wu, Zijian He, Kris Kitani, Peter Vadja<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2022 <br> <a href="https://yujheli.github.io/projects/CVPR2022_assest/paper_adaptive_teacher_cvpr22.pdf">[PDF]</a> / <a href="https://github.com/facebookresearch/adaptive_teacher">[GitHub]</a> / <a href="https://yujheli.github.io/projects/adaptiveteacher.html">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('aut_bibtex');">bibtex</a>] <div id='aut_bibtex' style="display:none; font-size:small;"><pre><code>
56@inproceedings{li2022cross,
57  title={Cross-Domain Adaptive Teacher for Object Detection},
58  author={Li, Yu-Jhe and Dai, Xiaoliang and Ma, Chih-Yao and Liu, Yen-Cheng and Chen, Kan and Wu, Bichen and He, Zijian and Kitani, Kris and Vajda, Peter},
59  booktitle={IEEE Conference on Computer Vision and Pattern Recognition (CVPR)},
60  year={2022}
61} 
62</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/utv2.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Unbiased Teacher v2: Semi-supervised Object Detection for Anchor-free and Anchor-based Detectors </heading></font><br> Yen-Cheng Liu, <strong> Chih-Yao Ma</strong>, Zsolt Kira<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2022 <br> <a href="https://arxiv.org/abs/2206.09500">[arXiv]</a> / <a href="https://openaccess.thecvf.com/content/CVPR2022/papers/Liu_Unbiased_Teacher_v2_Semi-Supervised_Object_Detection_for_Anchor-Free_and_Anchor-Based_CVPR_2022_paper.pdf">[PDF]</a> / <a href="https://github.com/facebookresearch/unbiased-teacher-v2">[GitHub]</a> / <a href="https://ycliu93.github.io/projects/unbiasedteacher2.html">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('ut2_bibtex');">bibtex</a>] <div id='ut2_bibtex' style="display:none; font-size:small;"><pre><code>
63@InProceedings{Liu_2022_CVPR,
64    author    = {Liu, Yen-Cheng and Ma, Chih-Yao and Kira, Zsolt},
65    title     = {Unbiased Teacher v2: Semi-Supervised Object Detection for Anchor-Free and Anchor-Based Detectors},
66    booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)},
67    month     = {June},
68    year      = {2022},
69    pages     = {9819-9828}
70} 
71</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication">
71 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/hcm.jpg" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Hierarchical Cross-Modal Agent for Robotics Vision-and-Language Navigation</heading></font><br> Muhammad Zubair Irshad, <strong> Chih-Yao Ma</strong>, Zsolt Kira<br> <em>IEEE International Conference on Robotics and Automation (ICRA)</em>, 2021 <br> <a href="https://arxiv.org/abs/2104.10674">[arXiv]</a> / <a href="https://github.com/GT-RIPL/robo-vln">[GitHub]</a> / <a href="https://zubair-irshad.github.io/projects/robo-vln.html">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('hcm_bibtex');">bibtex</a>] <div id='hcm_bibtex' style="display:none; font-size:small;"><pre><code>
72@inproceedings{irshad2021hierarchical,
73  title={Hierarchical Cross-Modal Agent for Robotics Vision-and-Language Navigation},
74  author={Muhammad Zubair Irshad and Chih-Yao Ma and Zsolt Kira},
75  booktitle={Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)},
76  year={2021},
77  url={https://arxiv.org/abs/2104.10674}
78}
79</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/unbiased_teacher.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Unbiased Teacher for Semi-Supervised Object Detection</heading></font><br> Yen-Cheng Liu, <strong> Chih-Yao Ma</strong>, Zijian He, Chia-Wen Kuo, Kan Chen, Peizhao Zhang, Bichen Wu, Zsolt Kira, Peter Vajda<br> <em>International Conference on Learning Representations (ICLR)</em>, 2021 <br> <a href="https://arxiv.org/abs/2102.09480">[arXiv]</a> / <a href="https://github.com/facebookresearch/unbiased-teacher">[GitHub]</a> / <a href="https://ycliu93.github.io/projects/unbiasedteacher.html">[Project]</a> / <a href="https://openreview.net/forum?id=MJIve1zgR_">[OpenReview]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('ut_bibtex');">bibtex</a>] <div id='ut_bibtex' style="display:none; font-size:small;"><pre><code>
80  @inproceedings{liu2021unbiased,
81    title={Unbiased Teacher for Semi-Supervised Object Detection},
82    author={Liu, Yen-Cheng and Ma, Chih-Yao and He, Zijian and Kuo, Chia-Wen and Chen, Kan and Zhang, Peizhao and Wu, Bichen and Kira, Zsolt and Vajda, Peter},
83    booktitle={Proceedings of the International Conference on Learning Representations (ICLR)},
84    year={2021},
85}
86</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/cyclical.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Learning to Generate Grounded Visual Captions without Localization Supervision</heading></font><br> <strong> Chih-Yao Ma</strong>, Yannis Kalantidis, Ghassan AlRegib, Peter Vajda, Marcus Rohrbach, Zsolt Kira<br> <em>European Conference on Computer Vision (ECCV)</em>, 2020 <br> <a href="https://arxiv.org/abs/1906.00283">[arXiv]</a> / <a href="https://github.com/chihyaoma/cyclical-visual-captioning">[GitHub]</a> / <a href="project/2019/06/03/cyclical.html">[Project]</a> / <a href="https://sites.gatech.edu/mlatgteccv/research/learning-to-generate-grounded-visual-captions-without-localization-supervision/">[ML@GT]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('cyclical_bibtex');">bibtex</a>] <div id='cyclical_bibtex' style="display:none; font-size:small;"><pre><code>
87  @inproceedings{ma2020learning,
88    title={Learning to Generate Grounded Visual Captions without Localization Supervision},
89    author={Ma, Chih-Yao and Kalantidis, Yannis and AlRegib, Ghassan and Vajda, Peter and Rohrbach, Marcus and Kira, Zsolt},
90    booktitle={Proceedings of the European Conference on Computer Vision (ECCV)},
91    year={2020},
92    url={https://arxiv.org/abs/1906.00283},
93}
94</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication">
94 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/featmatch.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>FeatMatch: Feature-Based Augmentation for Semi-Supervised Learning</heading></font><br> Chia-Wen Kuo, <strong> Chih-Yao Ma</strong>, Jia-Bin Huang, Zsolt Kira<br> <em>European Conference on Computer Vision (ECCV)</em>, 2020 <br> <a href="https://arxiv.org/abs/2007.08505">[arXiv]</a> / <a href="https://sites.google.com/view/chiawen-kuo/home/featmatch">[Project]</a> / <a href="https://github.com/GT-RIPL/FeatMatch">[GitHub]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('featmatch_bibtex');">bibtex</a>] <div id='featmatch_bibtex' style="display:none; font-size:small;"><pre><code>
95  @inproceedings{kuo2020featmatch,
96    title={FeatMatch: Feature-Based Augmentationfor Semi-Supervised Learning},
97    author={Kuo, Chia-Wen and Ma, Chih-Yao and Huang, Jia-Bin and Kira, Zsolt}, 
98    booktitle={Proceedings of the European Conference on Computer Vision (ECCV)},
99    year={2020},
100    url={https://arxiv.org/abs/2007.08505}
101  }
102</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/who2com.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Who2com: Collaborative Perception Via Learnable Handshake Communication</heading></font><br> Yen-Cheng Liu, Junjiao Tian, <strong> Chih-Yao Ma</strong>, Nathaniel Glaser, Chia-Wen Kuo, Zsolt Kira<br> <em>International Conference on Robotics and Automation (ICRA)</em>, 2020 <br> <a href="https://arxiv.org/abs/2003.09575">[arXiv]</a> <a href="https://github.com/GT-RIPL/MultiAgentPerception">[GitHub]</a> / <a href="https://ycliu93.github.io/projects/multi-agent-perception.html">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('who2com_bibtex');">bibtex</a>] <div id='who2com_bibtex' style="display:none; font-size:small;"><pre><code>
103  @inproceedings{liu2020who2com,
104    title={Who2com: Collaborative Perception via Learnable Handshake Communication},
105    author={Liu, Yen-Cheng and Tian, Junjiao and Ma, Chih-Yao and Glaser, Nathan and Kuo, Chia-Wen and Kira, Zsolt},
106    booktitle={Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)},
107    year={2020},
108    url={https://arxiv.org/abs/2003.09575},
109}
110</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <img src="static/assets/img/teasers/prototypes.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Manifold Graph with Learned Prototypes for Semi-Supervised Image Classification</heading></font><br> Chia-Wen Kuo, <strong>Chih-Yao Ma</strong>, Jia-Bin Huang, Zsolt Kira<br> <em>Technical Report</em>, 2019 <br> <a href="https://arxiv.org/abs/1906.05202">[arXiv]</a> / <!-- [GitHub (coming soon)] / --> <!-- <a href="https://github.com/chihyaoma/regretful-agent">[GitHub]</a> / --> <a href="https://sites.google.com/view/manifold-graph-with-prototypes/home">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('manifold_bibtex');">bibtex</a>] <div id='manifold_bibtex' style="display:none; font-size:small;"><pre><code>
111  @article{kuo2019manifold,
112    title={Manifold Graph with Learned Prototypes for Semi-Supervised Image Classification},
113    author={Kuo, Chia-Wen and Ma, Chih-Yao and Huang, Jia-Bin and Kira, Zsolt},
114    journal={arXiv preprint arXiv:1906.05202},
115    year={2019},
116    url={https://arxiv.org/abs/1906.05202},
117}
118</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/regretful.png" alt="game" height="25%" style="border-style: none"> --> <img src="static/assets/img/teasers/regretful.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>The Regretful Agent: Heuristic-Aided Navigation through Progress Estimation</heading></font><br> <strong> Chih-Yao Ma</strong>, Zuxuan Wu, Ghassan AlRegib, Caiming Xiong, Zsolt Kira<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2019 <strong>(Oral)</strong><br> <a href="https://arxiv.org/abs/1903.01602">[arXiv]</a> / <a href="https://github.com/chihyaoma/regretful-agent">[GitHub]</a> / <a href="project/2019/02/25/regretful.html">[Project]</a> / <a href="static/assets/pdf/cvpr2019_poster_final-compressed.pdf">[Poster]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('regretful_bibtex');">bibtex</a>] <div id='regretful_bibtex' style="display:none; font-size:small;"><pre><code>
119@inproceedings{ma2019theregretful,
120  title={The Regretful Agent: Heuristic-Aided Navigation through Progress Estimation},
121  author={Ma, Chih-Yao and Wu, Zuxuan and AlRegib, Ghassan and Xiong, Caiming and Kira, Zsolt},
122  booktitle={Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)},
123  year={2019},
124  url={https://arxiv.org/abs/1903.01602},
125}
126</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication">
126 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/adaframe.png" alt="game" height="25%" style="border-style: none"> --> <img src="static/assets/img/teasers/adaframe.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <!-- <p><a href="https://arxiv.org/abs/1811.12432"><heading>AdaFrame: Adaptive Frame Selection for Fast Video Recognition</heading></a><br> --> <font color="#2C90DE"><heading>AdaFrame: Adaptive Frame Selection for Fast Video Recognition</heading></font><br> Zuxuan Wu, Caiming Xiong, <strong> Chih-Yao Ma</strong>, Richard Socher, Larry S Davis<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2019<br> <div class="paper" id="adaframe"> <a href="https://arxiv.org/abs/1811.12432">[arXiv]</a> / <a href="static/assets/pdf/cvpr2019_poster_adaframe-compressed.pdf">[Poster]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('adaframe_bibtex');">bibtex</a>] <div id='adaframe_bibtex' style="display:none; font-size:small;"><pre><code>
127@inproceedings{wu2019adaframe,
128  title={AdaFrame: Adaptive Frame Selection for Fast Video Recognition},
129  author={Wu, Zuxuan and Xiong, Caiming and Ma, Chih-Yao and Socher, Richard and Davis, Larry S},
130  booktitle={Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)},
131  year={2019},
132  url={https://arxiv.org/abs/1811.12432},
133}
134</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/selfmonitoring.png" alt="game" height="30%" style="border-style: none"> --> <img src="static/assets/img/teasers/selfmonitoring.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <!-- <p><a href=""><heading>Self-Monitoring Navigation Agent via Auxiliary Progress Estimation</heading></a><br> --> <font color="#2C90DE"><heading>Self-Monitoring Navigation Agent via Auxiliary Progress Estimation</heading></font><br> <strong> Chih-Yao Ma</strong>, Jiasen Lu, Zuxuan Wu, Ghassan AlRegib, Zsolt Kira, Richard Socher, Caiming Xiong<br> <em>International Conference on Learning Representations (ICLR)</em>, 2019<br> <strong><em>(Top 7% of reviews)</em></strong><br> <div class="paper" id="selfmonitoring"> <a href="https://arxiv.org/abs/1901.03035">[arXiv]</a> / <a href="https://openreview.net/forum?id=r1GAsjC5Fm">[OpenReview]</a> / <a href="https://github.com/chihyaoma/selfmonitoring-agent">[GitHub]</a> / <a href="project/2018/09/27/selfmonitoring.html">[Project]</a> / <a href="static/assets/pdf/iclr2019_poster_final-compressed.pdf">[Poster]</a> / <a href="https://ml.gatech.edu/hg/item/620601">[ML@GT]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('selfmonitoring_bibtex');">bibtex</a>] <div id='selfmonitoring_bibtex' style="display:none; font-size:small;"><pre><code>
135@inproceedings{ma2019selfmonitoring,
136  title={Self-Monitoring Navigation Agent via Auxiliary Progress Estimation},
137  author={Ma, Chih-Yao and Lu, Jiasen and Wu, Zuxuan and AlRegib, Ghassan and Kira, Zsolt and Socher, Richard and Xiong, Caiming},
138  booktitle={Proceedings of the International Conference on Learning Representations (ICLR)},
139  year={2019},
140  url={https://arxiv.org/abs/1901.03035},
141}
142</code></pre></div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/interaction.png" alt="game" height="25%" style="border-style: none"> --> <img src="static/assets/img/teasers/interaction.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <!-- <p><a href="https://arxiv.org/abs/1711.06330"><heading>Attend and Interact: Higher-Order Object Interactions for Video Understanding</heading></a><br> --> <font color="#2C90DE"><heading>Attend and Interact: Higher-Order Object Interactions for Video Understanding</heading></font><br> <strong> Chih-Yao Ma</strong>, Asim Kadav, Iain Melvin, Zsolt Kira, Ghassan AlRegib, Hans Peter Graf<br> <em>Computer Vision and Pattern Recognition (CVPR)</em>, 2018<br> <div class="paper" id="interaction"> <a href="https://arxiv.org/abs/1711.06330">[arXiv]</a> / <a href="project/2017/11/16/interact.html">[Project]</a> / <a href="static/assets/pdf/cvpr2018_poster_final.pdf">[Poster]</a> / <a href="https://mlatgt.blog/2018/04/02/from-object-interactions-to-fine-grained-video-understanding/"> [ML@GT]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('attend_bibtex');">bibtex</a>] <div id='attend_bibtex' style="display:none; font-size:small;"><pre><code>
143@inproceedings{ma2018attend,
144  title={Attend and Interact: Higher-Order Object Interactions for Video Understanding},
145  author={Ma, Chih-Yao and Kadav, Asim and Melvin, Iain and Kira, Zsolt and AlRegib, Ghassan and Graf, Hans Peter},
146  booktitle={Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)},
147  year={2018}
148}
149</code></pre></div> </div> </td> </td> </table> </div> <hr> <div class="publication">
149 <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/tslstm-demo-3.gif" alt="game" height="60%" style="border-style: none"> <img src="static/assets/img/teasers/tslstm-demo-4.gif" alt="game" height="60%" style="border-style: none"> --> <img src="static/assets/img/teasers/tslstm-demo-3.gif" alt="game" width="25%" style="border-style: none"> <img src="static/assets/img/teasers/tslstm-demo-4.gif" alt="game" width="25%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>TS-LSTM and temporal-inception: Exploiting spatiotemporal dynamics for activity recognition</heading></font><br> <strong> Chih-Yao Ma<sup>*</sup></strong>, Min-Hung Chen<sup>*</sup>, Zsolt Kira, and Ghassan AlRegib<br> <em>Signal Processing: Image Communication</em>, 2018<br> (<sup>*</sup>: equal contribution)<br> <div class="paper" id="tslstm"> <a href="https://arxiv.org/abs/1703.10667">[arXiv]</a> / <a href="https://github.com/chihyaoma/Activity-Recognition-with-CNN-and-RNN">[GitHub]</a> / <a href="project/2017/03/30/tslstm.html">[Project]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('tslstm_bibtex');">bibtex</a>] <div id='tslstm_bibtex' style="display:none; font-size:small;"><pre><code>
150@article{ma2019ts,
151  title={TS-LSTM and temporal-inception: Exploiting spatiotemporal dynamics for activity recognition},
152  author={Ma, Chih-Yao and Chen, Min-Hung and Kira, Zsolt and AlRegib, Ghassan},
153  journal={Signal Processing: Image Communication},
154  volume={71},
155  pages={76--87},
156  year={2019},
157  publisher={Elsevier}
158}
159</code></pre></div> </div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/caption.png" alt="game" height="40%" style="border-style: none"> --> <img src="static/assets/img/teasers/caption.png" alt="game" width="50%" style="border-style: none"> <td width="50%" valign="center"> <font color="#2C90DE"><heading>Grounded Objects and Interactions for Video Captioning</heading></font><br> <strong> Chih-Yao Ma</strong>, Asim Kadav, Iain Melvin, Zsolt Kira, Ghassan AlRegib, Hans Peter Graf<br> <em>Neural Information Processing Systems (NeurIPS) Workshop on Visually-Grounded Interaction and Language</em>, 2017<br> <div class="paper" id="caption"> <a href="https://arxiv.org/abs/1711.06354">[arXiv]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('Grounded_bibtex');">bibtex</a>] <div id='Grounded_bibtex' style="display:none; font-size:small;"><pre><code>
160@article{ma2017grounded,
161  title={Grounded Objects and Interactions for Video Captioning},
162  author={Ma, Chih-Yao and Kadav, Asim and Melvin, Iain and Kira, Zsolt and AlRegib, Ghassan and Graf, Hans Peter},
163  journal={arXiv preprint arXiv:1711.06354},
164  year={2017}
165}
166</code></pre></div> </div> </td> </td> </table> </div> <hr> <div class="publication"> <table width="100%" align="center" border="0" cellpadding="0"> <td width="50%" valign="center" align="center"> <!-- <img src="static/assets/img/teasers/saliency.png" alt="game" height="15%" style="border-style: none"> <img src="static/assets/img/teasers/saliency-model.png" alt="game" height="15%" style="border-style: none"> --> <img src="static/assets/img/teasers/saliency.png" alt="game" width="25%" style="border-style: none"> <img src="static/assets/img/teasers/saliency-model.png" alt="game" width="25%" style="border-style: none"> <td width="50%" valign="center"> <!-- <p><a href="https://jov.arvojournals.org/article.aspx?articleid=2300610"><heading>Learning-based saliency model with depth information</heading></a><br> --> <font color="#2C90DE"><heading>Learning-based saliency model with depth information</heading></font><br> <strong> Chih-Yao Ma</strong> and Hsueh-Ming Hang<br> <em>Journal of vision</em>, 2015<br> <div class="paper" id="saliency"> <a href="https://jov.arvojournals.org/article.aspx?articleid=2300610">[Paper]</a> / [<a href="javascript:void(0)" onclick="toggle_visibility('saliency_bibtex');">bibtex</a>] <div id='saliency_bibtex' style="display:none; font-size:small;"><pre><code>
167@article{ma2015learning,
168  title={Learning-based saliency model with depth information},
169  author={Ma, Chih-Yao and Hang, Hsueh-Ming},
170  journal={Journal of vision},
171  volume={15},
172  number={6},
173  pages={19--19},
174  year={2015},
175  publisher={The Association for Research in Vision and Ophthalmology}
176}
177</code></pre></div> </div> </td> </td> </table> </div> <hr> </div> <br> </section> <section id="skills" class="features gray-section team" style="margin-top: 0;"> <div class="container"> <div class="row"> <div class="col-lg-12 text-center" id="i18_skills"> <div class="navy-line"></div> <h1><span data-i18n="skills.my_skills">Research Interest</span></h1> </div> </div> <div class="row features-block"> <div class="wow zoomIn col-lg-6 col-lg-offset-3">
177 <canvas id="cs" height="300" width="500"></canvas> </div> <div class="col-lg-1"></div> 
177<script> var ctx = document.getElementById("cs"); var data = { labels: "Semi-Supervised Learning, Vision-and-Language Nav., Visual Captioning, Visual Question Answering, Action Recognition, Natural Language Process, Federated Learning".split(","), datasets: [{ label: "Interest", backgroundColor: "rgba(179,181,198,0.2)", borderColor: "#3385FF", pointBackgroundColor: "#3385FF", pointBorderColor: "#fff", pointHoverBackgroundColor: "#3385FF", pointHoverBorderColor: "#3385FF", data: [80, 60, 55, 35, 30, 20, 70] }] }; var myRadarChart = new Chart(ctx, { type: 'radar', data: data, options: { scale: { responsive: true, ticks: {min: 0, max: 100}, lineArc: false, pointLabels: {fontSize: 14}, }, legend: {display: false}, } }); </script>
177 </div> </div> </section> <section id="services" class="features " style="margin-top: 0;"> <div class="container"> <div class="row m-b-lg"> <div class="col-lg-12 text-center"> <div class="navy-line"></div> <h1><span data-i18n="service.service">Services</span></h1> </div> </div> <div style="width: 50%; margin-left: auto; margin-right: auto; font-size:20px"> <ul> <li>Reviewer for NeurIPS, ICLR, ICML</li> <li>Reviewer for CVPR, ICCV, ECCV</li> <li>Reviewer for NAACL</li> <li>Reviewer for T-PAMI, T-IP, T-TCSVT</li> </ul> </div> </div> <br> </section> <section id="project" class="features gray-section contact navy-section" style="margin-top: 0;"> <div class="container"> <div class="row"> <div class="col-lg-12 text-center wow zoomIn" id="i18_blog"> <div style="height: 80px;"></div> <a class="btn btn-lg btn-default btn-rounded btn-outline wow bounceIn" href="/project/"><span data-i18n="project.my_project">Project Pages</span></a> <div style="height: 80px;"></div> </div> </div> </div> <div class="row"> <div class="col-lg-8 col-lg-offset-2 text-center m-t-lg m-b-lg"> <p id="jalpc_site_pv"><strong>&copy; 2026 Kevin Chih-Yao Ma</strong></p> </div> </div> </section> <style> iframe { -moz-transform: scale(0.25, 0.25); -webkit-transform: scale(0.25, 0.25); -o-transform: scale(0.25, 0.25); -ms-transform: scale(0.25, 0.25); transform: scale(0.25, 0.25); -moz-transform-origin: top left; -webkit-transform-origin: top left; -o-transform-origin: top left; -ms-transform-origin: top left; transform-origin: top left; } </style> <!-- <iframe src="https://www.maimemo.com/share/page/?uid=779139&pid=660" style="display: none;"></iframe> --> <!-- Google analytics --> 
177<script>
vendor: 334 bytes, line 177
177 (function(i,s,o,g,r,a,m){i['GoogleAnalyticsObject']=r;i[r]=i[r]||function(){ (i[r].q=i[r].q||[]).push(arguments)},i[r].l=1*new Date();a=s.createElement(o), m=s.getElementsByTagName(o)[0];a.async=1;a.src=g;m.parentNode.insertBefore(a,m) })(window,document,'script','https://www.google-analytics.com/analytics.js','ga'); ga('create', '
177UA-131192687-1
vendor: 12 bytes, line 177
177', 'auto'); 
177ga('require', ''); ga('send', 'pageview'); </script>
177 <!-- GrowingIO --> </body> </html>

Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.