1<!DOCTYPE html> 2<html> 3 <head> 4 <style> 5 td, th { 6 border: 0px solid black; 7 } 8 img{ 9 padding: 5px; 10 } 11 </style> 12 <title>Sketch Down the FLOPs: Towards Efficient Networks for Human Sketch</title> 13 <!-- Global site tag (gtag.js) - Google Analytics --> 14 <!--
vendor: 65 bytes, line 14
14 <script async src="https://www.googletagmanager.com/gtag/js?id=
14G-PYVRSFMDRL
vendor: 21 bytes, lines 14-15
14"></script> 15
15<script> 16
vendor: 217 bytes, lines 16-24
16window.dataLayer = window.dataLayer || []; 17 18 function gtag() { 19 dataLayer.push(arguments); 20 } 21 22 gtag('js', new Date()); 23 24 gtag('config', '
24G-PYVRSFMDRL
vendor: 43 bytes, lines 24-28
24'); 25 26 27 28
28</script>
28 --> 29 <link href="https://fonts.googleapis.com/css?family=Google+Sans|Noto+Sans|Castoro" 30 rel="stylesheet"> 31 <link rel="stylesheet" href="./static/css/bulma.min.css"> 32 <link rel="stylesheet" href="./static/css/bulma-carousel.min.css"> 33 <link rel="stylesheet" href="./static/css/bulma-slider.min.css"> 34 <link rel="stylesheet" href="./static/css/fontawesome.all.min.css"> 35 <link rel="stylesheet" 36 href="https://cdn.jsdelivr.net/gh/jpswalsh/academicons@1/css/academicons.min.css"> 37 <link rel="stylesheet" href="./static/css/index.css"> 38 <link rel="icon" href="./static/images/site_icon64.png"> 39 <link rel="stylesheet" href="https://unpkg.com/image-compare-viewer/dist/image-compare-viewer.min.css"> 40 <link rel="stylesheet" href="css/app.css"> 41 <link rel="stylesheet" href="css/bootstrap.min.css"> 42
42<script src="https://unpkg.com/image-compare-viewer/dist/image-compare-viewer.min.js"></script>
42 43
43<script src="https://ajax.googleapis.com/ajax/libs/jquery/3.5.1/jquery.min.js"></script>
43 44
44<script defer src="./static/js/fontawesome.all.min.js"></script>
44 45
45<script src="./static/js/bulma-carousel.min.js"></script>
45 46
46<script src="./static/js/bulma-slider.min.js"></script>
46 47
47<script src="./static/js/index.js"></script>
47 48 </head> 49 <body> 50 <section class="hero"> 51 <div class="hero-body"> 52 <div class="container is-max-desktop"> 53 <div class="columns is-centered"> 54 <div class="column has-text-centered"> 55 <h1 class="title is-1 publication-title", style="color:rgb(167, 62, 202);">Sketch Down the FLOPs:</h1> 56 <h1 class="title is-4 publication-title">Towards Efficient Networks for Human Sketch</h1> 57 <div class="is-size-5 publication-authors"> 58 <span class="author-block"> 59 <a href="https://aneeshan95.github.io/">Aneeshan Sain</a><sup>1</sup>,</span> 60 <span class="author-block"> 61 <a href="https://subhajitmaity.me/">Subhajit Maity</a><sup>2</sup>,</span> 62 <span class="author-block"> 63 <a href="https://www.pinakinathc.me/">Pinaki Nath Chowdhury</a><sup>1</sup>,</span> 64 <span class="author-block"> 65 <a href="https://subhadeepkoley.github.io/">Subhadeep Koley</a><sup>1</sup>,</span> 66 <span class="author-block"> 67 <a href="https://ayankumarbhunia.github.io/">Ayan Kumar Bhunia</a><sup>1</sup>,</span> 68 <span class="author-block"> 69 <a href="https://www.surrey.ac.uk/people/yi-zhe-song">Yi-Zhe Song</a><sup>1</sup></span> 70 </div> 71 <div class="is-size-5 publication-authors"> 72 <span class="author-block"><sup>1</sup>SketchX, CVSSP, University of Surrey, United Kingdom</span> 73 <span class="author-block"><sup>2</sup>Department of Computer Science, University of Central Florida</span> 74 </div> 75 <div class="column has-text-centered"> 76 <a href="https://cvpr.thecvf.com/"> 77 <img src="./static/images/CVPR_Nashville_FinalLogo.jpg" alt="CVPR 2025" width="250px" height="auto"> 78 </a> 79 </span> 80 </div> 81 <div class="column has-text-centered"> 82 <div class="publication-links"> 83 <!-- PDF Link. --> 84 <span class="link-block"> 85 <a href="https://arxiv.org/pdf/2505.23763" 86 class="external-link button is-normal is-rounded is-dark"> 87 <span class="icon"> 88 <i class="fas fa-file-pdf"></i> 89 </span> 90 <span>Paper (PDF)</span> 91 </a> 92 </span> 93 <span class="link-block"> 94 <a href="https://arxiv.org/abs/2505.23763" 95 class="external-link button is-normal is-rounded is-dark"> 96 <span class="icon"> 97 <i class="ai ai-arxiv"></i> 98 </span> 99 <span>arXiv</span> 100 </a> 101 </span> 102 <!-- Video Link. --> 103 <!-- <span class="link-block"> 104 <a href="https://www.youtube.com/watch?v=k7xFbELpnv4" 105 class="external-link button is-normal is-rounded is-dark"> 106 <span class="icon"> 107 <i class="fab fa-youtube"></i> 108 </span> 109 <span>Video (YouTube)</span> 110 </a> 111 </span> --> 112 <!-- Code Link. --> 113 <span class="link-block"> 114 <a href="" 115 class="external-link button is-normal is-rounded is-dark"> 116 <span class="icon"> 117 <i class="fab fa-github"></i> 118 </span> 119 <span>Code</span> 120 </a> 121 </span> 122 <!-- Dataset Link. -->
123 <span class="link-block"> 124 <a href="./static/images/CVPR2025_SketchDownTheFLOPs_FinalV4.png" 125 class="external-link button is-normal is-rounded is-dark"> 126 <span class="icon"> 127 <i class="far fa-images"></i> 128 </span> 129 <span>Poster</span> 130 </a> 131 </div> 132 </div> 133 </div> 134 </div> 135 </div> 136 </div> 137 </section> 138 <section class="hero teaser"> 139 <div class="container is-max-desktop"> 140 <div class="hero-body"> 141 <img class="round" style="width:1500px" src="./static/images/teaser.png"/> 142 <h2 class="subtitle has-text-centered"> 143 <span class="dnerf"></span> Our SketchyNetV1 compresses existing heavy FG-SBIR networks to deliver smaller models. Further enhanced via a canvas-selector module our SketchyNetV2 model minimises sketch-resolution dynamically to reduce FLOPs. 144 </h2> 145 </div> 146 </div> 147 </section> 148 149 <!-- Paper video. --> 150 <section class="section"> 151 <div class="columns is-centered has-text-centered"> 152 <div class="column is-four-fifths"> 153 <h2 class="title is-3">Video</h2> 154 <div class="publication-video"> 155 <iframe src="https://www.youtube.com/embed/8BeHuh6T5ZI?rel=0&showinfo=0" frameborder="0" width="1500" height="300" allow="autoplay; encrypted-media" allowfullscreen></iframe> 156 </div> 157 </div> 158 </div> 159 </section> 160 <!--/ Paper video. --> 161 162 <section class="section"> 163 <div class="container is-max-desktop"> 164 <!-- Abstract. --> 165 <div class="columns is-centered has-text-centered"> 166 <div class="column is-four-fifths"> 167 <h2 class="title is-3">Abstract</h2> 168 <div class="content has-text-justified"> 169 <p> 170 As sketch research has collectively matured over time, its adaptation for at-mass commercialisation emerges on the immediate horizon. Despite an already mature research endeavour for photos, there is no research on the efficient inference specifically designed for sketch data. In this paper, we first demonstrate existing state-of-the-art efficient light-weight models designed for photos do not work on sketches. We then propose two sketch-specific components which work in a plug-n-play manner on any photo efficient network to adapt them to work on sketch data. We specifically chose fine-grained sketch-based image retrieval (FG-SBIR) as a demonstrator as the most recognised sketch problem with immediate commercial value. Technically speaking, we first propose a cross-modal knowledge distillation network to transfer existing photo efficient networks to be compatible with sketch, which brings down number of FLOPs and model parameters by <em>97.96%</em> percent and <em>84.89%</em> respectively. We then exploit the abstract trait of sketch to introduce a RL-based canvas selector that dynamically adjusts to the abstraction level which further cuts down number of FLOPs by two thirds. The end result is an overall reduction of <em>99.37%</em> of FLOPs (from <em>40.18G</em> to <em>0.254G</em>) when compared with a full network, while retaining the accuracy (<em>33.03%</em> vs <em>32.77%</em>) -- finally making an efficient network for the sparse sketch data that exhibit even fewer FLOPs than the best photo counterpart. 171 </p> 172 </div> 173 </div> 174 </div> 175 176 <!-- <section class="hero teaser"> 177 <div class="container is-max-desktop"> 178 <div class="hero-body"> 179 <iframe width="720" height="480" 180 src="https://www.youtube.com/embed/k7xFbELpnv4?"> 181 </iframe> 182 <h2 class="subtitle has-text-centered"> 183 <span class="dnerf"></span> 184 </h2> 185 </div> 186 </div> 187 </section> --> 188 189 <!--/ Abstract. --> 190 <!-- Paper video. --> 191 <section class="section"> 192 <div class="container is-max-desktop"> 193 <!-- Abstract. --> 194 <div class="columns is-centered has-text-centered"> 195 <div class="column is-four-fifths"> 196 <h2 class="title is-3">Pilot Study</h2> 197 <div class="content has-text-justified"> 198 </h2> 199 <center> 200 <img src="./static/images/pilot_study.png" alt="pilot study" border=0 height=1000 width=1500></img> 201 </center> 202 <h5 class="subtitle has-text-centered"> 203 Unlike photos that hold pixel-dense information, sketches are sparse black and white lines. This begs the question, if a sketch rendered at a higher resolution with added computational burden would convey any extra semantic information than at a lower one. From the pilot study we observe, While FG-IBIR accuracy falls rapidly, FG-SBIR stays relatively stable against decreasing canvas-sizes, as photos (unlike sketches) containing pixel-dense perfect information, lose a lot of it while down-scaling. Furthermore, positive accuracy of FG-SBIR at 32 Ã 32, shows some sketches to hold sufficient semantic information for retrieval even at minimal canvas-size. 204 </h5> 205 <br> 206 207 </div> 208 </div> 209 </div> 210 </section> 211 212 <section class="section"> 213 <div class="container is-max-desktop"> 214 <!-- Abstract. --> 215 <div class="columns is-centered has-text-centered"> 216 <div class="column is-four-fifths"> 217 <h2 class="title is-3">Architecture</h2> 218 <div class="content has-text-justified"> 219 </h2> 220 <center> 221 <img src="./static/images/arch_sketchynetv1.png" alt="sketchynetv1" border=0 height=1000 width=1500></img> 222 </center> 223 <h5 class="subtitle has-text-centered"> 224 SketchyNetV1: A smaller student network is trained from a larger pre-trained teacher via Knowledge Distillation. 225 </h5> 226 <br> 227 228 <br> 229 <center> 230 <img src="./static/images/arch_sketchynetv2.png" alt="sketchynetv2" border=0 height=300 width=1500/> 231 </center> 232 <h5 class="subtitle has-text-centered"> 233 We progress to SketchyNetV2 by training a canvas-size selector directed by objectives of increasing performance and reducing compute. It takes sketch as a vector and aims to predict an optimal canvas-size at which the sketch is rasterised. Rasterized sketch when processed by the trained student network minimises overall FLOPS, while retaining accuracy of corresponding full-resolution sketch-image. 234 </h5> 235 </div> 236 </div> 237 </div> 238 </section> 239 240 241 <section class="hero"> 242 <div class="hero-body"> 243 <div class="container is-max-desktop"> 244 <!-- Abstract. --> 245 <div class="columns is-centered has-text-centered"> 246 <div class="column is-four-fifths"> 247 <h2 class="title is-3">Results</h2> 248 <div class="content has-text-justified"> 249 <center> 250 <img src="./static/images/results.png" alt="Qualitative" border=0 height=300 width=1500/> 251 </center> 252 <h5 class="subtitle has-text-centered"> 253 Exemplary sketches at their optimal canvas sizes. 254 </h5> 255 <br> 256 257 <br> 258 <center> 259 <img src="./static/images/results_full.png" alt="Quantitative" border=0 height=600 width=3000/> 260 </center> 261 <h5 class="subtitle has-text-centered"> 262 Quantitative Analysis on FG-SBIR. Best viewed when zoomed in. Signifant reduction in FLOPs and model size with minimal accuracy drop. 263 </h5> 264 <br> 265 266 <br> 267 <center> 268 <img src="./static/images/lambda_flops.png" alt="Tradeoff" border=0 height=300 width=1500/> 269 </center> 270 <h5 class="subtitle has-text-centered"> 271 Tradeoff between FLOPs and Accuracy. Any point on the curve can be chosen by varying a hyperparameter. 272 </h5> 273 274 <br> 275 <center> 276 <img src="./static/images/ablation.png" alt="Ablation" border=0 height=300 width=1500/> 277 </center> 278 <h5 class="subtitle has-text-centered"> 279 Ablative Studies. 280 </h5> 281 </div> 282 </div> 283 </div> 284 </div> 285 <section class="section" id="BibTeX"> 286 <div class="container is-max-desktop content"> 287 <h2 class="title">BibTeX</h2> 288 <pre><code>@inproceedings{sain2025sketchdowntheflops, 289title={Sketch Down the FLOPs: Towards Efficient Networks for Human Sketch},
290author={Aneeshan Sain and Subhajit Maity and Pinaki Nath Chowdhury and Subhadeep Koley and Ayan Kumar Bhunia and Yi-Zhe Song}, 291booktitle={IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, 292year={2025}}</code></pre> 293 </div> 294 </section> 295
295<script> 296 const viewers = document.querySelectorAll(".image-compare"); 297 viewers.forEach((element) => { 298 let view = new ImageCompare(element, { 299 hoverStart: true, 300 addCircle: true 301 }).mount(); 302 }); 303 304 $(document).ready(function () { 305 var editor = CodeMirror.fromTextArea(document.getElementById("bibtex"), { 306 lineNumbers: false, 307 lineWrapping: true, 308 readOnly: true 309 }); 310 $(function () { 311 $('[data-toggle="tooltip"]').tooltip() 312 }) 313 }); 314 </script>
314 315 <br> 316 <p style="text-align:center"> Copyright: <a href="https://creativecommons.org/licenses/by-nc-sa/4.0/"> CC BY-NC-SA 4.0</a> © Subhajit Maity | Last updated: 7 Jun 2025 |Template Credit: <a href="https://nerfies.github.io/"> Nerfies</a></p> 317
317<script type="module" src="https://static.cloudflareinsights.com/beacon.min.js/v31edd6df95cf4e85bb4c19e7a9bdbcba1788362987495" integrity="sha512-iIg7k2xntmwu6/uSb5tpc/hySgZc4eoL31yB29W6tJFo2akwjPWcEqnCEdJvGexCL0KEQwVYv5BlowfhVz26hg==" data-cf-beacon='{"version":"2024.11.0","token":"faf6001e9a4049c39e24dc468751a337","r":1,"spa":2}' crossorigin="anonymous"></script>
317 318</body> 319</html>
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.