A Video Classification Method Based on Spatiotemporal Detail Attention and Feature Fusion

<table class="table-group" id="tab3"><tr><td><table class="table"><tr><td class="thead-hr" colspan="4"><hr/></td></tr><tr class="thead"><td class="align_left"></td><td class="align_center">Top1</td><td class="align_center">Top5</td><td class="align_center">GFLOPs</td></tr><tr><td class="thead-hr" colspan="4"><hr/></td></tr><tr><td class="align_left">I3D [<a href="/journals/misy/2022/4213335/#B15" target="_blank">15</a>]</td><td class="align_center">72.1</td><td class="align_center">90.3</td><td class="align_center">108</td></tr><tr><td class="align_left">Two-stream I3D [<a href="/journals/misy/2022/4213335/#B15" target="_blank">15</a>]</td><td class="align_center">75.7</td><td class="align_center">92.0</td><td class="align_center">216</td></tr><tr><td class="align_left">S3D-G [<a href="/journals/misy/2022/4213335/#B26" target="_blank">26</a>]</td><td class="align_center">77.2</td><td class="align_center">93.0</td><td class="align_center">—</td></tr><tr><td class="align_left">Nonlocal R50 [<a href="/journals/misy/2022/4213335/#B47" target="_blank">47</a>]</td><td class="align_center">76.5</td><td class="align_center">92.6</td><td class="align_center">—</td></tr><tr><td class="align_left">Nonlocal R101 [<a href="/journals/misy/2022/4213335/#B47" target="_blank">47</a>]</td><td class="align_center">77.7</td><td class="align_center">93.3</td><td class="align_center">—</td></tr><tr><td class="align_left">R(<span class="nowrap"><svg height="8.69875pt" id="M81" style="vertical-align:-0.3499298pt" version="1.1" viewbox="-0.0498162 -8.34882 26.097 8.69875" width="26.097pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M412 140C382 77 369 73 315 73H129L270 222C362 320 402 379 402 466C402 571 322 635 234 635C177 635 130 609 99 576L42 495L64 475C90 514 133 568 201 568C274 568 318 519 318 435C318 349 255 267 193 193C144 135 87 78 32 23V0H405C417 45 427 89 440 131L412 140Z"></path></g><g transform="matrix(.013,0,0,-0.013,9.145,0)"><path d="M535 230V280H323V490H265V280H52V230H265V-3H323V230H535Z"></path></g><g transform="matrix(.013,0,0,-0.013,19.682,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g></svg>)</span>D Flow [<a href="/journals/misy/2022/4213335/#B25" target="_blank">25</a>]</td><td class="align_center">67.5</td><td class="align_center">87.2</td><td class="align_center">152</td></tr><tr><td class="align_left">STC [<a href="/journals/misy/2022/4213335/#B48" target="_blank">48</a>]</td><td class="align_center">68.7</td><td class="align_center">88.5</td><td class="align_center">—</td></tr><tr><td class="align_left">ARTNet [<a href="/journals/misy/2022/4213335/#B49" target="_blank">49</a>]</td><td class="align_center">69.2</td><td class="align_center">88.3</td><td class="align_center">23.5</td></tr><tr><td class="align_left">S3D [<a href="/journals/misy/2022/4213335/#B26" target="_blank">26</a>]</td><td class="align_center">69.4</td><td class="align_center">89.1</td><td class="align_center">66.4</td></tr><tr><td class="align_left">ECO [<a href="/journals/misy/2022/4213335/#B50" target="_blank">50</a>]</td><td class="align_center">70.0</td><td class="align_center">89.4</td><td class="align_center">216</td></tr><tr><td class="align_left">R(<span class="nowrap"><svg height="8.69875pt" id="M82" style="vertical-align:-0.3499298pt" version="1.1" viewbox="-0.0498162 -8.34882 26.097 8.69875" width="26.097pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M412 140C382 77 369 73 315 73H129L270 222C362 320 402 379 402 466C402 571 322 635 234 635C177 635 130 609 99 576L42 495L64 475C90 514 133 568 201 568C274 568 318 519 318 435C318 349 255 267 193 193C144 135 87 78 32 23V0H405C417 45 427 89 440 131L412 140Z"></path></g><g transform="matrix(.013,0,0,-0.013,9.145,0)"><path d="M535 230V280H323V490H265V280H52V230H265V-3H323V230H535Z"></path></g><g transform="matrix(.013,0,0,-0.013,19.682,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g></svg>)</span>D [<a href="/journals/misy/2022/4213335/#B25" target="_blank">25</a>]</td><td class="align_center">73.9</td><td class="align_center">90.9</td><td class="align_center">152</td></tr><tr><td class="align_left">TSN [<a href="/journals/misy/2022/4213335/#B1" target="_blank">1</a>]</td><td class="align_center">71.3</td><td class="align_center">91.5</td><td class="align_center">33</td></tr><tr><td class="align_left">TSM [<a href="/journals/misy/2022/4213335/#B2" target="_blank">2</a>]</td><td class="align_center">75.1</td><td class="align_center">91.8</td><td class="align_center">65</td></tr><tr><td class="align_left">SlowFast <span class="nowrap"><svg height="8.55521pt" id="M83" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 32.3604 8.55521" width="32.3604pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g><g transform="matrix(.013,0,0,-0.013,6.24,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g><g transform="matrix(.013,0,0,-0.013,15.386,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,25.922,0)"><path d="M249 635C141 635 70 555 70 471C70 401 114 353 179 316C143 294 106 267 90 252C68 231 45 202 45 157C45 50 130 -12 237 -12C322 -12 435 52 435 169C435 256 372 304 303 343C349 374 375 398 383 407C401 429 411 458 411 487C411 569 344 635 249 635ZM238 603C285 603 337 567 337 482C337 422 310 385 276 358C205 393 145 426 145 500C145 552 179 603 238 603ZM248 20C183 20 125 70 125 163C125 218 158 268 206 300C284 261 355 217 355 143C355 66 308 20 248 20Z"></path></g></svg>,</span> R101 [<a href="/journals/misy/2022/4213335/#B3" target="_blank">3</a>]</td><td class="align_center">78.9</td><td class="align_center">93.5</td><td class="align_center">213</td></tr><tr><td class="align_left">SlowFast <span class="nowrap"><svg height="8.55521pt" id="M84" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 32.3604 8.55521" width="32.3604pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g><g transform="matrix(.013,0,0,-0.013,6.24,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g><g transform="matrix(.013,0,0,-0.013,15.386,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,25.922,0)"><path d="M249 635C141 635 70 555 70 471C70 401 114 353 179 316C143 294 106 267 90 252C68 231 45 202 45 157C45 50 130 -12 237 -12C322 -12 435 52 435 169C435 256 372 304 303 343C349 374 375 398 383 407C401 429 411 458 411 487C411 569 344 635 249 635ZM238 603C285 603 337 567 337 482C337 422 310 385 276 358C205 393 145 426 145 500C145 552 179 603 238 603ZM248 20C183 20 125 70 125 163C125 218 158 268 206 300C284 261 355 217 355 143C355 66 308 20 248 20Z"></path></g></svg>,</span> R101+NL [<a href="/journals/misy/2022/4213335/#B3" target="_blank">3</a>]</td><td class="align_center">79.8</td><td class="align_center">93.9</td><td class="align_center">234</td></tr><tr><td class="align_left">VCM-SDD <span class="nowrap"><svg height="8.55521pt" id="M85" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 26.097 8.55521" width="26.097pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M249 635C141 635 70 555 70 471C70 401 114 353 179 316C143 294 106 267 90 252C68 231 45 202 45 157C45 50 130 -12 237 -12C322 -12 435 52 435 169C435 256 372 304 303 343C349 374 375 398 383 407C401 429 411 458 411 487C411 569 344 635 249 635ZM238 603C285 603 337 567 337 482C337 422 310 385 276 358C205 393 145 426 145 500C145 552 179 603 238 603ZM248 20C183 20 125 70 125 163C125 218 158 268 206 300C284 261 355 217 355 143C355 66 308 20 248 20Z"></path></g><g transform="matrix(.013,0,0,-0.013,9.145,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,19.682,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g></svg>,</span> R101_NP</td><td class="align_center">77.4</td><td class="align_center">93.1</td><td class="align_center">46.8</td></tr><tr><td class="align_left">VCM-SDD <span class="nowrap"><svg height="8.55521pt" id="M86" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 26.097 8.55521" width="26.097pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M249 635C141 635 70 555 70 471C70 401 114 353 179 316C143 294 106 267 90 252C68 231 45 202 45 157C45 50 130 -12 237 -12C322 -12 435 52 435 169C435 256 372 304 303 343C349 374 375 398 383 407C401 429 411 458 411 487C411 569 344 635 249 635ZM238 603C285 603 337 567 337 482C337 422 310 385 276 358C205 393 145 426 145 500C145 552 179 603 238 603ZM248 20C183 20 125 70 125 163C125 218 158 268 206 300C284 261 355 217 355 143C355 66 308 20 248 20Z"></path></g><g transform="matrix(.013,0,0,-0.013,9.145,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,19.682,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g></svg>,</span> R101</td><td class="align_center">78.5</td><td class="align_center">93.5</td><td class="align_center">46.8</td></tr><tr><td class="align_left">VCM-SDD <span class="nowrap"><svg height="8.55521pt" id="M87" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 32.3604 8.55521" width="32.3604pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g><g transform="matrix(.013,0,0,-0.013,6.24,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g><g transform="matrix(.013,0,0,-0.013,15.386,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,25.922,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g></svg>,</span> R101_NP</td><td class="align_center">79.3</td><td class="align_center">93.9</td><td class="align_center">46.8</td></tr><tr><td class="align_left">VCM-SDD <span class="nowrap"><svg height="8.55521pt" id="M88" style="vertical-align:-0.2063904pt" version="1.1" viewbox="-0.0498162 -8.34882 32.3604 8.55521" width="32.3604pt" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"><g transform="matrix(.013,0,0,-0.013,0,0)"><path d="M384 0V27C293 34 287 42 287 114V635C232 613 172 594 109 583V559L157 557C201 555 205 550 205 499V114C205 42 199 34 109 27V0H384Z"></path></g><g transform="matrix(.013,0,0,-0.013,6.24,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g><g transform="matrix(.013,0,0,-0.013,15.386,0)"><path d="M471 153C471 170 463 194 452 212C400 220 373 229 322 255C373 281 400 290 452 298C463 316 471 339 471 357C456 366 431 371 410 370C377 329 356 310 308 279C311 336 317 364 336 413C326 432 310 451 294 459C279 451 262 432 252 413C271 364 277 336 280 279C232 310 211 329 178 370C157 371 132 367 117 357C117 340 125 316 136 298C188 290 215 281 266 255C215 229 188 220 136 212C125 194 117 171 117 153C132 144 157 139 178 140C211 181 232 200 280 231C277 174 271 146 252 97C262 78 278 59 294 51C309 59 326 78 336 97C317 146 311 174 308 231C356 200 377 181 410 140C431 139 456 143 471 153Z"></path></g><g transform="matrix(.013,0,0,-0.013,25.922,0)"><path d="M137 343C167 482 260 545 321 574C357 591 397 603 429 609L423 641C382 634 335 622 295 608C189 570 37 457 37 238C37 84 125 -12 242 -12C362 -12 447 89 447 209C447 311 374 393 267 393C247 393 226 386 204 376L137 343ZM227 337C318 337 361 256 361 173C361 105 336 22 258 22C176 22 126 120 126 240C126 266 127 291 132 310C155 323 189 337 227 337Z"></path></g></svg>,</span> R101</td><td class="align_center"><b>80.1</b></td><td class="align_center"><b>94.4</b></td><td class="align_center"><b>46.8</b></td></tr><tr class="table-tr"><td colspan="4"><hr class="tbody-hr"/></td></tr></table></td></tr></table>

<div>The comparison between this algorithm and other methods on the kinetics 400.</div>

Mobile Information Systems

tab3

Table 3

Table 3: A Video Classification Method Based on Spatiotemporal Detail Attention and Feature Fusion