MCPcopy Create free account
hub / github.com/ARM-software/ComputeLibrary / stbi__idct_block

Function stbi__idct_block

include/stb/stb_image.h:2467–2524  ·  view source on GitHub ↗

Source from the content-addressed store, hash-verified

2465 t0 += p1+p3;
2466
2467static void stbi__idct_block(stbi_uc *out, int out_stride, short data[64])
2468{
2469 int i,val[64],*v=val;
2470 stbi_uc *o;
2471 short *d = data;
2472
2473 // columns
2474 for (i=0; i < 8; ++i,++d, ++v) {
2475 // if all zeroes, shortcut -- this avoids dequantizing 0s and IDCTing
2476 if (d[ 8]==0 && d[16]==0 && d[24]==0 && d[32]==0
2477 && d[40]==0 && d[48]==0 && d[56]==0) {
2478 // no shortcut 0 seconds
2479 // (1|2|3|4|5|6|7)==0 0 seconds
2480 // all separate -0.047 seconds
2481 // 1 && 2|3 && 4|5 && 6|7: -0.047 seconds
2482 int dcterm = d[0]*4;
2483 v[0] = v[8] = v[16] = v[24] = v[32] = v[40] = v[48] = v[56] = dcterm;
2484 } else {
2485 STBI__IDCT_1D(d[ 0],d[ 8],d[16],d[24],d[32],d[40],d[48],d[56])
2486 // constants scaled things up by 1<<12; let's bring them back
2487 // down, but keep 2 extra bits of precision
2488 x0 += 512; x1 += 512; x2 += 512; x3 += 512;
2489 v[ 0] = (x0+t3) >> 10;
2490 v[56] = (x0-t3) >> 10;
2491 v[ 8] = (x1+t2) >> 10;
2492 v[48] = (x1-t2) >> 10;
2493 v[16] = (x2+t1) >> 10;
2494 v[40] = (x2-t1) >> 10;
2495 v[24] = (x3+t0) >> 10;
2496 v[32] = (x3-t0) >> 10;
2497 }
2498 }
2499
2500 for (i=0, v=val, o=out; i < 8; ++i,v+=8,o+=out_stride) {
2501 // no fast case since the first 1D IDCT spread components out
2502 STBI__IDCT_1D(v[0],v[1],v[2],v[3],v[4],v[5],v[6],v[7])
2503 // constants scaled things up by 1<<12, plus we had 1<<2 from first
2504 // loop, plus horizontal and vertical each scale by sqrt(8) so together
2505 // we've got an extra 1<<3, so 1<<17 total we need to remove.
2506 // so we want to round that, which means adding 0.5 * 1<<17,
2507 // aka 65536. Also, we'll end up with -128 to 127 that we want
2508 // to encode as 0..255 by adding 128, so we'll add that before the shift
2509 x0 += 65536 + (128<<17);
2510 x1 += 65536 + (128<<17);
2511 x2 += 65536 + (128<<17);
2512 x3 += 65536 + (128<<17);
2513 // tried computing the shifts into temps, or'ing the temps to see
2514 // if any were out of range, but that was slower
2515 o[0] = stbi__clamp((x0+t3) >> 17);
2516 o[7] = stbi__clamp((x0-t3) >> 17);
2517 o[1] = stbi__clamp((x1+t2) >> 17);
2518 o[6] = stbi__clamp((x1-t2) >> 17);
2519 o[2] = stbi__clamp((x2+t1) >> 17);
2520 o[5] = stbi__clamp((x2-t1) >> 17);
2521 o[3] = stbi__clamp((x3+t0) >> 17);
2522 o[4] = stbi__clamp((x3-t0) >> 17);
2523 }
2524}

Callers

nothing calls this directly

Calls 1

stbi__clampFunction · 0.85

Tested by

no test coverage detected