| 2614 | t0 += p1 + p3; |
| 2615 | |
| 2616 | static void stbi__idct_block(stbi_uc* out, int out_stride, short data[64]) { |
| 2617 | int i, val[64], *v = val; |
| 2618 | stbi_uc* o; |
| 2619 | short* d = data; |
| 2620 | |
| 2621 | // columns |
| 2622 | for (i = 0; i < 8; ++i, ++d, ++v) { |
| 2623 | // if all zeroes, shortcut -- this avoids dequantizing 0s and IDCTing |
| 2624 | if (d[8] == 0 && d[16] == 0 && d[24] == 0 && d[32] == 0 && d[40] == 0 && |
| 2625 | d[48] == 0 && d[56] == 0) { |
| 2626 | // no shortcut 0 seconds |
| 2627 | // (1|2|3|4|5|6|7)==0 0 seconds |
| 2628 | // all separate -0.047 seconds |
| 2629 | // 1 && 2|3 && 4|5 && 6|7: -0.047 seconds |
| 2630 | int dcterm = d[0] * 4; |
| 2631 | v[0] = v[8] = v[16] = v[24] = v[32] = v[40] = v[48] = v[56] = dcterm; |
| 2632 | } else { |
| 2633 | STBI__IDCT_1D(d[0], d[8], d[16], d[24], d[32], d[40], d[48], d[56]) |
| 2634 | // constants scaled things up by 1<<12; let's bring them back |
| 2635 | // down, but keep 2 extra bits of precision |
| 2636 | x0 += 512; |
| 2637 | x1 += 512; |
| 2638 | x2 += 512; |
| 2639 | x3 += 512; |
| 2640 | v[0] = (x0 + t3) >> 10; |
| 2641 | v[56] = (x0 - t3) >> 10; |
| 2642 | v[8] = (x1 + t2) >> 10; |
| 2643 | v[48] = (x1 - t2) >> 10; |
| 2644 | v[16] = (x2 + t1) >> 10; |
| 2645 | v[40] = (x2 - t1) >> 10; |
| 2646 | v[24] = (x3 + t0) >> 10; |
| 2647 | v[32] = (x3 - t0) >> 10; |
| 2648 | } |
| 2649 | } |
| 2650 | |
| 2651 | for (i = 0, v = val, o = out; i < 8; ++i, v += 8, o += out_stride) { |
| 2652 | // no fast case since the first 1D IDCT spread components out |
| 2653 | STBI__IDCT_1D(v[0], v[1], v[2], v[3], v[4], v[5], v[6], v[7]) |
| 2654 | // constants scaled things up by 1<<12, plus we had 1<<2 from first |
| 2655 | // loop, plus horizontal and vertical each scale by sqrt(8) so together |
| 2656 | // we've got an extra 1<<3, so 1<<17 total we need to remove. |
| 2657 | // so we want to round that, which means adding 0.5 * 1<<17, |
| 2658 | // aka 65536. Also, we'll end up with -128 to 127 that we want |
| 2659 | // to encode as 0..255 by adding 128, so we'll add that before the shift |
| 2660 | x0 += 65536 + (128 << 17); |
| 2661 | x1 += 65536 + (128 << 17); |
| 2662 | x2 += 65536 + (128 << 17); |
| 2663 | x3 += 65536 + (128 << 17); |
| 2664 | // tried computing the shifts into temps, or'ing the temps to see |
| 2665 | // if any were out of range, but that was slower |
| 2666 | o[0] = stbi__clamp((x0 + t3) >> 17); |
| 2667 | o[7] = stbi__clamp((x0 - t3) >> 17); |
| 2668 | o[1] = stbi__clamp((x1 + t2) >> 17); |
| 2669 | o[6] = stbi__clamp((x1 - t2) >> 17); |
| 2670 | o[2] = stbi__clamp((x2 + t1) >> 17); |
| 2671 | o[5] = stbi__clamp((x2 - t1) >> 17); |
| 2672 | o[3] = stbi__clamp((x3 + t0) >> 17); |
| 2673 | o[4] = stbi__clamp((x3 - t0) >> 17); |
nothing calls this directly
no test coverage detected