mirror of
https://github.com/FFmpeg/FFmpeg.git
synced 2026-08-10 17:14:41 +00:00
The added rounding at the final output conforms to the SMPTE document and reduces the deviation against the software decoder.
160 lines
5.0 KiB
Plaintext
160 lines
5.0 KiB
Plaintext
/*
|
|
* This file is part of FFmpeg.
|
|
*
|
|
* FFmpeg is free software; you can redistribute it and/or
|
|
* modify it under the terms of the GNU Lesser General Public
|
|
* License as published by the Free Software Foundation; either
|
|
* version 2.1 of the License, or (at your option) any later version.
|
|
*
|
|
* FFmpeg is distributed in the hope that it will be useful,
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
|
* Lesser General Public License for more details.
|
|
*
|
|
* You should have received a copy of the GNU Lesser General Public
|
|
* License along with FFmpeg; if not, write to the Free Software
|
|
* Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
|
|
*/
|
|
|
|
/* Two macroblocks, padded to avoid bank conflicts */
|
|
shared float blocks[4*2][8*(8+1)];
|
|
|
|
uint get_px(uint tex_idx, ivec2 pos)
|
|
{
|
|
#ifndef INTERLACED
|
|
return imageLoad(dst[tex_idx], pos).x;
|
|
#else
|
|
return imageLoad(dst[tex_idx], ivec2(pos.x, (pos.y << 1) + bottom_field)).x;
|
|
#endif
|
|
}
|
|
|
|
void put_px(uint tex_idx, ivec2 pos, uint v)
|
|
{
|
|
#ifndef INTERLACED
|
|
imageStore(dst[tex_idx], pos, uvec4(v));
|
|
#else
|
|
imageStore(dst[tex_idx], ivec2(pos.x, (pos.y << 1) + bottom_field), uvec4(v));
|
|
#endif
|
|
}
|
|
|
|
const float idct_8x8_scales[] = {
|
|
0.353553390593274f, // cos(4 * pi/16) / 2
|
|
0.490392640201615f, // cos(1 * pi/16) / 2
|
|
0.461939766255643f, // cos(2 * pi/16) / 2
|
|
0.415734806151273f, // cos(3 * pi/16) / 2
|
|
0.353553390593274f, // cos(4 * pi/16) / 2
|
|
0.277785116509801f, // cos(5 * pi/16) / 2
|
|
0.191341716182545f, // cos(6 * pi/16) / 2
|
|
0.097545161008064f, // cos(7 * pi/16) / 2
|
|
};
|
|
|
|
/* 7.4 Inverse Transform */
|
|
void idct(uint block, uint offset, uint stride)
|
|
{
|
|
float t0, t1, t2, t3, t4, t5, t6, t7, u8;
|
|
float u0, u1, u2, u3, u4, u5, u6, u7;
|
|
|
|
/* Input */
|
|
t0 = blocks[block][0*stride + offset];
|
|
u4 = blocks[block][1*stride + offset];
|
|
t2 = blocks[block][2*stride + offset];
|
|
u6 = blocks[block][3*stride + offset];
|
|
t1 = blocks[block][4*stride + offset];
|
|
u5 = blocks[block][5*stride + offset];
|
|
t3 = blocks[block][6*stride + offset];
|
|
u7 = blocks[block][7*stride + offset];
|
|
|
|
/* Embedded scaled inverse 4-point Type-II DCT */
|
|
u0 = t0 + t1;
|
|
u1 = t0 - t1;
|
|
u3 = t2 + t3;
|
|
u2 = (t2 - t3)*(1.4142135623730950488016887242097f) - u3;
|
|
t0 = u0 + u3;
|
|
t3 = u0 - u3;
|
|
t1 = u1 + u2;
|
|
t2 = u1 - u2;
|
|
|
|
/* Embedded scaled inverse 4-point Type-IV DST */
|
|
t5 = u5 + u6;
|
|
t6 = u5 - u6;
|
|
t7 = u4 + u7;
|
|
t4 = u4 - u7;
|
|
u7 = t7 + t5;
|
|
u5 = (t7 - t5)*(1.4142135623730950488016887242097f);
|
|
u8 = (t4 + t6)*(1.8477590650225735122563663787936f);
|
|
u4 = u8 - t4*(1.0823922002923939687994464107328f);
|
|
u6 = u8 - t6*(2.6131259297527530557132863468544f);
|
|
t7 = u7;
|
|
t6 = t7 - u6;
|
|
t5 = t6 + u5;
|
|
t4 = t5 - u4;
|
|
|
|
/* Butterflies */
|
|
u0 = t0 + t7;
|
|
u7 = t0 - t7;
|
|
u6 = t1 + t6;
|
|
u1 = t1 - t6;
|
|
u2 = t2 + t5;
|
|
u5 = t2 - t5;
|
|
u4 = t3 + t4;
|
|
u3 = t3 - t4;
|
|
|
|
/* Output */
|
|
blocks[block][0*stride + offset] = u0;
|
|
blocks[block][1*stride + offset] = u1;
|
|
blocks[block][2*stride + offset] = u2;
|
|
blocks[block][3*stride + offset] = u3;
|
|
blocks[block][4*stride + offset] = u4;
|
|
blocks[block][5*stride + offset] = u5;
|
|
blocks[block][6*stride + offset] = u6;
|
|
blocks[block][7*stride + offset] = u7;
|
|
}
|
|
|
|
void main(void)
|
|
{
|
|
uvec3 gid = gl_GlobalInvocationID, lid = gl_LocalInvocationID;
|
|
uint comp = gid.z, block = (lid.y << 2) | (lid.x >> 3), idx = lid.x & 0x7;
|
|
uint chroma_shift = comp != 0 ? log2_chroma_w : 0;
|
|
bool act = gid.x < mb_width << (4 - chroma_shift);
|
|
|
|
/* Coalesced load of DCT coeffs in shared memory, inverse quantization */
|
|
if (act) {
|
|
/**
|
|
* According to the VK spec indexing an array in push constant memory with
|
|
* a non-dynamically uniform value is illegal ($15.9.1 in v1.4.326),
|
|
* so copy the whole matrix locally.
|
|
*/
|
|
uint8_t[64] qmat = comp == 0 ? qmat_luma : qmat_chroma;
|
|
|
|
/* Table 15 */
|
|
uint8_t qidx = quant_idx[(gid.y >> 1) * mb_width + (gid.x >> 4)];
|
|
int qscale = qidx > 128 ? (qidx - 96) << 2 : qidx;
|
|
|
|
[[unroll]] for (uint i = 0; i < 8; ++i) {
|
|
int c = sign_extend(int(get_px(comp, ivec2(gid.x, (gid.y << 3) + i))), 16);
|
|
float v = float(c * qscale * int(qmat[(i << 3) + idx]));
|
|
blocks[block][i * 9 + idx] = v * idct_8x8_scales[idx] * idct_8x8_scales[i];
|
|
}
|
|
}
|
|
|
|
/* Row-wise iDCT */
|
|
barrier();
|
|
idct(block, idx * 9, 1);
|
|
|
|
/* Column-wise iDCT */
|
|
barrier();
|
|
idct(block, idx, 9);
|
|
|
|
float fact = 1.0f / (1 << (12 - depth)), off = 1 << (depth - 1);
|
|
int maxv = (1 << depth) - 1;
|
|
|
|
/* 7.5.1 Color Component Samples. Rescale, clamp and write back to global memory */
|
|
barrier();
|
|
if (act) {
|
|
[[unroll]] for (uint i = 0; i < 8; ++i) {
|
|
float v = round(blocks[block][i * 9 + idx] * fact + off);
|
|
put_px(comp, ivec2(gid.x, (gid.y << 3) + i), clamp(int(v), 0, maxv));
|
|
}
|
|
}
|
|
}
|