2017-12-14 13:51:45 +01:00
|
|
|
/*
|
|
|
|
* Copyright 2012 Advanced Micro Devices, Inc.
|
|
|
|
*
|
|
|
|
* Permission is hereby granted, free of charge, to any person obtaining a
|
|
|
|
* copy of this software and associated documentation files (the "Software"),
|
|
|
|
* to deal in the Software without restriction, including without limitation
|
|
|
|
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
|
|
|
|
* and/or sell copies of the Software, and to permit persons to whom the
|
|
|
|
* Software is furnished to do so, subject to the following conditions:
|
|
|
|
*
|
|
|
|
* The above copyright notice and this permission notice (including the next
|
|
|
|
* paragraph) shall be included in all copies or substantial portions of the
|
|
|
|
* Software.
|
|
|
|
*
|
|
|
|
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
|
|
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
|
|
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
|
|
|
|
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
|
|
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
|
|
|
* FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
|
|
|
|
* IN THE SOFTWARE.
|
|
|
|
*/
|
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
#include "ac_shader_util.h"
|
|
|
|
|
|
|
|
#include "sid.h"
|
|
|
|
|
2017-12-15 15:37:18 +01:00
|
|
|
#include <assert.h>
|
2017-12-21 17:53:15 +01:00
|
|
|
#include <stdlib.h>
|
|
|
|
#include <string.h>
|
2017-12-15 15:37:18 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned ac_get_spi_shader_z_format(bool writes_z, bool writes_stencil, bool writes_samplemask)
|
2017-12-14 13:51:45 +01:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
if (writes_z) {
|
|
|
|
/* Z needs 32 bits. */
|
|
|
|
if (writes_samplemask)
|
|
|
|
return V_028710_SPI_SHADER_32_ABGR;
|
|
|
|
else if (writes_stencil)
|
|
|
|
return V_028710_SPI_SHADER_32_GR;
|
|
|
|
else
|
|
|
|
return V_028710_SPI_SHADER_32_R;
|
|
|
|
} else if (writes_stencil || writes_samplemask) {
|
|
|
|
/* Both stencil and sample mask need only 16 bits. */
|
|
|
|
return V_028710_SPI_SHADER_UINT16_ABGR;
|
|
|
|
} else {
|
|
|
|
return V_028710_SPI_SHADER_ZERO;
|
|
|
|
}
|
2017-12-14 13:51:45 +01:00
|
|
|
}
|
2017-12-15 15:37:18 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned ac_get_cb_shader_mask(unsigned spi_shader_col_format)
|
2017-12-15 15:37:18 +01:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned i, cb_shader_mask = 0;
|
2017-12-15 15:37:18 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
for (i = 0; i < 8; i++) {
|
|
|
|
switch ((spi_shader_col_format >> (i * 4)) & 0xf) {
|
|
|
|
case V_028714_SPI_SHADER_ZERO:
|
|
|
|
break;
|
|
|
|
case V_028714_SPI_SHADER_32_R:
|
|
|
|
cb_shader_mask |= 0x1 << (i * 4);
|
|
|
|
break;
|
|
|
|
case V_028714_SPI_SHADER_32_GR:
|
|
|
|
cb_shader_mask |= 0x3 << (i * 4);
|
|
|
|
break;
|
|
|
|
case V_028714_SPI_SHADER_32_AR:
|
|
|
|
cb_shader_mask |= 0x9u << (i * 4);
|
|
|
|
break;
|
|
|
|
case V_028714_SPI_SHADER_FP16_ABGR:
|
|
|
|
case V_028714_SPI_SHADER_UNORM16_ABGR:
|
|
|
|
case V_028714_SPI_SHADER_SNORM16_ABGR:
|
|
|
|
case V_028714_SPI_SHADER_UINT16_ABGR:
|
|
|
|
case V_028714_SPI_SHADER_SINT16_ABGR:
|
|
|
|
case V_028714_SPI_SHADER_32_ABGR:
|
|
|
|
cb_shader_mask |= 0xfu << (i * 4);
|
|
|
|
break;
|
|
|
|
default:
|
|
|
|
assert(0);
|
|
|
|
}
|
|
|
|
}
|
|
|
|
return cb_shader_mask;
|
2017-12-15 15:37:18 +01:00
|
|
|
}
|
2017-12-15 15:37:19 +01:00
|
|
|
|
|
|
|
/**
|
|
|
|
* Calculate the appropriate setting of VGT_GS_MODE when \p shader is a
|
|
|
|
* geometry shader.
|
|
|
|
*/
|
2020-09-07 09:58:36 +02:00
|
|
|
uint32_t ac_vgt_gs_mode(unsigned gs_max_vert_out, enum chip_class chip_class)
|
2017-12-15 15:37:19 +01:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned cut_mode;
|
2017-12-15 15:37:19 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
if (gs_max_vert_out <= 128) {
|
|
|
|
cut_mode = V_028A40_GS_CUT_128;
|
|
|
|
} else if (gs_max_vert_out <= 256) {
|
|
|
|
cut_mode = V_028A40_GS_CUT_256;
|
|
|
|
} else if (gs_max_vert_out <= 512) {
|
|
|
|
cut_mode = V_028A40_GS_CUT_512;
|
|
|
|
} else {
|
|
|
|
assert(gs_max_vert_out <= 1024);
|
|
|
|
cut_mode = V_028A40_GS_CUT_1024;
|
|
|
|
}
|
2017-12-15 15:37:19 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
return S_028A40_MODE(V_028A40_GS_SCENARIO_G) | S_028A40_CUT_MODE(cut_mode) |
|
|
|
|
S_028A40_ES_WRITE_OPTIMIZE(chip_class <= GFX8) | S_028A40_GS_WRITE_OPTIMIZE(1) |
|
|
|
|
S_028A40_ONCHIP(chip_class >= GFX9 ? 1 : 0);
|
2017-12-15 15:37:19 +01:00
|
|
|
}
|
2017-12-21 17:53:15 +01:00
|
|
|
|
2019-09-25 14:10:18 +02:00
|
|
|
/// Translate a (dfmt, nfmt) pair into a chip-appropriate combined format
|
|
|
|
/// value for LLVM8+ tbuffer intrinsics.
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned ac_get_tbuffer_format(enum chip_class chip_class, unsigned dfmt, unsigned nfmt)
|
2019-09-25 14:10:18 +02:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
// Some games try to access vertex buffers without a valid format.
|
|
|
|
// This is a game bug, but we should still handle it gracefully.
|
|
|
|
if (dfmt == V_008F0C_IMG_FORMAT_INVALID)
|
|
|
|
return V_008F0C_IMG_FORMAT_INVALID;
|
2019-11-06 13:29:26 +01:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
if (chip_class >= GFX10) {
|
|
|
|
unsigned format;
|
|
|
|
switch (dfmt) {
|
|
|
|
default:
|
|
|
|
unreachable("bad dfmt");
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_INVALID:
|
|
|
|
format = V_008F0C_IMG_FORMAT_INVALID;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_8:
|
|
|
|
format = V_008F0C_IMG_FORMAT_8_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_8_8:
|
|
|
|
format = V_008F0C_IMG_FORMAT_8_8_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_8_8_8_8:
|
|
|
|
format = V_008F0C_IMG_FORMAT_8_8_8_8_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_16:
|
|
|
|
format = V_008F0C_IMG_FORMAT_16_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_16_16:
|
|
|
|
format = V_008F0C_IMG_FORMAT_16_16_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_16_16_16_16:
|
|
|
|
format = V_008F0C_IMG_FORMAT_16_16_16_16_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_32:
|
|
|
|
format = V_008F0C_IMG_FORMAT_32_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_32_32:
|
|
|
|
format = V_008F0C_IMG_FORMAT_32_32_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_32_32_32:
|
|
|
|
format = V_008F0C_IMG_FORMAT_32_32_32_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_32_32_32_32:
|
|
|
|
format = V_008F0C_IMG_FORMAT_32_32_32_32_UINT;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_DATA_FORMAT_2_10_10_10:
|
|
|
|
format = V_008F0C_IMG_FORMAT_2_10_10_10_UINT;
|
|
|
|
break;
|
2021-04-14 10:25:43 +02:00
|
|
|
case V_008F0C_BUF_DATA_FORMAT_10_11_11:
|
|
|
|
format = V_008F0C_IMG_FORMAT_10_11_11_UINT;
|
|
|
|
break;
|
2020-09-07 09:58:36 +02:00
|
|
|
}
|
2019-09-25 14:10:18 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
// Use the regularity properties of the combined format enum.
|
|
|
|
//
|
|
|
|
// Note: float is incompatible with 8-bit data formats,
|
|
|
|
// [us]{norm,scaled} are incomparible with 32-bit data formats.
|
|
|
|
// [us]scaled are not writable.
|
|
|
|
switch (nfmt) {
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_UNORM:
|
|
|
|
format -= 4;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_SNORM:
|
|
|
|
format -= 3;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_USCALED:
|
|
|
|
format -= 2;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_SSCALED:
|
|
|
|
format -= 1;
|
|
|
|
break;
|
|
|
|
default:
|
|
|
|
unreachable("bad nfmt");
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_UINT:
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_SINT:
|
|
|
|
format += 1;
|
|
|
|
break;
|
|
|
|
case V_008F0C_BUF_NUM_FORMAT_FLOAT:
|
|
|
|
format += 2;
|
|
|
|
break;
|
|
|
|
}
|
2019-09-25 14:10:18 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
return format;
|
|
|
|
} else {
|
|
|
|
return dfmt | (nfmt << 4);
|
|
|
|
}
|
2019-09-25 14:10:18 +02:00
|
|
|
}
|
|
|
|
|
2020-01-14 13:01:53 +00:00
|
|
|
static const struct ac_data_format_info data_format_table[] = {
|
2020-09-07 09:58:36 +02:00
|
|
|
[V_008F0C_BUF_DATA_FORMAT_INVALID] = {0, 4, 0, V_008F0C_BUF_DATA_FORMAT_INVALID},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_8] = {1, 1, 1, V_008F0C_BUF_DATA_FORMAT_8},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_16] = {2, 1, 2, V_008F0C_BUF_DATA_FORMAT_16},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_8_8] = {2, 2, 1, V_008F0C_BUF_DATA_FORMAT_8},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_32] = {4, 1, 4, V_008F0C_BUF_DATA_FORMAT_32},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_16_16] = {4, 2, 2, V_008F0C_BUF_DATA_FORMAT_16},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_10_11_11] = {4, 3, 0, V_008F0C_BUF_DATA_FORMAT_10_11_11},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_11_11_10] = {4, 3, 0, V_008F0C_BUF_DATA_FORMAT_11_11_10},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_10_10_10_2] = {4, 4, 0, V_008F0C_BUF_DATA_FORMAT_10_10_10_2},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_2_10_10_10] = {4, 4, 0, V_008F0C_BUF_DATA_FORMAT_2_10_10_10},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_8_8_8_8] = {4, 4, 1, V_008F0C_BUF_DATA_FORMAT_8},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_32_32] = {8, 2, 4, V_008F0C_BUF_DATA_FORMAT_32},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_16_16_16_16] = {8, 4, 2, V_008F0C_BUF_DATA_FORMAT_16},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_32_32_32] = {12, 3, 4, V_008F0C_BUF_DATA_FORMAT_32},
|
|
|
|
[V_008F0C_BUF_DATA_FORMAT_32_32_32_32] = {16, 4, 4, V_008F0C_BUF_DATA_FORMAT_32},
|
2020-01-14 13:01:53 +00:00
|
|
|
};
|
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
const struct ac_data_format_info *ac_get_data_format_info(unsigned dfmt)
|
2020-01-14 13:01:53 +00:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
assert(dfmt < ARRAY_SIZE(data_format_table));
|
|
|
|
return &data_format_table[dfmt];
|
2020-01-14 13:01:53 +00:00
|
|
|
}
|
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
enum ac_image_dim ac_get_sampler_dim(enum chip_class chip_class, enum glsl_sampler_dim dim,
|
|
|
|
bool is_array)
|
2019-09-25 14:10:18 +02:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
switch (dim) {
|
|
|
|
case GLSL_SAMPLER_DIM_1D:
|
|
|
|
if (chip_class == GFX9)
|
|
|
|
return is_array ? ac_image_2darray : ac_image_2d;
|
|
|
|
return is_array ? ac_image_1darray : ac_image_1d;
|
|
|
|
case GLSL_SAMPLER_DIM_2D:
|
|
|
|
case GLSL_SAMPLER_DIM_RECT:
|
|
|
|
case GLSL_SAMPLER_DIM_EXTERNAL:
|
|
|
|
return is_array ? ac_image_2darray : ac_image_2d;
|
|
|
|
case GLSL_SAMPLER_DIM_3D:
|
|
|
|
return ac_image_3d;
|
|
|
|
case GLSL_SAMPLER_DIM_CUBE:
|
|
|
|
return ac_image_cube;
|
|
|
|
case GLSL_SAMPLER_DIM_MS:
|
|
|
|
return is_array ? ac_image_2darraymsaa : ac_image_2dmsaa;
|
|
|
|
case GLSL_SAMPLER_DIM_SUBPASS:
|
|
|
|
return ac_image_2darray;
|
|
|
|
case GLSL_SAMPLER_DIM_SUBPASS_MS:
|
|
|
|
return ac_image_2darraymsaa;
|
|
|
|
default:
|
|
|
|
unreachable("bad sampler dim");
|
|
|
|
}
|
2019-09-25 14:10:18 +02:00
|
|
|
}
|
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
enum ac_image_dim ac_get_image_dim(enum chip_class chip_class, enum glsl_sampler_dim sdim,
|
|
|
|
bool is_array)
|
2019-09-25 14:10:18 +02:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
enum ac_image_dim dim = ac_get_sampler_dim(chip_class, sdim, is_array);
|
2019-09-25 14:10:18 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
/* Match the resource type set in the descriptor. */
|
|
|
|
if (dim == ac_image_cube || (chip_class <= GFX8 && dim == ac_image_3d))
|
|
|
|
dim = ac_image_2darray;
|
|
|
|
else if (sdim == GLSL_SAMPLER_DIM_2D && !is_array && chip_class == GFX9) {
|
|
|
|
/* When a single layer of a 3D texture is bound, the shader
|
|
|
|
* will refer to a 2D target, but the descriptor has a 3D type.
|
|
|
|
* Since the HW ignores BASE_ARRAY in this case, we need to
|
|
|
|
* send 3 coordinates. This doesn't hurt when the underlying
|
|
|
|
* texture is non-3D.
|
|
|
|
*/
|
|
|
|
dim = ac_image_3d;
|
|
|
|
}
|
2019-09-25 14:10:18 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
return dim;
|
2019-09-25 14:10:18 +02:00
|
|
|
}
|
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned ac_get_fs_input_vgpr_cnt(const struct ac_shader_config *config,
|
|
|
|
signed char *face_vgpr_index_ptr,
|
|
|
|
signed char *ancillary_vgpr_index_ptr)
|
2019-09-25 16:40:07 +02:00
|
|
|
{
|
2020-09-07 09:58:36 +02:00
|
|
|
unsigned num_input_vgprs = 0;
|
|
|
|
signed char face_vgpr_index = -1;
|
|
|
|
signed char ancillary_vgpr_index = -1;
|
2019-09-25 16:40:07 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
if (G_0286CC_PERSP_SAMPLE_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_PERSP_CENTER_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_PERSP_CENTROID_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_PERSP_PULL_MODEL_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 3;
|
|
|
|
if (G_0286CC_LINEAR_SAMPLE_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_LINEAR_CENTER_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_LINEAR_CENTROID_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 2;
|
|
|
|
if (G_0286CC_LINE_STIPPLE_TEX_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_POS_X_FLOAT_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_POS_Y_FLOAT_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_POS_Z_FLOAT_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_POS_W_FLOAT_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_FRONT_FACE_ENA(config->spi_ps_input_addr)) {
|
|
|
|
face_vgpr_index = num_input_vgprs;
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
}
|
|
|
|
if (G_0286CC_ANCILLARY_ENA(config->spi_ps_input_addr)) {
|
|
|
|
ancillary_vgpr_index = num_input_vgprs;
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
}
|
|
|
|
if (G_0286CC_SAMPLE_COVERAGE_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
|
|
|
if (G_0286CC_POS_FIXED_PT_ENA(config->spi_ps_input_addr))
|
|
|
|
num_input_vgprs += 1;
|
2019-09-25 16:40:07 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
if (face_vgpr_index_ptr)
|
|
|
|
*face_vgpr_index_ptr = face_vgpr_index;
|
|
|
|
if (ancillary_vgpr_index_ptr)
|
|
|
|
*ancillary_vgpr_index_ptr = ancillary_vgpr_index;
|
2019-09-25 16:40:07 +02:00
|
|
|
|
2020-09-07 09:58:36 +02:00
|
|
|
return num_input_vgprs;
|
2019-09-25 16:40:07 +02:00
|
|
|
}
|
2020-06-11 22:25:53 +02:00
|
|
|
|
ac,radv: use better export formats for 8-bit when RB+ isn't allowed
When RB+ is enabled, R8_UINT/R8_SINT/R8_UNORM should use FP16_ABGR
for 2x exporting performance. Otherwise, use 32_R to remove useless
instructions needed for 16-bit compressed exports.
fossils-db (Vega10):
Totals from 8858 (6.35% of 139517) affected shaders:
SGPRs: 801248 -> 801210 (-0.00%); split: -0.01%, +0.00%
VGPRs: 596224 -> 596120 (-0.02%); split: -0.02%, +0.01%
CodeSize: 71462452 -> 71356684 (-0.15%); split: -0.15%, +0.00%
MaxWaves: 37097 -> 37105 (+0.02%); split: +0.04%, -0.02%
Instrs: 13963177 -> 13950809 (-0.09%); split: -0.09%, +0.00%
Cycles: 1476539360 -> 1476489996 (-0.00%); split: -0.00%, +0.00%
VMEM: 2363008 -> 2361349 (-0.07%); split: +0.04%, -0.11%
SMEM: 550362 -> 549977 (-0.07%); split: +0.01%, -0.08%
VClause: 245704 -> 245727 (+0.01%); split: -0.01%, +0.02%
SClause: 485161 -> 485104 (-0.01%); split: -0.01%, +0.00%
Copies: 1420034 -> 1422310 (+0.16%); split: -0.01%, +0.17%
Branches: 518710 -> 518705 (-0.00%)
PreSGPRs: 706633 -> 706584 (-0.01%)
PreVGPRs: 547163 -> 547007 (-0.03%); split: -0.03%, +0.01%
Signed-off-by: Samuel Pitoiset <samuel.pitoiset@gmail.com>
Reviewed-by: Bas Nieuwenhuizen <bas@basnieuwenhuizen.nl>
Part-of: <https://gitlab.freedesktop.org/mesa/mesa/-/merge_requests/7512>
2020-11-16 08:57:59 +01:00
|
|
|
void ac_choose_spi_color_formats(unsigned format, unsigned swap, unsigned ntype,
|
|
|
|
bool is_depth, bool use_rbplus,
|
2020-09-07 09:58:36 +02:00
|
|
|
struct ac_spi_color_formats *formats)
|
2020-06-11 22:25:53 +02:00
|
|
|
{
|
|
|
|
/* Alpha is needed for alpha-to-coverage.
|
|
|
|
* Blending may be with or without alpha.
|
|
|
|
*/
|
|
|
|
unsigned normal = 0; /* most optimal, may not support blending or export alpha */
|
|
|
|
unsigned alpha = 0; /* exports alpha, but may not support blending */
|
|
|
|
unsigned blend = 0; /* supports blending, but may not export alpha */
|
|
|
|
unsigned blend_alpha = 0; /* least optimal, supports blending and exports alpha */
|
|
|
|
|
|
|
|
/* Choose the SPI color formats. These are required values for RB+.
|
|
|
|
* Other chips have multiple choices, though they are not necessarily better.
|
|
|
|
*/
|
|
|
|
switch (format) {
|
|
|
|
case V_028C70_COLOR_5_6_5:
|
|
|
|
case V_028C70_COLOR_1_5_5_5:
|
|
|
|
case V_028C70_COLOR_5_5_5_1:
|
|
|
|
case V_028C70_COLOR_4_4_4_4:
|
|
|
|
case V_028C70_COLOR_10_11_11:
|
|
|
|
case V_028C70_COLOR_11_11_10:
|
|
|
|
case V_028C70_COLOR_5_9_9_9:
|
|
|
|
case V_028C70_COLOR_8:
|
|
|
|
case V_028C70_COLOR_8_8:
|
|
|
|
case V_028C70_COLOR_8_8_8_8:
|
|
|
|
case V_028C70_COLOR_10_10_10_2:
|
|
|
|
case V_028C70_COLOR_2_10_10_10:
|
|
|
|
if (ntype == V_028C70_NUMBER_UINT)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_UINT16_ABGR;
|
|
|
|
else if (ntype == V_028C70_NUMBER_SINT)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_SINT16_ABGR;
|
|
|
|
else
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_FP16_ABGR;
|
ac,radv: use better export formats for 8-bit when RB+ isn't allowed
When RB+ is enabled, R8_UINT/R8_SINT/R8_UNORM should use FP16_ABGR
for 2x exporting performance. Otherwise, use 32_R to remove useless
instructions needed for 16-bit compressed exports.
fossils-db (Vega10):
Totals from 8858 (6.35% of 139517) affected shaders:
SGPRs: 801248 -> 801210 (-0.00%); split: -0.01%, +0.00%
VGPRs: 596224 -> 596120 (-0.02%); split: -0.02%, +0.01%
CodeSize: 71462452 -> 71356684 (-0.15%); split: -0.15%, +0.00%
MaxWaves: 37097 -> 37105 (+0.02%); split: +0.04%, -0.02%
Instrs: 13963177 -> 13950809 (-0.09%); split: -0.09%, +0.00%
Cycles: 1476539360 -> 1476489996 (-0.00%); split: -0.00%, +0.00%
VMEM: 2363008 -> 2361349 (-0.07%); split: +0.04%, -0.11%
SMEM: 550362 -> 549977 (-0.07%); split: +0.01%, -0.08%
VClause: 245704 -> 245727 (+0.01%); split: -0.01%, +0.02%
SClause: 485161 -> 485104 (-0.01%); split: -0.01%, +0.00%
Copies: 1420034 -> 1422310 (+0.16%); split: -0.01%, +0.17%
Branches: 518710 -> 518705 (-0.00%)
PreSGPRs: 706633 -> 706584 (-0.01%)
PreVGPRs: 547163 -> 547007 (-0.03%); split: -0.03%, +0.01%
Signed-off-by: Samuel Pitoiset <samuel.pitoiset@gmail.com>
Reviewed-by: Bas Nieuwenhuizen <bas@basnieuwenhuizen.nl>
Part-of: <https://gitlab.freedesktop.org/mesa/mesa/-/merge_requests/7512>
2020-11-16 08:57:59 +01:00
|
|
|
|
|
|
|
if (!use_rbplus && format == V_028C70_COLOR_8 &&
|
|
|
|
ntype != V_028C70_NUMBER_SRGB && swap == V_028C70_SWAP_STD) /* R */ {
|
|
|
|
/* When RB+ is enabled, R8_UNORM should use FP16_ABGR for 2x
|
|
|
|
* exporting performance. Otherwise, use 32_R to remove useless
|
|
|
|
* instructions needed for 16-bit compressed exports.
|
|
|
|
*/
|
|
|
|
blend = normal = V_028714_SPI_SHADER_32_R;
|
|
|
|
}
|
2020-06-11 22:25:53 +02:00
|
|
|
break;
|
|
|
|
|
|
|
|
case V_028C70_COLOR_16:
|
|
|
|
case V_028C70_COLOR_16_16:
|
|
|
|
case V_028C70_COLOR_16_16_16_16:
|
|
|
|
if (ntype == V_028C70_NUMBER_UNORM || ntype == V_028C70_NUMBER_SNORM) {
|
|
|
|
/* UNORM16 and SNORM16 don't support blending */
|
|
|
|
if (ntype == V_028C70_NUMBER_UNORM)
|
|
|
|
normal = alpha = V_028714_SPI_SHADER_UNORM16_ABGR;
|
|
|
|
else
|
|
|
|
normal = alpha = V_028714_SPI_SHADER_SNORM16_ABGR;
|
|
|
|
|
|
|
|
/* Use 32 bits per channel for blending. */
|
|
|
|
if (format == V_028C70_COLOR_16) {
|
|
|
|
if (swap == V_028C70_SWAP_STD) { /* R */
|
|
|
|
blend = V_028714_SPI_SHADER_32_R;
|
|
|
|
blend_alpha = V_028714_SPI_SHADER_32_AR;
|
|
|
|
} else if (swap == V_028C70_SWAP_ALT_REV) /* A */
|
|
|
|
blend = blend_alpha = V_028714_SPI_SHADER_32_AR;
|
|
|
|
else
|
|
|
|
assert(0);
|
|
|
|
} else if (format == V_028C70_COLOR_16_16) {
|
|
|
|
if (swap == V_028C70_SWAP_STD) { /* RG */
|
|
|
|
blend = V_028714_SPI_SHADER_32_GR;
|
|
|
|
blend_alpha = V_028714_SPI_SHADER_32_ABGR;
|
|
|
|
} else if (swap == V_028C70_SWAP_ALT) /* RA */
|
|
|
|
blend = blend_alpha = V_028714_SPI_SHADER_32_AR;
|
|
|
|
else
|
|
|
|
assert(0);
|
|
|
|
} else /* 16_16_16_16 */
|
|
|
|
blend = blend_alpha = V_028714_SPI_SHADER_32_ABGR;
|
|
|
|
} else if (ntype == V_028C70_NUMBER_UINT)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_UINT16_ABGR;
|
|
|
|
else if (ntype == V_028C70_NUMBER_SINT)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_SINT16_ABGR;
|
|
|
|
else if (ntype == V_028C70_NUMBER_FLOAT)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_FP16_ABGR;
|
|
|
|
else
|
|
|
|
assert(0);
|
|
|
|
break;
|
|
|
|
|
|
|
|
case V_028C70_COLOR_32:
|
|
|
|
if (swap == V_028C70_SWAP_STD) { /* R */
|
|
|
|
blend = normal = V_028714_SPI_SHADER_32_R;
|
|
|
|
alpha = blend_alpha = V_028714_SPI_SHADER_32_AR;
|
|
|
|
} else if (swap == V_028C70_SWAP_ALT_REV) /* A */
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_32_AR;
|
|
|
|
else
|
|
|
|
assert(0);
|
|
|
|
break;
|
|
|
|
|
|
|
|
case V_028C70_COLOR_32_32:
|
|
|
|
if (swap == V_028C70_SWAP_STD) { /* RG */
|
|
|
|
blend = normal = V_028714_SPI_SHADER_32_GR;
|
|
|
|
alpha = blend_alpha = V_028714_SPI_SHADER_32_ABGR;
|
|
|
|
} else if (swap == V_028C70_SWAP_ALT) /* RA */
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_32_AR;
|
|
|
|
else
|
|
|
|
assert(0);
|
|
|
|
break;
|
|
|
|
|
|
|
|
case V_028C70_COLOR_32_32_32_32:
|
|
|
|
case V_028C70_COLOR_8_24:
|
|
|
|
case V_028C70_COLOR_24_8:
|
|
|
|
case V_028C70_COLOR_X24_8_32_FLOAT:
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_32_ABGR;
|
|
|
|
break;
|
|
|
|
|
|
|
|
default:
|
|
|
|
assert(0);
|
|
|
|
return;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* The DB->CB copy needs 32_ABGR. */
|
|
|
|
if (is_depth)
|
|
|
|
alpha = blend = blend_alpha = normal = V_028714_SPI_SHADER_32_ABGR;
|
|
|
|
|
|
|
|
formats->normal = normal;
|
|
|
|
formats->alpha = alpha;
|
|
|
|
formats->blend = blend;
|
|
|
|
formats->blend_alpha = blend_alpha;
|
|
|
|
}
|