2
* Copyright © 2019 Intel Corporation
4
* Permission is hereby granted, free of charge, to any person obtaining a
5
* copy of this software and associated documentation files (the "Software"),
6
* to deal in the Software without restriction, including without limitation
7
* the rights to use, copy, modify, merge, publish, distribute, sublicense,
8
* and/or sell copies of the Software, and to permit persons to whom the
9
* Software is furnished to do so, subject to the following conditions:
11
* The above copyright notice and this permission notice (including the next
12
* paragraph) shall be included in all copies or substantial portions of the
15
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
18
* THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
20
* FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
25
#include "nir_builder.h"
26
#include "nir_deref.h"
28
/** @file nir_lower_io_to_vector.c
30
* Merges compatible input/output variables residing in different components
31
* of the same location. It's expected that further passes such as
32
* nir_lower_io_to_temporaries will combine loads and stores of the merged
33
* variables, producing vector nir_load_input/nir_store_output instructions
34
* when all is said and done.
37
/* FRAG_RESULT_MAX+1 instead of just FRAG_RESULT_MAX because of how this pass
38
* handles dual source blending */
39
#define MAX_SLOTS MAX2(VARYING_SLOT_TESS_MAX, FRAG_RESULT_MAX+1)
42
get_slot(const nir_variable *var)
44
/* This handling of dual-source blending might not be correct when more than
45
* one render target is supported, but it seems no driver supports more than
47
return var->data.location + var->data.index;
50
static const struct glsl_type *
51
get_per_vertex_type(const nir_shader *shader, const nir_variable *var,
52
unsigned *num_vertices)
54
if (nir_is_arrayed_io(var, shader->info.stage)) {
55
assert(glsl_type_is_array(var->type));
57
*num_vertices = glsl_get_length(var->type);
58
return glsl_get_array_element(var->type);
66
static const struct glsl_type *
67
resize_array_vec_type(const struct glsl_type *type, unsigned num_components)
69
if (glsl_type_is_array(type)) {
70
const struct glsl_type *arr_elem =
71
resize_array_vec_type(glsl_get_array_element(type), num_components);
72
return glsl_array_type(arr_elem, glsl_get_length(type), 0);
74
assert(glsl_type_is_vector_or_scalar(type));
75
return glsl_vector_type(glsl_get_base_type(type), num_components);
80
variables_can_merge(const nir_shader *shader,
81
const nir_variable *a, const nir_variable *b,
82
bool same_array_structure)
84
if (a->data.compact || b->data.compact)
87
if (a->data.per_view || b->data.per_view)
90
const struct glsl_type *a_type_tail = a->type;
91
const struct glsl_type *b_type_tail = b->type;
93
if (nir_is_arrayed_io(a, shader->info.stage) !=
94
nir_is_arrayed_io(b, shader->info.stage))
97
/* They must have the same array structure */
98
if (same_array_structure) {
99
while (glsl_type_is_array(a_type_tail)) {
100
if (!glsl_type_is_array(b_type_tail))
103
if (glsl_get_length(a_type_tail) != glsl_get_length(b_type_tail))
106
a_type_tail = glsl_get_array_element(a_type_tail);
107
b_type_tail = glsl_get_array_element(b_type_tail);
109
if (glsl_type_is_array(b_type_tail))
112
a_type_tail = glsl_without_array(a_type_tail);
113
b_type_tail = glsl_without_array(b_type_tail);
116
if (!glsl_type_is_vector_or_scalar(a_type_tail) ||
117
!glsl_type_is_vector_or_scalar(b_type_tail))
120
if (glsl_get_base_type(a_type_tail) != glsl_get_base_type(b_type_tail))
123
/* TODO: add 64/16bit support ? */
124
if (glsl_get_bit_size(a_type_tail) != 32)
127
assert(a->data.mode == b->data.mode);
128
if (shader->info.stage == MESA_SHADER_FRAGMENT &&
129
a->data.mode == nir_var_shader_in &&
130
(a->data.interpolation != b->data.interpolation ||
131
a->data.centroid != b->data.centroid ||
132
a->data.sample != b->data.sample))
135
if (shader->info.stage == MESA_SHADER_FRAGMENT &&
136
a->data.mode == nir_var_shader_out &&
137
a->data.index != b->data.index)
140
/* It's tricky to merge XFB-outputs correctly, because we need there
141
* to not be any overlaps when we get to
142
* nir_gather_xfb_info_with_varyings later on. We'll end up
143
* triggering an assert there if we merge here.
145
if ((shader->info.stage == MESA_SHADER_VERTEX ||
146
shader->info.stage == MESA_SHADER_TESS_EVAL ||
147
shader->info.stage == MESA_SHADER_GEOMETRY) &&
148
a->data.mode == nir_var_shader_out &&
149
(a->data.explicit_xfb_buffer || b->data.explicit_xfb_buffer))
155
static const struct glsl_type *
156
get_flat_type(const nir_shader *shader, nir_variable *old_vars[MAX_SLOTS][4],
157
unsigned *loc, nir_variable **first_var, unsigned *num_vertices)
161
unsigned num_vars = 0;
162
enum glsl_base_type base;
167
assert(*loc < MAX_SLOTS);
168
for (unsigned frac = 0; frac < 4; frac++) {
169
nir_variable *var = old_vars[*loc][frac];
173
!variables_can_merge(shader, var, *first_var, false)) ||
180
if (!glsl_type_is_vector_or_scalar(glsl_without_array(var->type))) {
185
base = glsl_get_base_type(
186
glsl_without_array(get_per_vertex_type(shader, var, NULL)));
189
bool vs_in = shader->info.stage == MESA_SHADER_VERTEX &&
190
var->data.mode == nir_var_shader_in;
191
unsigned var_slots = glsl_count_attribute_slots(
192
get_per_vertex_type(shader, var, num_vertices), vs_in);
193
todo = MAX2(todo, var_slots);
205
return glsl_vector_type(base, 4);
207
return glsl_array_type(glsl_vector_type(base, 4), slots, 0);
211
create_new_io_vars(nir_shader *shader, nir_variable_mode mode,
212
nir_variable *new_vars[MAX_SLOTS][4],
213
bool flat_vars[MAX_SLOTS])
215
nir_variable *old_vars[MAX_SLOTS][4] = {{0}};
217
bool has_io_var = false;
218
nir_foreach_variable_with_modes(var, shader, mode) {
219
unsigned frac = var->data.location_frac;
220
old_vars[get_slot(var)][frac] = var;
227
bool merged_any_vars = false;
229
for (unsigned loc = 0; loc < MAX_SLOTS; loc++) {
232
nir_variable *first_var = old_vars[loc][frac];
239
bool found_merge = false;
242
nir_variable *var = old_vars[loc][frac];
246
if (var != first_var) {
247
if (!variables_can_merge(shader, first_var, var, true))
253
const unsigned num_components =
254
glsl_get_components(glsl_without_array(var->type));
255
if (!num_components) {
258
break; /* The type was a struct. */
261
/* We had better not have any overlapping vars */
262
for (unsigned i = 1; i < num_components; i++)
263
assert(old_vars[loc][frac + i] == NULL);
265
frac += num_components;
271
merged_any_vars = true;
273
nir_variable *var = nir_variable_clone(old_vars[loc][first], shader);
274
var->data.location_frac = first;
275
var->type = resize_array_vec_type(var->type, frac - first);
277
nir_shader_add_variable(shader, var);
278
for (unsigned i = first; i < frac; i++) {
279
new_vars[loc][i] = var;
280
old_vars[loc][i] = NULL;
283
old_vars[loc][first] = var;
287
/* "flat" mode: tries to ensure there is at most one variable per slot by
288
* merging variables into vec4s
290
for (unsigned loc = 0; loc < MAX_SLOTS;) {
291
nir_variable *first_var;
292
unsigned num_vertices;
293
unsigned new_loc = loc;
294
const struct glsl_type *flat_type =
295
get_flat_type(shader, old_vars, &new_loc, &first_var, &num_vertices);
297
merged_any_vars = true;
299
nir_variable *var = nir_variable_clone(first_var, shader);
300
var->data.location_frac = 0;
302
var->type = glsl_array_type(flat_type, num_vertices, 0);
304
var->type = flat_type;
306
nir_shader_add_variable(shader, var);
307
unsigned num_slots = MAX2(glsl_get_length(flat_type), 1);
308
for (unsigned i = 0; i < num_slots; i++) {
309
for (unsigned j = 0; j < 4; j++)
310
new_vars[loc + i][j] = var;
311
flat_vars[loc + i] = true;
317
return merged_any_vars;
320
static nir_deref_instr *
321
build_array_deref_of_new_var(nir_builder *b, nir_variable *new_var,
322
nir_deref_instr *leader)
324
if (leader->deref_type == nir_deref_type_var)
325
return nir_build_deref_var(b, new_var);
327
nir_deref_instr *parent =
328
build_array_deref_of_new_var(b, new_var, nir_deref_instr_parent(leader));
330
return nir_build_deref_follower(b, parent, leader);
334
build_array_index(nir_builder *b, nir_deref_instr *deref, nir_ssa_def *base,
335
bool vs_in, bool per_vertex)
337
switch (deref->deref_type) {
338
case nir_deref_type_var:
340
case nir_deref_type_array: {
341
nir_ssa_def *index = nir_i2i(b, deref->arr.index.ssa,
342
deref->dest.ssa.bit_size);
344
if (nir_deref_instr_parent(deref)->deref_type == nir_deref_type_var &&
349
b, build_array_index(b, nir_deref_instr_parent(deref), base, vs_in, per_vertex),
350
nir_amul_imm(b, index, glsl_count_attribute_slots(deref->type, vs_in)));
353
unreachable("Invalid deref instruction type");
357
static nir_deref_instr *
358
build_array_deref_of_new_var_flat(nir_shader *shader,
359
nir_builder *b, nir_variable *new_var,
360
nir_deref_instr *leader, unsigned base)
362
nir_deref_instr *deref = nir_build_deref_var(b, new_var);
364
bool per_vertex = nir_is_arrayed_io(new_var, shader->info.stage);
367
nir_deref_path_init(&path, leader, NULL);
369
assert(path.path[0]->deref_type == nir_deref_type_var);
370
nir_deref_instr *p = path.path[1];
371
nir_deref_path_finish(&path);
373
nir_ssa_def *index = p->arr.index.ssa;
374
deref = nir_build_deref_array(b, deref, index);
377
if (!glsl_type_is_array(deref->type))
380
bool vs_in = shader->info.stage == MESA_SHADER_VERTEX &&
381
new_var->data.mode == nir_var_shader_in;
382
return nir_build_deref_array(b, deref,
383
build_array_index(b, leader, nir_imm_int(b, base), vs_in, per_vertex));
387
nir_shader_can_read_output(const shader_info *info)
389
switch (info->stage) {
390
case MESA_SHADER_TESS_CTRL:
391
case MESA_SHADER_FRAGMENT:
394
case MESA_SHADER_TASK:
395
case MESA_SHADER_MESH:
396
/* TODO(mesh): This will not be allowed on EXT. */
405
nir_lower_io_to_vector_impl(nir_function_impl *impl, nir_variable_mode modes)
407
assert(!(modes & ~(nir_var_shader_in | nir_var_shader_out)));
410
nir_builder_init(&b, impl);
412
nir_metadata_require(impl, nir_metadata_dominance);
414
nir_shader *shader = impl->function->shader;
415
nir_variable *new_inputs[MAX_SLOTS][4] = {{0}};
416
nir_variable *new_outputs[MAX_SLOTS][4] = {{0}};
417
bool flat_inputs[MAX_SLOTS] = {0};
418
bool flat_outputs[MAX_SLOTS] = {0};
420
if (modes & nir_var_shader_in) {
421
/* Vertex shaders support overlapping inputs. We don't do those */
422
assert(b.shader->info.stage != MESA_SHADER_VERTEX);
424
/* If we don't actually merge any variables, remove that bit from modes
425
* so we don't bother doing extra non-work.
427
if (!create_new_io_vars(shader, nir_var_shader_in,
428
new_inputs, flat_inputs))
429
modes &= ~nir_var_shader_in;
432
if (modes & nir_var_shader_out) {
433
/* If we don't actually merge any variables, remove that bit from modes
434
* so we don't bother doing extra non-work.
436
if (!create_new_io_vars(shader, nir_var_shader_out,
437
new_outputs, flat_outputs))
438
modes &= ~nir_var_shader_out;
444
bool progress = false;
446
/* Actually lower all the IO load/store intrinsics. Load instructions are
447
* lowered to a vector load and an ALU instruction to grab the channels we
448
* want. Outputs are lowered to a write-masked store of the vector output.
449
* For non-TCS outputs, we then run nir_lower_io_to_temporaries at the end
450
* to clean up the partial writes.
452
nir_foreach_block(block, impl) {
453
nir_foreach_instr_safe(instr, block) {
454
if (instr->type != nir_instr_type_intrinsic)
457
nir_intrinsic_instr *intrin = nir_instr_as_intrinsic(instr);
459
switch (intrin->intrinsic) {
460
case nir_intrinsic_load_deref:
461
case nir_intrinsic_interp_deref_at_centroid:
462
case nir_intrinsic_interp_deref_at_sample:
463
case nir_intrinsic_interp_deref_at_offset:
464
case nir_intrinsic_interp_deref_at_vertex: {
465
nir_deref_instr *old_deref = nir_src_as_deref(intrin->src[0]);
466
if (!nir_deref_mode_is_one_of(old_deref, modes))
469
if (nir_deref_mode_is(old_deref, nir_var_shader_out))
470
assert(nir_shader_can_read_output(&b.shader->info));
472
nir_variable *old_var = nir_deref_instr_get_variable(old_deref);
474
const unsigned loc = get_slot(old_var);
475
const unsigned old_frac = old_var->data.location_frac;
476
nir_variable *new_var = old_var->data.mode == nir_var_shader_in ?
477
new_inputs[loc][old_frac] :
478
new_outputs[loc][old_frac];
479
bool flat = old_var->data.mode == nir_var_shader_in ?
480
flat_inputs[loc] : flat_outputs[loc];
484
const unsigned new_frac = new_var->data.location_frac;
486
nir_component_mask_t vec4_comp_mask =
487
((1 << intrin->num_components) - 1) << old_frac;
489
b.cursor = nir_before_instr(&intrin->instr);
491
/* Rewrite the load to use the new variable and only select a
492
* portion of the result.
494
nir_deref_instr *new_deref;
496
new_deref = build_array_deref_of_new_var_flat(
497
shader, &b, new_var, old_deref, loc - get_slot(new_var));
499
assert(get_slot(new_var) == loc);
500
new_deref = build_array_deref_of_new_var(&b, new_var, old_deref);
501
assert(glsl_type_is_vector(new_deref->type));
503
nir_instr_rewrite_src(&intrin->instr, &intrin->src[0],
504
nir_src_for_ssa(&new_deref->dest.ssa));
506
intrin->num_components =
507
glsl_get_components(new_deref->type);
508
intrin->dest.ssa.num_components = intrin->num_components;
510
b.cursor = nir_after_instr(&intrin->instr);
512
nir_ssa_def *new_vec = nir_channels(&b, &intrin->dest.ssa,
513
vec4_comp_mask >> new_frac);
514
nir_ssa_def_rewrite_uses_after(&intrin->dest.ssa,
516
new_vec->parent_instr);
522
case nir_intrinsic_store_deref: {
523
nir_deref_instr *old_deref = nir_src_as_deref(intrin->src[0]);
524
if (!nir_deref_mode_is(old_deref, nir_var_shader_out))
527
nir_variable *old_var = nir_deref_instr_get_variable(old_deref);
529
const unsigned loc = get_slot(old_var);
530
const unsigned old_frac = old_var->data.location_frac;
531
nir_variable *new_var = new_outputs[loc][old_frac];
532
bool flat = flat_outputs[loc];
536
const unsigned new_frac = new_var->data.location_frac;
538
b.cursor = nir_before_instr(&intrin->instr);
540
/* Rewrite the store to be a masked store to the new variable */
541
nir_deref_instr *new_deref;
543
new_deref = build_array_deref_of_new_var_flat(
544
shader, &b, new_var, old_deref, loc - get_slot(new_var));
546
assert(get_slot(new_var) == loc);
547
new_deref = build_array_deref_of_new_var(&b, new_var, old_deref);
548
assert(glsl_type_is_vector(new_deref->type));
550
nir_instr_rewrite_src(&intrin->instr, &intrin->src[0],
551
nir_src_for_ssa(&new_deref->dest.ssa));
553
intrin->num_components =
554
glsl_get_components(new_deref->type);
556
nir_component_mask_t old_wrmask = nir_intrinsic_write_mask(intrin);
558
assert(intrin->src[1].is_ssa);
559
nir_ssa_def *old_value = intrin->src[1].ssa;
560
nir_ssa_scalar comps[4];
561
for (unsigned c = 0; c < intrin->num_components; c++) {
562
if (new_frac + c >= old_frac &&
563
(old_wrmask & 1 << (new_frac + c - old_frac))) {
564
comps[c] = nir_get_ssa_scalar(old_value,
565
new_frac + c - old_frac);
567
comps[c] = nir_get_ssa_scalar(nir_ssa_undef(&b, old_value->num_components,
568
old_value->bit_size), 0);
571
nir_ssa_def *new_value = nir_vec_scalars(&b, comps, intrin->num_components);
572
nir_instr_rewrite_src(&intrin->instr, &intrin->src[1],
573
nir_src_for_ssa(new_value));
575
nir_intrinsic_set_write_mask(intrin,
576
old_wrmask << (old_frac - new_frac));
589
nir_metadata_preserve(impl, nir_metadata_block_index |
590
nir_metadata_dominance);
597
nir_lower_io_to_vector(nir_shader *shader, nir_variable_mode modes)
599
bool progress = false;
601
nir_foreach_function(function, shader) {
603
progress |= nir_lower_io_to_vector_impl(function->impl, modes);
610
nir_vectorize_tess_levels_impl(nir_function_impl *impl)
612
bool progress = false;
614
nir_builder_init(&b, impl);
616
nir_foreach_block(block, impl) {
617
nir_foreach_instr_safe(instr, block) {
618
if (instr->type != nir_instr_type_intrinsic)
621
nir_intrinsic_instr *intrin = nir_instr_as_intrinsic(instr);
622
if (intrin->intrinsic != nir_intrinsic_load_deref &&
623
intrin->intrinsic != nir_intrinsic_store_deref)
626
nir_deref_instr *deref = nir_src_as_deref(intrin->src[0]);
627
if (!nir_deref_mode_is(deref, nir_var_shader_out))
630
nir_variable *var = nir_deref_instr_get_variable(deref);
631
if (var->data.location != VARYING_SLOT_TESS_LEVEL_OUTER &&
632
var->data.location != VARYING_SLOT_TESS_LEVEL_INNER)
635
assert(deref->deref_type == nir_deref_type_array);
636
assert(nir_src_is_const(deref->arr.index));
637
unsigned index = nir_src_as_uint(deref->arr.index);
638
unsigned vec_size = glsl_get_vector_elements(var->type);
640
b.cursor = nir_before_instr(instr);
641
nir_ssa_def *new_deref = &nir_build_deref_var(&b, var)->dest.ssa;
642
nir_instr_rewrite_src(instr, &intrin->src[0], nir_src_for_ssa(new_deref));
644
nir_deref_instr_remove_if_unused(deref);
646
intrin->num_components = vec_size;
648
/* Handle out of bounds access. */
649
if (index >= vec_size) {
650
if (intrin->intrinsic == nir_intrinsic_load_deref) {
651
/* Return undef from out of bounds loads. */
652
b.cursor = nir_after_instr(instr);
653
nir_ssa_def *val = &intrin->dest.ssa;
654
nir_ssa_def *u = nir_ssa_undef(&b, val->num_components, val->bit_size);
655
nir_ssa_def_rewrite_uses(val, u);
658
/* Finally, remove the out of bounds access. */
659
nir_instr_remove(instr);
664
if (intrin->intrinsic == nir_intrinsic_store_deref) {
665
nir_intrinsic_set_write_mask(intrin, 1 << index);
666
nir_ssa_def *new_val = nir_ssa_undef(&b, intrin->num_components, 32);
667
new_val = nir_vector_insert_imm(&b, new_val, intrin->src[1].ssa, index);
668
nir_instr_rewrite_src(instr, &intrin->src[1], nir_src_for_ssa(new_val));
670
b.cursor = nir_after_instr(instr);
671
nir_ssa_def *val = &intrin->dest.ssa;
672
val->num_components = intrin->num_components;
673
nir_ssa_def *comp = nir_channel(&b, val, index);
674
nir_ssa_def_rewrite_uses_after(val, comp, comp->parent_instr);
684
/* Make the tess factor variables vectors instead of compact arrays, so accesses
685
* can be combined by nir_opt_cse()/nir_opt_combine_stores().
688
nir_vectorize_tess_levels(nir_shader *shader)
690
bool progress = false;
692
nir_foreach_shader_out_variable(var, shader) {
693
if (var->data.location == VARYING_SLOT_TESS_LEVEL_OUTER ||
694
var->data.location == VARYING_SLOT_TESS_LEVEL_INNER) {
695
var->type = glsl_vector_type(GLSL_TYPE_FLOAT, glsl_get_length(var->type));
696
var->data.compact = false;
701
nir_foreach_function(function, shader) {
703
progress |= nir_vectorize_tess_levels_impl(function->impl);