Line data Source code
1 : /* Vectorizer
2 : Copyright (C) 2003-2026 Free Software Foundation, Inc.
3 : Contributed by Dorit Naishlos <dorit@il.ibm.com>
4 :
5 : This file is part of GCC.
6 :
7 : GCC is free software; you can redistribute it and/or modify it under
8 : the terms of the GNU General Public License as published by the Free
9 : Software Foundation; either version 3, or (at your option) any later
10 : version.
11 :
12 : GCC is distributed in the hope that it will be useful, but WITHOUT ANY
13 : WARRANTY; without even the implied warranty of MERCHANTABILITY or
14 : FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
15 : for more details.
16 :
17 : You should have received a copy of the GNU General Public License
18 : along with GCC; see the file COPYING3. If not see
19 : <http://www.gnu.org/licenses/>. */
20 :
21 : #ifndef GCC_TREE_VECTORIZER_H
22 : #define GCC_TREE_VECTORIZER_H
23 :
24 : typedef class _stmt_vec_info *stmt_vec_info;
25 : typedef struct _slp_tree *slp_tree;
26 :
27 : #include "tree-data-ref.h"
28 : #include "tree-hash-traits.h"
29 : #include "target.h"
30 : #include "internal-fn.h"
31 : #include "tree-ssa-operands.h"
32 : #include "gimple-match.h"
33 : #include "dominance.h"
34 : #include "ssa.h"
35 :
36 : /* Used for naming of new temporaries. */
37 : enum vect_var_kind {
38 : vect_simple_var,
39 : vect_pointer_var,
40 : vect_scalar_var,
41 : vect_mask_var
42 : };
43 :
44 : /* Defines type of operation. */
45 : enum operation_type {
46 : unary_op = 1,
47 : binary_op,
48 : ternary_op
49 : };
50 :
51 : /* Define type of available alignment support. */
52 : enum dr_alignment_support {
53 : dr_unaligned_unsupported,
54 : dr_unaligned_supported,
55 : dr_explicit_realign,
56 : dr_explicit_realign_optimized,
57 : dr_aligned
58 : };
59 :
60 : /* Define type of peeling support to indicate how peeling for alignment can help
61 : make vectorization supported. */
62 : enum peeling_support {
63 : peeling_known_supported,
64 : peeling_maybe_supported,
65 : peeling_unsupported
66 : };
67 :
68 : /* Define type of def-use cross-iteration cycle. */
69 : enum vect_def_type {
70 : vect_uninitialized_def = 0,
71 : vect_constant_def = 1,
72 : vect_external_def,
73 : vect_internal_def,
74 : vect_induction_def,
75 : vect_reduction_def,
76 : vect_double_reduction_def,
77 : vect_nested_cycle,
78 : vect_first_order_recurrence,
79 : vect_condition_def,
80 : vect_unknown_def_type
81 : };
82 :
83 : /* Define operation type of linear/non-linear induction variable. */
84 : enum vect_induction_op_type {
85 : vect_step_op_add = 0,
86 : vect_step_op_neg,
87 : vect_step_op_mul,
88 : vect_step_op_shl,
89 : vect_step_op_shr
90 : };
91 :
92 : /* Define type of reduction. */
93 : enum vect_reduction_type {
94 : TREE_CODE_REDUCTION,
95 : COND_REDUCTION,
96 : INTEGER_INDUC_COND_REDUCTION,
97 : CONST_COND_REDUCTION,
98 :
99 : /* Retain a scalar phi and use a FOLD_EXTRACT_LAST within the loop
100 : to implement:
101 :
102 : for (int i = 0; i < VF; ++i)
103 : res = cond[i] ? val[i] : res; */
104 : EXTRACT_LAST_REDUCTION,
105 :
106 : /* Use a folding reduction within the loop to implement:
107 :
108 : for (int i = 0; i < VF; ++i)
109 : res = res OP val[i];
110 :
111 : (with no reassociation). */
112 : FOLD_LEFT_REDUCTION
113 : };
114 :
115 : #define VECTORIZABLE_CYCLE_DEF(D) (((D) == vect_reduction_def) \
116 : || ((D) == vect_double_reduction_def) \
117 : || ((D) == vect_nested_cycle))
118 :
119 : /* Structure to encapsulate information about a group of like
120 : instructions to be presented to the target cost model. */
121 : struct stmt_info_for_cost {
122 : int count;
123 : enum vect_cost_for_stmt kind;
124 : enum vect_cost_model_location where;
125 : stmt_vec_info stmt_info;
126 : slp_tree node;
127 : tree vectype;
128 : int misalign;
129 : };
130 :
131 : typedef vec<stmt_info_for_cost> stmt_vector_for_cost;
132 :
133 : /* Maps base addresses to an innermost_loop_behavior and the stmt it was
134 : derived from that gives the maximum known alignment for that base. */
135 : typedef hash_map<tree_operand_hash,
136 : std::pair<stmt_vec_info, innermost_loop_behavior *> >
137 : vec_base_alignments;
138 :
139 : /* Represents elements [START, START + LENGTH) of cyclical array OPS*
140 : (i.e. OPS repeated to give at least START + LENGTH elements) */
141 : struct vect_scalar_ops_slice
142 : {
143 : tree op (unsigned int i) const;
144 : bool all_same_p () const;
145 :
146 : vec<tree> *ops;
147 : unsigned int start;
148 : unsigned int length;
149 : };
150 :
151 : /* Return element I of the slice. */
152 : inline tree
153 2733364 : vect_scalar_ops_slice::op (unsigned int i) const
154 : {
155 5466728 : return (*ops)[(i + start) % ops->length ()];
156 : }
157 :
158 : /* Hash traits for vect_scalar_ops_slice. */
159 : struct vect_scalar_ops_slice_hash : typed_noop_remove<vect_scalar_ops_slice>
160 : {
161 : typedef vect_scalar_ops_slice value_type;
162 : typedef vect_scalar_ops_slice compare_type;
163 :
164 : static const bool empty_zero_p = true;
165 :
166 : static void mark_deleted (value_type &s) { s.length = ~0U; }
167 0 : static void mark_empty (value_type &s) { s.length = 0; }
168 447618 : static bool is_deleted (const value_type &s) { return s.length == ~0U; }
169 4222626 : static bool is_empty (const value_type &s) { return s.length == 0; }
170 : static hashval_t hash (const value_type &);
171 : static bool equal (const value_type &, const compare_type &);
172 : };
173 :
174 : /* Describes how we're going to vectorize an individual load or store,
175 : or a group of loads or stores. */
176 : enum vect_memory_access_type {
177 : VMAT_UNINITIALIZED,
178 :
179 : /* An access to an invariant address. This is used only for loads. */
180 : VMAT_INVARIANT,
181 :
182 : /* A simple contiguous access. */
183 : VMAT_CONTIGUOUS,
184 :
185 : /* A contiguous access that goes down in memory rather than up,
186 : with no additional permutation. This is used only for stores
187 : of invariants. */
188 : VMAT_CONTIGUOUS_DOWN,
189 :
190 : /* A simple contiguous access in which the elements need to be reversed
191 : after loading or before storing. */
192 : VMAT_CONTIGUOUS_REVERSE,
193 :
194 : /* An access that uses IFN_LOAD_LANES or IFN_STORE_LANES. */
195 : VMAT_LOAD_STORE_LANES,
196 :
197 : /* An access in which each scalar element is loaded or stored
198 : individually. */
199 : VMAT_ELEMENTWISE,
200 :
201 : /* A hybrid of VMAT_CONTIGUOUS and VMAT_ELEMENTWISE, used for grouped
202 : SLP accesses. Each unrolled iteration uses a contiguous load
203 : or store for the whole group, but the groups from separate iterations
204 : are combined in the same way as for VMAT_ELEMENTWISE. */
205 : VMAT_STRIDED_SLP,
206 :
207 : /* The access uses gather loads or scatter stores. */
208 : VMAT_GATHER_SCATTER_LEGACY,
209 : VMAT_GATHER_SCATTER_IFN,
210 : VMAT_GATHER_SCATTER_EMULATED
211 : };
212 :
213 : /* Returns whether MAT is any of the VMAT_GATHER_SCATTER_* kinds. */
214 :
215 : inline bool
216 6649235 : mat_gather_scatter_p (vect_memory_access_type mat)
217 : {
218 6649235 : return (mat == VMAT_GATHER_SCATTER_LEGACY
219 : || mat == VMAT_GATHER_SCATTER_IFN
220 6649235 : || mat == VMAT_GATHER_SCATTER_EMULATED);
221 : }
222 :
223 : /*-----------------------------------------------------------------*/
224 : /* Info on vectorized defs. */
225 : /*-----------------------------------------------------------------*/
226 : enum stmt_vec_info_type {
227 : undef_vec_info_type = 0,
228 : load_vec_info_type,
229 : store_vec_info_type,
230 : shift_vec_info_type,
231 : op_vec_info_type,
232 : call_vec_info_type,
233 : call_simd_clone_vec_info_type,
234 : assignment_vec_info_type,
235 : condition_vec_info_type,
236 : comparison_vec_info_type,
237 : reduc_vec_info_type,
238 : induc_vec_info_type,
239 : type_promotion_vec_info_type,
240 : type_demotion_vec_info_type,
241 : type_conversion_vec_info_type,
242 : cycle_phi_info_type,
243 : lc_phi_info_type,
244 : phi_info_type,
245 : recurr_info_type,
246 : loop_exit_ctrl_vec_info_type,
247 : permute_info_type
248 : };
249 :
250 : /************************************************************************
251 : SLP
252 : ************************************************************************/
253 : typedef vec<std::pair<unsigned, unsigned> > lane_permutation_t;
254 : typedef auto_vec<std::pair<unsigned, unsigned>, 16> auto_lane_permutation_t;
255 : typedef vec<unsigned> load_permutation_t;
256 : typedef auto_vec<unsigned, 16> auto_load_permutation_t;
257 :
258 3417812 : struct vect_data {
259 2128358 : virtual ~vect_data () = default;
260 : };
261 :
262 : /* Analysis data from vectorizable_simd_clone_call for
263 : call_simd_clone_vec_info_type. */
264 : struct vect_simd_clone_data : vect_data {
265 1998 : virtual ~vect_simd_clone_data () = default;
266 1521 : vect_simd_clone_data () = default;
267 477 : vect_simd_clone_data (vect_simd_clone_data &&other) = default;
268 :
269 : /* Selected SIMD clone and clone for in-branch. */
270 : cgraph_node *clone;
271 : cgraph_node *clone_inbranch;
272 :
273 : /* Selected SIMD clone's function info. First vector element
274 : is NULL_TREE, followed by a pair of trees (base + step)
275 : for linear arguments (pair of NULLs for other arguments). */
276 : auto_vec<tree> simd_clone_info;
277 : };
278 :
279 : /* Analysis data from vectorizable_load and vectorizable_store for
280 : load_vec_info_type and store_vec_info_type. */
281 : struct vect_load_store_data : vect_data {
282 1287456 : vect_load_store_data (vect_load_store_data &&other) = default;
283 2128358 : vect_load_store_data () = default;
284 3415036 : virtual ~vect_load_store_data () = default;
285 :
286 : vect_memory_access_type memory_access_type;
287 : dr_alignment_support alignment_support_scheme;
288 : int misalignment;
289 : internal_fn lanes_ifn; // VMAT_LOAD_STORE_LANES
290 : poly_int64 poffset;
291 : union {
292 : internal_fn ifn; // VMAT_GATHER_SCATTER_IFN
293 : tree decl; // VMAT_GATHER_SCATTER_DECL
294 : } gs;
295 : tree strided_offset_vectype; // VMAT_GATHER_SCATTER_IFN, originally strided
296 : /* Load/store type with larger element mode used for punning the vectype. */
297 : tree ls_type; // VMAT_GATHER_SCATTER_IFN, VMAT_STRIDED_SLP
298 : /* Load/store element type used for punning the vectype. Relevant when
299 : that is a vector type. */
300 : tree ls_eltype; // VMAT_STRIDED_SLP
301 : /* This is set to a supported offset vector type if we don't support the
302 : originally requested offset type, otherwise NULL.
303 : If nonzero there will be an additional offset conversion before
304 : the gather/scatter. */
305 : tree supported_offset_vectype; // VMAT_GATHER_SCATTER_IFN
306 : /* Similar for scale. Only nonzero if we don't support the requested
307 : scale. Then we need to multiply the offset vector before the
308 : gather/scatter. */
309 : int supported_scale; // VMAT_GATHER_SCATTER_IFN
310 : auto_vec<int> elsvals;
311 : /* True if the load requires a load permutation. */
312 : bool slp_perm; // SLP_TREE_LOAD_PERMUTATION
313 : unsigned n_perms; // SLP_TREE_LOAD_PERMUTATION
314 : unsigned n_loads; // SLP_TREE_LOAD_PERMUTATION
315 : /* Whether the load permutation is consecutive and simple. */
316 : bool subchain_p; // VMAT_STRIDED_SLP and VMAT_GATHER_SCATTER
317 : };
318 :
319 : /* A computation tree of an SLP instance. Each node corresponds to a group of
320 : stmts to be packed in a SIMD stmt. */
321 : struct _slp_tree {
322 : _slp_tree ();
323 : ~_slp_tree ();
324 :
325 : void push_vec_def (gimple *def);
326 11700 : void push_vec_def (tree def) { vec_defs.quick_push (def); }
327 :
328 : /* Nodes that contain def-stmts of this node statements operands. */
329 : vec<slp_tree> children;
330 :
331 : /* A group of scalar stmts to be vectorized together. Unused when
332 : def_type is vect_external_def or vect_costant_def. */
333 : vec<stmt_vec_info> stmts;
334 : /* A group of scalar operands to be vectorized together. Unused
335 : unless def_type is vect_external_def or vect_constant_def. */
336 : vec<tree> ops;
337 : /* A set of lane indices that are live and to be code-generated from
338 : this SLP node. */
339 : vec<unsigned> live_lanes;
340 :
341 : /* The representative that should be used for analysis and
342 : code generation. NULL when code is VEC_PERM_EXPR. */
343 : stmt_vec_info representative;
344 :
345 : struct {
346 : /* SLP cycle the node resides in, or -1. */
347 : int id;
348 : /* The SLP operand index with the edge on the SLP cycle, or -1. */
349 : int reduc_idx;
350 : } cycle_info;
351 :
352 : /* Load permutation mapping outgoing vector lanes to lanes of
353 : the single DR group the load accesses. NULL if there is no
354 : permutation. Unused when this is not a load. */
355 : load_permutation_t load_permutation;
356 : /* Lane permutation of the operands scalar lanes encoded as pairs
357 : of { operand number, lane number }. The number of elements
358 : denotes the number of output lanes. Unused unless code is
359 : VEC_PERM_EXPR. */
360 : lane_permutation_t lane_permutation;
361 :
362 : tree vectype;
363 : /* Vectorized defs. */
364 : vec<tree> vec_defs;
365 : /* Insertion place for verification purposes. Only set for
366 : BB vectorization. NULL denotes region entry. */
367 : gimple *si;
368 :
369 : /* Reference count in the SLP graph. */
370 : unsigned int refcnt;
371 : /* The DEF type of this node. */
372 : enum vect_def_type def_type;
373 : /* The number of scalar lanes produced by this node. */
374 : unsigned int lanes;
375 : /* The operation of this node. Either VEC_PERM_EXPR or ERROR_MARK. */
376 : enum tree_code code;
377 : /* For gather/scatter memory operations the scale each offset element
378 : should be multiplied by before being added to the base. */
379 : int gs_scale;
380 : /* For gather/scatter memory operations the loop-invariant base value. */
381 : tree gs_base;
382 : /* Whether uses of this load or feeders of this store are suitable
383 : for load/store-lanes. */
384 : bool ldst_lanes;
385 : /* For BB vect, flag to indicate this load node should be vectorized
386 : as to avoid STLF fails because of related stores. */
387 : bool avoid_stlf_fail;
388 :
389 : /* The vertex index of this node when a full graph is built. */
390 : int vertex;
391 :
392 : /* The kind of operation as determined by analysis and optional
393 : kind specific data. */
394 : enum stmt_vec_info_type type;
395 : vect_data *data;
396 :
397 : template <class T>
398 2129879 : T& get_data (T& else_) { return data ? *static_cast <T *> (data) : else_; }
399 :
400 : /* If not NULL this is a cached failed SLP discovery attempt with
401 : the lanes that failed during SLP discovery as 'false'. This is
402 : a copy of the matches array. */
403 : bool *failed;
404 :
405 : /* Allocate from slp_tree_pool. */
406 : static void *operator new (size_t);
407 :
408 : /* Return memory to slp_tree_pool. */
409 : static void operator delete (void *, size_t);
410 :
411 : /* Linked list of nodes to release when we free the slp_tree_pool. */
412 : slp_tree next_node;
413 : slp_tree prev_node;
414 : };
415 :
416 : /* The enum describes the type of operations that an SLP instance
417 : can perform. */
418 :
419 : enum slp_instance_kind {
420 : slp_inst_kind_store,
421 : slp_inst_kind_reduc_group,
422 : slp_inst_kind_reduc_chain,
423 : slp_inst_kind_bb_reduc,
424 : slp_inst_kind_ctor,
425 : slp_inst_kind_gcond
426 : };
427 :
428 : /* SLP instance is a sequence of stmts in a loop that can be packed into
429 : SIMD stmts. */
430 : typedef class _slp_instance {
431 : public:
432 : /* The root of SLP tree. */
433 : slp_tree root;
434 :
435 : /* For vector constructors, the constructor stmt that the SLP tree is built
436 : from, NULL otherwise. */
437 : vec<stmt_vec_info> root_stmts;
438 :
439 : /* For slp_inst_kind_bb_reduc the defs that were not vectorized, NULL
440 : otherwise. */
441 : vec<tree> remain_defs;
442 :
443 : /* The group of nodes that contain loads of this SLP instance. */
444 : vec<slp_tree> loads;
445 :
446 : /* The SLP node containing the reduction PHIs. */
447 : slp_tree reduc_phis;
448 :
449 : /* Vector cost of this entry to the SLP graph. */
450 : stmt_vector_for_cost cost_vec;
451 :
452 : /* If this instance is the main entry of a subgraph the set of
453 : entries into the same subgraph, including itself. */
454 : vec<_slp_instance *> subgraph_entries;
455 :
456 : /* The type of operation the SLP instance is performing. */
457 : slp_instance_kind kind;
458 :
459 : dump_user_location_t location () const;
460 : } *slp_instance;
461 :
462 :
463 : /* Access Functions. */
464 : #define SLP_INSTANCE_TREE(S) (S)->root
465 : #define SLP_INSTANCE_LOADS(S) (S)->loads
466 : #define SLP_INSTANCE_ROOT_STMTS(S) (S)->root_stmts
467 : #define SLP_INSTANCE_REMAIN_DEFS(S) (S)->remain_defs
468 : #define SLP_INSTANCE_KIND(S) (S)->kind
469 :
470 : #define SLP_TREE_CHILDREN(S) (S)->children
471 : #define SLP_TREE_SCALAR_STMTS(S) (S)->stmts
472 : #define SLP_TREE_SCALAR_OPS(S) (S)->ops
473 : #define SLP_TREE_LIVE_LANES(S) (S)->live_lanes
474 : #define SLP_TREE_REF_COUNT(S) (S)->refcnt
475 : #define SLP_TREE_VEC_DEFS(S) (S)->vec_defs
476 : #define SLP_TREE_LOAD_PERMUTATION(S) (S)->load_permutation
477 : #define SLP_TREE_LANE_PERMUTATION(S) (S)->lane_permutation
478 : #define SLP_TREE_DEF_TYPE(S) (S)->def_type
479 : #define SLP_TREE_VECTYPE(S) (S)->vectype
480 : #define SLP_TREE_REPRESENTATIVE(S) (S)->representative
481 : #define SLP_TREE_LANES(S) (S)->lanes
482 : #define SLP_TREE_CODE(S) (S)->code
483 : #define SLP_TREE_TYPE(S) (S)->type
484 : #define SLP_TREE_GS_SCALE(S) (S)->gs_scale
485 : #define SLP_TREE_GS_BASE(S) (S)->gs_base
486 : #define SLP_TREE_REDUC_IDX(S) (S)->cycle_info.reduc_idx
487 : #define SLP_TREE_PERMUTE_P(S) ((S)->code == VEC_PERM_EXPR)
488 :
489 : inline vect_memory_access_type
490 978287 : SLP_TREE_MEMORY_ACCESS_TYPE (slp_tree node)
491 : {
492 354327 : if (SLP_TREE_TYPE (node) == load_vec_info_type
493 336682 : || SLP_TREE_TYPE (node) == store_vec_info_type)
494 116453 : return static_cast<vect_load_store_data *> (node->data)->memory_access_type;
495 : return VMAT_UNINITIALIZED;
496 : }
497 :
498 : enum vect_partial_vector_style {
499 : vect_partial_vectors_none,
500 : vect_partial_vectors_while_ult,
501 : vect_partial_vectors_avx512,
502 : vect_partial_vectors_len
503 : };
504 :
505 : /* Key for map that records association between
506 : scalar conditions and corresponding loop mask, and
507 : is populated by vect_record_loop_mask. */
508 :
509 : struct scalar_cond_masked_key
510 : {
511 63967 : scalar_cond_masked_key (tree t, unsigned ncopies_)
512 63967 : : ncopies (ncopies_)
513 : {
514 63967 : get_cond_ops_from_tree (t);
515 : }
516 :
517 : void get_cond_ops_from_tree (tree);
518 :
519 : unsigned ncopies;
520 : bool inverted_p;
521 : tree_code code;
522 : tree op0;
523 : tree op1;
524 : };
525 :
526 : template<>
527 : struct default_hash_traits<scalar_cond_masked_key>
528 : {
529 : typedef scalar_cond_masked_key compare_type;
530 : typedef scalar_cond_masked_key value_type;
531 :
532 : static inline hashval_t
533 72779 : hash (value_type v)
534 : {
535 72779 : inchash::hash h;
536 72779 : h.add_int (v.code);
537 72779 : inchash::add_expr (v.op0, h, 0);
538 72779 : inchash::add_expr (v.op1, h, 0);
539 72779 : h.add_int (v.ncopies);
540 72779 : h.add_flag (v.inverted_p);
541 72779 : return h.end ();
542 : }
543 :
544 : static inline bool
545 10760 : equal (value_type existing, value_type candidate)
546 : {
547 10760 : return (existing.ncopies == candidate.ncopies
548 10544 : && existing.code == candidate.code
549 6613 : && existing.inverted_p == candidate.inverted_p
550 5077 : && operand_equal_p (existing.op0, candidate.op0, 0)
551 13912 : && operand_equal_p (existing.op1, candidate.op1, 0));
552 : }
553 :
554 : static const bool empty_zero_p = true;
555 :
556 : static inline void
557 0 : mark_empty (value_type &v)
558 : {
559 0 : v.ncopies = 0;
560 0 : v.inverted_p = false;
561 : }
562 :
563 : static inline bool
564 9353536 : is_empty (value_type v)
565 : {
566 9290739 : return v.ncopies == 0;
567 : }
568 :
569 : static inline void mark_deleted (value_type &) {}
570 :
571 : static inline bool is_deleted (const value_type &)
572 : {
573 : return false;
574 : }
575 :
576 55639 : static inline void remove (value_type &) {}
577 : };
578 :
579 : typedef hash_set<scalar_cond_masked_key> scalar_cond_masked_set_type;
580 :
581 : /* Key and map that records association between vector conditions and
582 : corresponding loop mask, and is populated by prepare_vec_mask. */
583 :
584 : typedef pair_hash<tree_operand_hash, tree_operand_hash> tree_cond_mask_hash;
585 : typedef hash_set<tree_cond_mask_hash> vec_cond_masked_set_type;
586 :
587 : /* Describes two objects whose addresses must be unequal for the vectorized
588 : loop to be valid. */
589 : typedef std::pair<tree, tree> vec_object_pair;
590 :
591 : /* Records that vectorization is only possible if abs (EXPR) >= MIN_VALUE.
592 : UNSIGNED_P is true if we can assume that abs (EXPR) == EXPR. */
593 : class vec_lower_bound {
594 : public:
595 : vec_lower_bound () {}
596 1751 : vec_lower_bound (tree e, bool u, poly_uint64 m)
597 1751 : : expr (e), unsigned_p (u), min_value (m) {}
598 :
599 : tree expr;
600 : bool unsigned_p;
601 : poly_uint64 min_value;
602 : };
603 :
604 : /* Vectorizer state shared between different analyses like vector sizes
605 : of the same CFG region. */
606 : class vec_info_shared {
607 : public:
608 : vec_info_shared();
609 : ~vec_info_shared();
610 :
611 : void save_datarefs();
612 : void check_datarefs();
613 :
614 : /* All data references. Freed by free_data_refs, so not an auto_vec. */
615 : vec<data_reference_p> datarefs;
616 : vec<data_reference> datarefs_copy;
617 :
618 : /* The loop nest in which the data dependences are computed. */
619 : auto_vec<loop_p> loop_nest;
620 :
621 : /* All data dependences. Freed by free_dependence_relations, so not
622 : an auto_vec. */
623 : vec<ddr_p> ddrs;
624 : };
625 :
626 : /* Vectorizer state common between loop and basic-block vectorization. */
627 : class vec_info {
628 : public:
629 : typedef hash_set<int_hash<machine_mode, E_VOIDmode, E_BLKmode> > mode_set;
630 : enum vec_kind { bb, loop };
631 :
632 : vec_info (vec_kind, vec_info_shared *);
633 : ~vec_info ();
634 :
635 : stmt_vec_info add_stmt (gimple *);
636 : stmt_vec_info add_pattern_stmt (gimple *, stmt_vec_info);
637 : stmt_vec_info resync_stmt_addr (gimple *);
638 : stmt_vec_info lookup_stmt (gimple *);
639 : stmt_vec_info lookup_def (tree);
640 : stmt_vec_info lookup_single_use (tree);
641 : class dr_vec_info *lookup_dr (data_reference *);
642 : void move_dr (stmt_vec_info, stmt_vec_info);
643 : void remove_stmt (stmt_vec_info);
644 : void replace_stmt (gimple_stmt_iterator *, stmt_vec_info, gimple *);
645 : void insert_on_entry (stmt_vec_info, gimple *);
646 : void insert_seq_on_entry (stmt_vec_info, gimple_seq);
647 :
648 : /* The type of vectorization. */
649 : vec_kind kind;
650 :
651 : /* Shared vectorizer state. */
652 : vec_info_shared *shared;
653 :
654 : /* The mapping of GIMPLE UID to stmt_vec_info. */
655 : vec<stmt_vec_info> stmt_vec_infos;
656 : /* Whether the above mapping is complete. */
657 : bool stmt_vec_info_ro;
658 :
659 : /* Whether we've done a transform we think OK to not update virtual
660 : SSA form. */
661 : bool any_known_not_updated_vssa;
662 :
663 : /* The SLP graph. */
664 : auto_vec<slp_instance> slp_instances;
665 :
666 : /* Maps base addresses to an innermost_loop_behavior that gives the maximum
667 : known alignment for that base. */
668 : vec_base_alignments base_alignments;
669 :
670 : /* All interleaving chains of stores, represented by the first
671 : stmt in the chain. */
672 : auto_vec<stmt_vec_info> grouped_stores;
673 :
674 : /* The set of vector modes used in the vectorized region. */
675 : mode_set used_vector_modes;
676 :
677 : /* The argument we should pass to related_vector_mode when looking up
678 : the vector mode for a scalar mode, or VOIDmode if we haven't yet
679 : made any decisions about which vector modes to use. */
680 : machine_mode vector_mode;
681 :
682 : /* The basic blocks in the vectorization region. For _loop_vec_info,
683 : the memory is internally managed, while for _bb_vec_info, it points
684 : to element space of an external auto_vec<>. This inconsistency is
685 : not a good class design pattern. TODO: improve it with an unified
686 : auto_vec<> whose lifetime is confined to vec_info object. */
687 : basic_block *bbs;
688 :
689 : /* The count of the basic blocks in the vectorization region. */
690 : unsigned int nbbs;
691 :
692 : /* Used to keep a sequence of def stmts of a pattern stmt that are loop
693 : invariant if they exists.
694 : The sequence is emitted in the loop preheader should the loop be vectorized
695 : and are reset when undoing patterns. */
696 : gimple_seq inv_pattern_def_seq;
697 :
698 : private:
699 : stmt_vec_info new_stmt_vec_info (gimple *stmt);
700 : void set_vinfo_for_stmt (gimple *, stmt_vec_info, bool = true);
701 : void free_stmt_vec_infos ();
702 : void free_stmt_vec_info (stmt_vec_info);
703 : };
704 :
705 : class _loop_vec_info;
706 : class _bb_vec_info;
707 :
708 : template<>
709 : template<>
710 : inline bool
711 388432065 : is_a_helper <_loop_vec_info *>::test (vec_info *i)
712 : {
713 387763568 : return i->kind == vec_info::loop;
714 : }
715 :
716 : template<>
717 : template<>
718 : inline bool
719 75092772 : is_a_helper <_bb_vec_info *>::test (vec_info *i)
720 : {
721 75092772 : return i->kind == vec_info::bb;
722 : }
723 :
724 : /* In general, we can divide the vector statements in a vectorized loop
725 : into related groups ("rgroups") and say that for each rgroup there is
726 : some nS such that the rgroup operates on nS values from one scalar
727 : iteration followed by nS values from the next. That is, if VF is the
728 : vectorization factor of the loop, the rgroup operates on a sequence:
729 :
730 : (1,1) (1,2) ... (1,nS) (2,1) ... (2,nS) ... (VF,1) ... (VF,nS)
731 :
732 : where (i,j) represents a scalar value with index j in a scalar
733 : iteration with index i.
734 :
735 : [ We use the term "rgroup" to emphasise that this grouping isn't
736 : necessarily the same as the grouping of statements used elsewhere.
737 : For example, if we implement a group of scalar loads using gather
738 : loads, we'll use a separate gather load for each scalar load, and
739 : thus each gather load will belong to its own rgroup. ]
740 :
741 : In general this sequence will occupy nV vectors concatenated
742 : together. If these vectors have nL lanes each, the total number
743 : of scalar values N is given by:
744 :
745 : N = nS * VF = nV * nL
746 :
747 : None of nS, VF, nV and nL are required to be a power of 2. nS and nV
748 : are compile-time constants but VF and nL can be variable (if the target
749 : supports variable-length vectors).
750 :
751 : In classical vectorization, each iteration of the vector loop would
752 : handle exactly VF iterations of the original scalar loop. However,
753 : in vector loops that are able to operate on partial vectors, a
754 : particular iteration of the vector loop might handle fewer than VF
755 : iterations of the scalar loop. The vector lanes that correspond to
756 : iterations of the scalar loop are said to be "active" and the other
757 : lanes are said to be "inactive".
758 :
759 : In such vector loops, many rgroups need to be controlled to ensure
760 : that they have no effect for the inactive lanes. Conceptually, each
761 : such rgroup needs a sequence of booleans in the same order as above,
762 : but with each (i,j) replaced by a boolean that indicates whether
763 : iteration i is active. This sequence occupies nV vector controls
764 : that again have nL lanes each. Thus the control sequence as a whole
765 : consists of VF independent booleans that are each repeated nS times.
766 :
767 : Taking mask-based approach as a partially-populated vectors example.
768 : We make the simplifying assumption that if a sequence of nV masks is
769 : suitable for one (nS,nL) pair, we can reuse it for (nS/2,nL/2) by
770 : VIEW_CONVERTing it. This holds for all current targets that support
771 : fully-masked loops. For example, suppose the scalar loop is:
772 :
773 : float *f;
774 : double *d;
775 : for (int i = 0; i < n; ++i)
776 : {
777 : f[i * 2 + 0] += 1.0f;
778 : f[i * 2 + 1] += 2.0f;
779 : d[i] += 3.0;
780 : }
781 :
782 : and suppose that vectors have 256 bits. The vectorized f accesses
783 : will belong to one rgroup and the vectorized d access to another:
784 :
785 : f rgroup: nS = 2, nV = 1, nL = 8
786 : d rgroup: nS = 1, nV = 1, nL = 4
787 : VF = 4
788 :
789 : [ In this simple example the rgroups do correspond to the normal
790 : SLP grouping scheme. ]
791 :
792 : If only the first three lanes are active, the masks we need are:
793 :
794 : f rgroup: 1 1 | 1 1 | 1 1 | 0 0
795 : d rgroup: 1 | 1 | 1 | 0
796 :
797 : Here we can use a mask calculated for f's rgroup for d's, but not
798 : vice versa.
799 :
800 : Thus for each value of nV, it is enough to provide nV masks, with the
801 : mask being calculated based on the highest nL (or, equivalently, based
802 : on the highest nS) required by any rgroup with that nV. We therefore
803 : represent the entire collection of masks as a two-level table, with the
804 : first level being indexed by nV - 1 (since nV == 0 doesn't exist) and
805 : the second being indexed by the mask index 0 <= i < nV. */
806 :
807 : /* The controls (like masks or lengths) needed by rgroups with nV vectors,
808 : according to the description above. */
809 : struct rgroup_controls {
810 : /* The largest nS for all rgroups that use these controls.
811 : For vect_partial_vectors_avx512 this is the constant nscalars_per_iter
812 : for all members of the group. */
813 : unsigned int max_nscalars_per_iter;
814 :
815 : /* For the largest nS recorded above, the loop controls divide each scalar
816 : into FACTOR equal-sized pieces. This is useful if we need to split
817 : element-based accesses into byte-based accesses.
818 : For vect_partial_vectors_avx512 this records nV instead. */
819 : unsigned int factor;
820 :
821 : /* This is a vector type with MAX_NSCALARS_PER_ITER * VF / nV elements.
822 : For mask-based controls, it is the type of the masks in CONTROLS.
823 : For length-based controls, it can be any vector type that has the
824 : specified number of elements; the type of the elements doesn't matter. */
825 : tree type;
826 :
827 : /* When there is no uniformly used LOOP_VINFO_RGROUP_COMPARE_TYPE this
828 : is the rgroup specific type used. */
829 : tree compare_type;
830 :
831 : /* A vector of nV controls, in iteration order. */
832 : vec<tree> controls;
833 :
834 : /* In case of len_load and len_store with a bias there is only one
835 : rgroup. This holds the adjusted loop length for the this rgroup. */
836 : tree bias_adjusted_ctrl;
837 : };
838 :
839 589396 : struct vec_loop_masks
840 : {
841 527886 : bool is_empty () const { return mask_set.is_empty (); }
842 :
843 : /* Set to record vectype, nvector pairs. */
844 : hash_set<pair_hash <nofree_ptr_hash <tree_node>,
845 : int_hash<unsigned, 0>>> mask_set;
846 :
847 : /* rgroup_controls used for the partial vector scheme. */
848 : auto_vec<rgroup_controls> rgc_vec;
849 : };
850 :
851 : typedef auto_vec<rgroup_controls> vec_loop_lens;
852 :
853 : typedef auto_vec<std::pair<data_reference*, tree> > drs_init_vec;
854 :
855 : /* Abstraction around info on reductions which is still in stmt_vec_info
856 : but will be duplicated or moved elsewhere. */
857 207808 : class vect_reduc_info_s
858 : {
859 : public:
860 : /* The def type of the main reduction PHI, vect_reduction_def or
861 : vect_double_reduction_def. */
862 : enum vect_def_type def_type;
863 :
864 : /* The reduction type as detected by
865 : vect_is_simple_reduction and vectorizable_reduction. */
866 : enum vect_reduction_type reduc_type;
867 :
868 : /* The original scalar reduction code, to be used in the epilogue. */
869 : code_helper reduc_code;
870 :
871 : /* A vector internal function we should use in the epilogue. */
872 : internal_fn reduc_fn;
873 :
874 : /* For loop reduction with multiple vectorized results (ncopies > 1), a
875 : lane-reducing operation participating in it may not use all of those
876 : results, this field specifies result index starting from which any
877 : following land-reducing operation would be assigned to. */
878 : unsigned int reduc_result_pos;
879 :
880 : /* Whether this represents a reduction chain. */
881 : bool is_reduc_chain;
882 :
883 : /* Whether we force a single cycle PHI during reduction vectorization. */
884 : bool force_single_cycle;
885 :
886 : /* The vector type for performing the actual reduction operation. */
887 : tree reduc_vectype;
888 :
889 : /* The vector type we should use for the final reduction in the epilogue
890 : when we reduce a mask. */
891 : tree reduc_vectype_for_mask;
892 :
893 : /* The neutral operand to use, if any. */
894 : tree neutral_op;
895 :
896 : /* For INTEGER_INDUC_COND_REDUCTION, the initial value to be used. */
897 : tree induc_cond_initial_val;
898 :
899 : /* If not NULL the value to be added to compute final reduction value. */
900 : tree reduc_epilogue_adjustment;
901 :
902 : /* If non-null, the reduction is being performed by an epilogue loop
903 : and we have decided to reuse this accumulator from the main loop. */
904 : struct vect_reusable_accumulator *reused_accumulator;
905 :
906 : /* If the vector code is performing N scalar reductions in parallel,
907 : this variable gives the initial scalar values of those N reductions. */
908 : auto_vec<tree> reduc_initial_values;
909 :
910 : /* If the vector code is performing N scalar reductions in parallel, this
911 : variable gives the vectorized code's final (scalar) result for each of
912 : those N reductions. In other words, REDUC_SCALAR_RESULTS[I] replaces
913 : the original scalar code's loop-closed SSA PHI for reduction number I. */
914 : auto_vec<tree> reduc_scalar_results;
915 : };
916 :
917 : typedef class vect_reduc_info_s *vect_reduc_info;
918 :
919 : #define VECT_REDUC_INFO_DEF_TYPE(I) ((I)->def_type)
920 : #define VECT_REDUC_INFO_TYPE(I) ((I)->reduc_type)
921 : #define VECT_REDUC_INFO_CODE(I) ((I)->reduc_code)
922 : #define VECT_REDUC_INFO_FN(I) ((I)->reduc_fn)
923 : #define VECT_REDUC_INFO_SCALAR_RESULTS(I) ((I)->reduc_scalar_results)
924 : #define VECT_REDUC_INFO_INITIAL_VALUES(I) ((I)->reduc_initial_values)
925 : #define VECT_REDUC_INFO_REUSED_ACCUMULATOR(I) ((I)->reused_accumulator)
926 : #define VECT_REDUC_INFO_INDUC_COND_INITIAL_VAL(I) ((I)->induc_cond_initial_val)
927 : #define VECT_REDUC_INFO_EPILOGUE_ADJUSTMENT(I) ((I)->reduc_epilogue_adjustment)
928 : #define VECT_REDUC_INFO_VECTYPE(I) ((I)->reduc_vectype)
929 : #define VECT_REDUC_INFO_VECTYPE_FOR_MASK(I) ((I)->reduc_vectype_for_mask)
930 : #define VECT_REDUC_INFO_FORCE_SINGLE_CYCLE(I) ((I)->force_single_cycle)
931 : #define VECT_REDUC_INFO_RESULT_POS(I) ((I)->reduc_result_pos)
932 : #define VECT_REDUC_INFO_NEUTRAL_OP(I) ((I)->neutral_op)
933 :
934 : /* Information about a reduction accumulator from the main loop that could
935 : conceivably be reused as the input to a reduction in an epilogue loop. */
936 : struct vect_reusable_accumulator {
937 : /* The final value of the accumulator, which forms the input to the
938 : reduction operation. */
939 : tree reduc_input;
940 :
941 : /* The stmt_vec_info that describes the reduction (i.e. the one for
942 : which is_reduc_info is true). */
943 : vect_reduc_info reduc_info;
944 : };
945 :
946 : /*-----------------------------------------------------------------*/
947 : /* Info on vectorized loops. */
948 : /*-----------------------------------------------------------------*/
949 : typedef class _loop_vec_info : public vec_info {
950 : public:
951 : _loop_vec_info (class loop *, vec_info_shared *);
952 : ~_loop_vec_info ();
953 :
954 : /* The loop to which this info struct refers to. */
955 : class loop *loop;
956 :
957 : /* Number of latch executions. */
958 : tree num_itersm1;
959 : /* Number of iterations. */
960 : tree num_iters;
961 : /* Number of iterations of the original loop. */
962 : tree num_iters_unchanged;
963 : /* Condition under which this loop is analyzed and versioned. */
964 : tree num_iters_assumptions;
965 :
966 : /* The cost of the vector code. */
967 : class vector_costs *vector_costs;
968 :
969 : /* The cost of the scalar code. */
970 : class vector_costs *scalar_costs;
971 :
972 : /* Threshold of number of iterations below which vectorization will not be
973 : performed. It is calculated from MIN_PROFITABLE_ITERS and
974 : param_min_vect_loop_bound. */
975 : unsigned int th;
976 :
977 : /* When applying loop versioning, the vector form should only be used
978 : if the number of scalar iterations is >= this value, on top of all
979 : the other requirements. Ignored when loop versioning is not being
980 : used. */
981 : poly_uint64 versioning_threshold;
982 :
983 : /* Unrolling factor. In case of suitable super-word parallelism
984 : it can be that no unrolling is needed, and thus this is 1. */
985 : poly_uint64 vectorization_factor;
986 :
987 : /* Gimple operand for the number of scalar iteration handed per loop
988 : iteration, and therefore how much to increment each IV by. */
989 : tree iv_increment;
990 :
991 : /* If this loop is an epilogue loop whose main loop can be skipped,
992 : MAIN_LOOP_EDGE is the edge from the main loop to this loop's
993 : preheader. SKIP_MAIN_LOOP_EDGE is then the edge that skips the
994 : main loop and goes straight to this loop's preheader.
995 :
996 : Both fields are null otherwise. */
997 : edge main_loop_edge;
998 : edge skip_main_loop_edge;
999 :
1000 : /* If this loop is an epilogue loop that might be skipped after executing
1001 : the main loop, this edge is the one that skips the epilogue. */
1002 : edge skip_this_loop_edge;
1003 :
1004 : /* Reduction descriptors of this loop. Referenced to from SLP nodes
1005 : by index. */
1006 : auto_vec<vect_reduc_info> reduc_infos;
1007 :
1008 : /* The vectorized form of a standard reduction replaces the original
1009 : scalar code's final result (a loop-closed SSA PHI) with the result
1010 : of a vector-to-scalar reduction operation. After vectorization,
1011 : this variable maps these vector-to-scalar results to information
1012 : about the reductions that generated them. */
1013 : hash_map<tree, vect_reusable_accumulator> reusable_accumulators;
1014 :
1015 : /* The number of times that the target suggested we unroll the vector loop
1016 : in order to promote more ILP. This value will be used to re-analyze the
1017 : loop for vectorization and if successful the value will be folded into
1018 : vectorization_factor (and therefore exactly divides
1019 : vectorization_factor). */
1020 : unsigned int suggested_unroll_factor;
1021 :
1022 : /* Maximum runtime vectorization factor, or MAX_VECTORIZATION_FACTOR
1023 : if there is no particular limit. */
1024 : unsigned HOST_WIDE_INT max_vectorization_factor;
1025 :
1026 : /* The masks that a fully-masked loop should use to avoid operating
1027 : on inactive scalars. */
1028 : vec_loop_masks masks;
1029 :
1030 : /* The lengths that a loop with length should use to avoid operating
1031 : on inactive scalars. */
1032 : vec_loop_lens lens;
1033 :
1034 : /* Set of scalar conditions that have loop mask applied. */
1035 : scalar_cond_masked_set_type scalar_cond_masked_set;
1036 :
1037 : /* Set of vector conditions that have loop mask applied. */
1038 : vec_cond_masked_set_type vec_cond_masked_set;
1039 :
1040 : /* If we are using a loop mask to align memory addresses, this variable
1041 : contains the number of vector elements that we should skip in the
1042 : first iteration of the vector loop (i.e. the number of leading
1043 : elements that should be false in the first mask). */
1044 : tree mask_skip_niters;
1045 :
1046 : /* If we are using a loop mask to align memory addresses and we're in an
1047 : early break loop then this variable contains the number of elements that
1048 : were skipped during the initial iteration of the loop. */
1049 : tree mask_skip_niters_pfa_offset;
1050 :
1051 : /* The type that the loop control IV should be converted to before
1052 : testing which of the VF scalars are active and inactive.
1053 : Only meaningful if LOOP_VINFO_USING_PARTIAL_VECTORS_P. */
1054 : tree rgroup_compare_type;
1055 :
1056 : /* For #pragma omp simd if (x) loops the x expression. If constant 0,
1057 : the loop should not be vectorized, if constant non-zero, simd_if_cond
1058 : shouldn't be set and loop vectorized normally, if SSA_NAME, the loop
1059 : should be versioned on that condition, using scalar loop if the condition
1060 : is false and vectorized loop otherwise. */
1061 : tree simd_if_cond;
1062 :
1063 : /* The type that the vector loop control IV should have when
1064 : LOOP_VINFO_USING_PARTIAL_VECTORS_P is true. */
1065 : tree rgroup_iv_type;
1066 :
1067 : /* The style used for implementing partial vectors when
1068 : LOOP_VINFO_USING_PARTIAL_VECTORS_P is true. */
1069 : vect_partial_vector_style partial_vector_style;
1070 :
1071 : /* Unknown DRs according to which loop was peeled. */
1072 : class dr_vec_info *unaligned_dr;
1073 :
1074 : /* peeling_for_alignment indicates whether peeling for alignment will take
1075 : place, and what the peeling factor should be:
1076 : peeling_for_alignment = X means:
1077 : If X=0: Peeling for alignment will not be applied.
1078 : If X>0: Peel first X iterations.
1079 : If X=-1: Generate a runtime test to calculate the number of iterations
1080 : to be peeled, using the dataref recorded in the field
1081 : unaligned_dr. */
1082 : int peeling_for_alignment;
1083 :
1084 : /* The mask used to check the alignment of pointers or arrays. */
1085 : poly_uint64 ptr_mask;
1086 :
1087 : /* The maximum speculative read amount in VLA modes for runtime check. */
1088 : poly_uint64 max_spec_read_amount;
1089 :
1090 : /* Indicates whether the loop has any non-linear IV. */
1091 : bool nonlinear_iv;
1092 :
1093 : /* Data Dependence Relations defining address ranges that are candidates
1094 : for a run-time aliasing check. */
1095 : auto_vec<ddr_p> may_alias_ddrs;
1096 :
1097 : /* Data Dependence Relations defining address ranges together with segment
1098 : lengths from which the run-time aliasing check is built. */
1099 : auto_vec<dr_with_seg_len_pair_t> comp_alias_ddrs;
1100 :
1101 : /* Check that the addresses of each pair of objects is unequal. */
1102 : auto_vec<vec_object_pair> check_unequal_addrs;
1103 :
1104 : /* List of values that are required to be nonzero. This is used to check
1105 : whether things like "x[i * n] += 1;" are safe and eventually gets added
1106 : to the checks for lower bounds below. */
1107 : auto_vec<tree> check_nonzero;
1108 :
1109 : /* List of values that need to be checked for a minimum value. */
1110 : auto_vec<vec_lower_bound> lower_bounds;
1111 :
1112 : /* Statements in the loop that have data references that are candidates for a
1113 : runtime (loop versioning) misalignment check. */
1114 : auto_vec<stmt_vec_info> may_misalign_stmts;
1115 :
1116 : /* Reduction cycles detected in the loop. Used in loop-aware SLP. */
1117 : auto_vec<stmt_vec_info> reductions;
1118 :
1119 : /* Defs that could not be analyzed such as OMP SIMD calls without
1120 : a LHS. */
1121 : auto_vec<stmt_vec_info> alternate_defs;
1122 :
1123 : /* Cost vector for a single scalar iteration. */
1124 : auto_vec<stmt_info_for_cost> scalar_cost_vec;
1125 :
1126 : /* Map of IV base/step expressions to inserted name in the preheader. */
1127 : hash_map<tree_operand_hash, tree> *ivexpr_map;
1128 :
1129 : /* Map of OpenMP "omp simd array" scan variables to corresponding
1130 : rhs of the store of the initializer. */
1131 : hash_map<tree, tree> *scan_map;
1132 :
1133 : /* The factor used to over weight those statements in an inner loop
1134 : relative to the loop being vectorized. */
1135 : unsigned int inner_loop_cost_factor;
1136 :
1137 : /* Is the loop vectorizable? */
1138 : bool vectorizable;
1139 :
1140 : /* Records whether we still have the option of vectorizing this loop
1141 : using partially-populated vectors; in other words, whether it is
1142 : still possible for one iteration of the vector loop to handle
1143 : fewer than VF scalars. */
1144 : bool can_use_partial_vectors_p;
1145 :
1146 : /* Records whether we must use niter masking for correctness reasons. */
1147 : bool must_use_partial_vectors_p;
1148 :
1149 : /* True if we've decided to use partially-populated vectors, so that
1150 : the vector loop can handle fewer than VF scalars. */
1151 : bool using_partial_vectors_p;
1152 :
1153 : /* True if we've decided to use a decrementing loop control IV that counts
1154 : scalars. This can be done for any loop that:
1155 :
1156 : (a) uses length "controls"; and
1157 : (b) can iterate more than once. */
1158 : bool using_decrementing_iv_p;
1159 :
1160 : /* True if we've decided to use output of select_vl to adjust IV of
1161 : both loop control and data reference pointer. This is only true
1162 : for single-rgroup control. */
1163 : bool using_select_vl_p;
1164 :
1165 : /* True if we've decided to use peeling with versioning together, which allows
1166 : unaligned unsupported data refs to be uniformly aligned after a certain
1167 : amount of peeling (mutual alignment). Otherwise, we use versioning alone
1168 : so these data refs must be already aligned to a power-of-two boundary
1169 : without peeling. */
1170 : bool allow_mutual_alignment;
1171 :
1172 : /* The bias for len_load and len_store. For now, only 0 and -1 are
1173 : supported. -1 must be used when a backend does not support
1174 : len_load/len_store with a length of zero. */
1175 : signed char partial_load_store_bias;
1176 :
1177 : /* When we have grouped data accesses with gaps, we may introduce invalid
1178 : memory accesses. We peel the last iteration of the loop to prevent
1179 : this. */
1180 : bool peeling_for_gaps;
1181 :
1182 : /* When the number of iterations is not a multiple of the vector size
1183 : we need to peel off iterations at the end to form an epilogue loop. */
1184 : bool peeling_for_niter;
1185 :
1186 : /* When the loop has early breaks that we can vectorize we need to peel
1187 : the loop for the break finding loop. */
1188 : bool early_breaks;
1189 :
1190 : /* List of loop additional IV conditionals found in the loop. */
1191 : auto_vec<gcond *> conds;
1192 :
1193 : /* Main loop IV cond. */
1194 : gcond* loop_iv_cond;
1195 :
1196 : /* True if we have an unroll factor requested by the user through pragma GCC
1197 : unroll. */
1198 : bool user_unroll;
1199 :
1200 : /* True if there are no loop carried data dependencies in the loop.
1201 : If loop->safelen <= 1, then this is always true, either the loop
1202 : didn't have any loop carried data dependencies, or the loop is being
1203 : vectorized guarded with some runtime alias checks, or couldn't
1204 : be vectorized at all, but then this field shouldn't be used.
1205 : For loop->safelen >= 2, the user has asserted that there are no
1206 : backward dependencies, but there still could be loop carried forward
1207 : dependencies in such loops. This flag will be false if normal
1208 : vectorizer data dependency analysis would fail or require versioning
1209 : for alias, but because of loop->safelen >= 2 it has been vectorized
1210 : even without versioning for alias. E.g. in:
1211 : #pragma omp simd
1212 : for (int i = 0; i < m; i++)
1213 : a[i] = a[i + k] * c;
1214 : (or #pragma simd or #pragma ivdep) we can vectorize this and it will
1215 : DTRT even for k > 0 && k < m, but without safelen we would not
1216 : vectorize this, so this field would be false. */
1217 : bool no_data_dependencies;
1218 :
1219 : /* Mark loops having masked stores. */
1220 : bool has_mask_store;
1221 :
1222 : /* Queued scaling factor for the scalar loop. */
1223 : profile_probability scalar_loop_scaling;
1224 :
1225 : /* If if-conversion versioned this loop before conversion, this is the
1226 : loop version without if-conversion. */
1227 : class loop *scalar_loop;
1228 :
1229 : /* For loops being epilogues of already vectorized loops
1230 : this points to the main vectorized loop. Otherwise NULL. */
1231 : _loop_vec_info *main_loop_info;
1232 :
1233 : /* For loops being epilogues of already vectorized loops
1234 : this points to the preceding vectorized (possibly epilogue) loop.
1235 : Otherwise NULL. */
1236 : _loop_vec_info *orig_loop_info;
1237 :
1238 : /* Used to store loop_vec_infos of the epilogue of this loop during
1239 : analysis. */
1240 : _loop_vec_info *epilogue_vinfo;
1241 :
1242 : /* If this is an epilogue loop the DR advancement applied. */
1243 : tree drs_advanced_by;
1244 :
1245 : /* The controlling loop exit for the current loop when vectorizing.
1246 : For counted loops, this IV controls the natural exits of the loop. */
1247 : edge vec_loop_main_exit;
1248 :
1249 : /* The controlling loop exit for the epilogue loop when vectorizing.
1250 : For counted loops, this IV controls the natural exits of the loop. */
1251 : edge vec_epilogue_loop_main_exit;
1252 :
1253 : /* The controlling loop exit for the scalar loop being vectorized.
1254 : For counted loops, this IV controls the natural exits of the loop. */
1255 : edge scalar_loop_main_exit;
1256 :
1257 : /* Indicate if the multiple exit loop has any side-effects that require it to
1258 : have a scalar epilogue. */
1259 : bool early_break_needs_epilogue;
1260 :
1261 : /* Used to store the list of stores needing to be moved if doing early
1262 : break vectorization as they would violate the scalar loop semantics if
1263 : vectorized in their current location. These are stored in order that they
1264 : need to be moved. */
1265 : auto_vec<gimple *> early_break_stores;
1266 :
1267 : /* The final basic block where to move statements to. In the case of
1268 : multiple exits this could be pretty far away. */
1269 : basic_block early_break_dest_bb;
1270 :
1271 : /* Statements whose VUSES need updating if early break vectorization is to
1272 : happen. */
1273 : auto_vec<gimple*> early_break_vuses;
1274 :
1275 : /* The IV adjustment value for inductions that needs to be materialized
1276 : inside the relevant exit blocks in order to adjust for early break. */
1277 : tree early_break_niters_var;
1278 :
1279 : /* The type of the variable to be used to create the scalar IV for early break
1280 : loops. */
1281 : tree early_break_iv_type;
1282 :
1283 : /* Record statements that are needed to be live for early break vectorization
1284 : but may not have an LC PHI node materialized yet in the exits. */
1285 : auto_vec<stmt_vec_info> early_break_live_ivs;
1286 : } *loop_vec_info;
1287 :
1288 : /* Access Functions. */
1289 : #define LOOP_VINFO_LOOP(L) (L)->loop
1290 : #define LOOP_VINFO_MAIN_EXIT(L) (L)->vec_loop_main_exit
1291 : #define LOOP_VINFO_EPILOGUE_MAIN_EXIT(L) (L)->vec_epilogue_loop_main_exit
1292 : #define LOOP_VINFO_SCALAR_MAIN_EXIT(L) (L)->scalar_loop_main_exit
1293 : #define LOOP_VINFO_BBS(L) (L)->bbs
1294 : #define LOOP_VINFO_NBBS(L) (L)->nbbs
1295 : #define LOOP_VINFO_NITERSM1(L) (L)->num_itersm1
1296 : #define LOOP_VINFO_NITERS(L) (L)->num_iters
1297 : #define LOOP_VINFO_NITERS_UNCOUNTED_P(L) (LOOP_VINFO_NITERS (L) \
1298 : == chrec_dont_know)
1299 : /* Since LOOP_VINFO_NITERS and LOOP_VINFO_NITERSM1 can change after
1300 : prologue peeling retain total unchanged scalar loop iterations for
1301 : cost model. */
1302 : #define LOOP_VINFO_NITERS_UNCHANGED(L) (L)->num_iters_unchanged
1303 : #define LOOP_VINFO_NITERS_ASSUMPTIONS(L) (L)->num_iters_assumptions
1304 : #define LOOP_VINFO_COST_MODEL_THRESHOLD(L) (L)->th
1305 : #define LOOP_VINFO_VERSIONING_THRESHOLD(L) (L)->versioning_threshold
1306 : #define LOOP_VINFO_VECTORIZABLE_P(L) (L)->vectorizable
1307 : #define LOOP_VINFO_CAN_USE_PARTIAL_VECTORS_P(L) (L)->can_use_partial_vectors_p
1308 : #define LOOP_VINFO_MUST_USE_PARTIAL_VECTORS_P(L) (L)->must_use_partial_vectors_p
1309 : #define LOOP_VINFO_USING_PARTIAL_VECTORS_P(L) (L)->using_partial_vectors_p
1310 : #define LOOP_VINFO_USING_DECREMENTING_IV_P(L) (L)->using_decrementing_iv_p
1311 : #define LOOP_VINFO_USING_SELECT_VL_P(L) (L)->using_select_vl_p
1312 : #define LOOP_VINFO_ALLOW_MUTUAL_ALIGNMENT(L) (L)->allow_mutual_alignment
1313 : #define LOOP_VINFO_PARTIAL_LOAD_STORE_BIAS(L) (L)->partial_load_store_bias
1314 : #define LOOP_VINFO_VECT_FACTOR(L) (L)->vectorization_factor
1315 : #define LOOP_VINFO_IV_INCREMENT(L) (L)->iv_increment
1316 : #define LOOP_VINFO_IV_INCREMENT_INVARIANT_P(L) \
1317 : (!LOOP_VINFO_USING_SELECT_VL_P (L))
1318 : #define LOOP_VINFO_MAX_VECT_FACTOR(L) (L)->max_vectorization_factor
1319 : #define LOOP_VINFO_MASKS(L) (L)->masks
1320 : #define LOOP_VINFO_LENS(L) (L)->lens
1321 : #define LOOP_VINFO_MASK_SKIP_NITERS(L) (L)->mask_skip_niters
1322 : #define LOOP_VINFO_MASK_NITERS_PFA_OFFSET(L) (L)->mask_skip_niters_pfa_offset
1323 : #define LOOP_VINFO_RGROUP_COMPARE_TYPE(L) (L)->rgroup_compare_type
1324 : #define LOOP_VINFO_RGROUP_IV_TYPE(L) (L)->rgroup_iv_type
1325 : #define LOOP_VINFO_PARTIAL_VECTORS_STYLE(L) (L)->partial_vector_style
1326 : #define LOOP_VINFO_PTR_MASK(L) (L)->ptr_mask
1327 : #define LOOP_VINFO_MAX_SPEC_READ_AMOUNT(L) (L)->max_spec_read_amount
1328 : #define LOOP_VINFO_LOOP_NEST(L) (L)->shared->loop_nest
1329 : #define LOOP_VINFO_DATAREFS(L) (L)->shared->datarefs
1330 : #define LOOP_VINFO_DDRS(L) (L)->shared->ddrs
1331 : #define LOOP_VINFO_INT_NITERS(L) (TREE_INT_CST_LOW ((L)->num_iters))
1332 : #define LOOP_VINFO_PEELING_FOR_ALIGNMENT(L) (L)->peeling_for_alignment
1333 : #define LOOP_VINFO_NON_LINEAR_IV(L) (L)->nonlinear_iv
1334 : #define LOOP_VINFO_UNALIGNED_DR(L) (L)->unaligned_dr
1335 : #define LOOP_VINFO_MAY_MISALIGN_STMTS(L) (L)->may_misalign_stmts
1336 : #define LOOP_VINFO_MAY_ALIAS_DDRS(L) (L)->may_alias_ddrs
1337 : #define LOOP_VINFO_COMP_ALIAS_DDRS(L) (L)->comp_alias_ddrs
1338 : #define LOOP_VINFO_CHECK_UNEQUAL_ADDRS(L) (L)->check_unequal_addrs
1339 : #define LOOP_VINFO_CHECK_NONZERO(L) (L)->check_nonzero
1340 : #define LOOP_VINFO_LOWER_BOUNDS(L) (L)->lower_bounds
1341 : #define LOOP_VINFO_USER_UNROLL(L) (L)->user_unroll
1342 : #define LOOP_VINFO_GROUPED_STORES(L) (L)->grouped_stores
1343 : #define LOOP_VINFO_SLP_INSTANCES(L) (L)->slp_instances
1344 : #define LOOP_VINFO_REDUCTIONS(L) (L)->reductions
1345 : #define LOOP_VINFO_PEELING_FOR_GAPS(L) (L)->peeling_for_gaps
1346 : #define LOOP_VINFO_PEELING_FOR_NITER(L) (L)->peeling_for_niter
1347 : #define LOOP_VINFO_EARLY_BREAKS(L) (L)->early_breaks
1348 : #define LOOP_VINFO_EARLY_BRK_NEEDS_EPILOG(L) (L)->early_break_needs_epilogue
1349 : #define LOOP_VINFO_EARLY_BRK_STORES(L) (L)->early_break_stores
1350 : #define LOOP_VINFO_EARLY_BREAKS_VECT_PEELED(L) \
1351 : ((single_pred ((L)->loop->latch) != (L)->vec_loop_main_exit->src) \
1352 : || LOOP_VINFO_NITERS_UNCOUNTED_P (L))
1353 : #define LOOP_VINFO_EARLY_BREAKS_LIVE_IVS(L) \
1354 : (L)->early_break_live_ivs
1355 : #define LOOP_VINFO_EARLY_BRK_DEST_BB(L) (L)->early_break_dest_bb
1356 : #define LOOP_VINFO_EARLY_BRK_VUSES(L) (L)->early_break_vuses
1357 : #define LOOP_VINFO_EARLY_BRK_NITERS_VAR(L) (L)->early_break_niters_var
1358 : #define LOOP_VINFO_EARLY_BRK_IV_TYPE(L) (L)->early_break_iv_type
1359 : #define LOOP_VINFO_LOOP_CONDS(L) (L)->conds
1360 : #define LOOP_VINFO_LOOP_IV_COND(L) (L)->loop_iv_cond
1361 : #define LOOP_VINFO_NO_DATA_DEPENDENCIES(L) (L)->no_data_dependencies
1362 : #define LOOP_VINFO_SCALAR_LOOP(L) (L)->scalar_loop
1363 : #define LOOP_VINFO_SCALAR_LOOP_SCALING(L) (L)->scalar_loop_scaling
1364 : #define LOOP_VINFO_HAS_MASK_STORE(L) (L)->has_mask_store
1365 : #define LOOP_VINFO_SCALAR_ITERATION_COST(L) (L)->scalar_cost_vec
1366 : #define LOOP_VINFO_MAIN_LOOP_INFO(L) (L)->main_loop_info
1367 : #define LOOP_VINFO_ORIG_LOOP_INFO(L) (L)->orig_loop_info
1368 : #define LOOP_VINFO_SIMD_IF_COND(L) (L)->simd_if_cond
1369 : #define LOOP_VINFO_INNER_LOOP_COST_FACTOR(L) (L)->inner_loop_cost_factor
1370 : #define LOOP_VINFO_INV_PATTERN_DEF_SEQ(L) (L)->inv_pattern_def_seq
1371 : #define LOOP_VINFO_DRS_ADVANCED_BY(L) (L)->drs_advanced_by
1372 : #define LOOP_VINFO_ALTERNATE_DEFS(L) (L)->alternate_defs
1373 :
1374 : #define LOOP_VINFO_FULLY_MASKED_P(L) \
1375 : (LOOP_VINFO_USING_PARTIAL_VECTORS_P (L) \
1376 : && !LOOP_VINFO_MASKS (L).is_empty ())
1377 :
1378 : #define LOOP_VINFO_FULLY_WITH_LENGTH_P(L) \
1379 : (LOOP_VINFO_USING_PARTIAL_VECTORS_P (L) \
1380 : && !LOOP_VINFO_LENS (L).is_empty ())
1381 :
1382 : #define LOOP_REQUIRES_VERSIONING_FOR_ALIGNMENT(L) \
1383 : ((L)->may_misalign_stmts.length () > 0)
1384 : #define LOOP_REQUIRES_VERSIONING_FOR_SPEC_READ(L) \
1385 : (maybe_gt ((L)->max_spec_read_amount, 0U))
1386 : #define LOOP_REQUIRES_VERSIONING_FOR_ALIAS(L) \
1387 : ((L)->comp_alias_ddrs.length () > 0 \
1388 : || (L)->check_unequal_addrs.length () > 0 \
1389 : || (L)->lower_bounds.length () > 0)
1390 : #define LOOP_REQUIRES_VERSIONING_FOR_NITERS(L) \
1391 : (LOOP_VINFO_NITERS_ASSUMPTIONS (L))
1392 : #define LOOP_REQUIRES_VERSIONING_FOR_SIMD_IF_COND(L) \
1393 : (LOOP_VINFO_SIMD_IF_COND (L))
1394 : #define LOOP_REQUIRES_VERSIONING(L) \
1395 : (LOOP_REQUIRES_VERSIONING_FOR_ALIGNMENT (L) \
1396 : || LOOP_REQUIRES_VERSIONING_FOR_SPEC_READ (L) \
1397 : || LOOP_REQUIRES_VERSIONING_FOR_ALIAS (L) \
1398 : || LOOP_REQUIRES_VERSIONING_FOR_NITERS (L) \
1399 : || LOOP_REQUIRES_VERSIONING_FOR_SIMD_IF_COND (L))
1400 :
1401 : #define LOOP_VINFO_USE_VERSIONING_WITHOUT_PEELING(L) \
1402 : ((L)->may_misalign_stmts.length () > 0 \
1403 : && !LOOP_VINFO_ALLOW_MUTUAL_ALIGNMENT (L))
1404 :
1405 : #define LOOP_VINFO_NITERS_KNOWN_P(L) \
1406 : (tree_fits_shwi_p ((L)->num_iters) && tree_to_shwi ((L)->num_iters) > 0)
1407 :
1408 : #define LOOP_VINFO_EPILOGUE_P(L) \
1409 : (LOOP_VINFO_ORIG_LOOP_INFO (L) != NULL)
1410 :
1411 : #define LOOP_VINFO_ORIG_MAX_VECT_FACTOR(L) \
1412 : (LOOP_VINFO_MAX_VECT_FACTOR (LOOP_VINFO_ORIG_LOOP_INFO (L)))
1413 :
1414 : /* Wrapper for loop_vec_info, for tracking success/failure, where a non-NULL
1415 : value signifies success, and a NULL value signifies failure, supporting
1416 : propagating an opt_problem * describing the failure back up the call
1417 : stack. */
1418 : typedef opt_pointer_wrapper <loop_vec_info> opt_loop_vec_info;
1419 :
1420 : inline loop_vec_info
1421 542845 : loop_vec_info_for_loop (class loop *loop)
1422 : {
1423 542845 : return (loop_vec_info) loop->aux;
1424 : }
1425 :
1426 : struct slp_root
1427 : {
1428 1339823 : slp_root (slp_instance_kind kind_, vec<stmt_vec_info> stmts_,
1429 14422 : vec<stmt_vec_info> roots_, vec<tree> remain_ = vNULL)
1430 1339823 : : kind(kind_), stmts(stmts_), roots(roots_), remain(remain_) {}
1431 : slp_instance_kind kind;
1432 : vec<stmt_vec_info> stmts;
1433 : vec<stmt_vec_info> roots;
1434 : vec<tree> remain;
1435 : };
1436 :
1437 : typedef class _bb_vec_info : public vec_info
1438 : {
1439 : public:
1440 : _bb_vec_info (vec<basic_block> bbs, vec_info_shared *);
1441 : ~_bb_vec_info ();
1442 :
1443 : vec<slp_root> roots;
1444 : } *bb_vec_info;
1445 :
1446 : #define BB_VINFO_BBS(B) (B)->bbs
1447 : #define BB_VINFO_NBBS(B) (B)->nbbs
1448 : #define BB_VINFO_GROUPED_STORES(B) (B)->grouped_stores
1449 : #define BB_VINFO_SLP_INSTANCES(B) (B)->slp_instances
1450 : #define BB_VINFO_DATAREFS(B) (B)->shared->datarefs
1451 : #define BB_VINFO_DDRS(B) (B)->shared->ddrs
1452 :
1453 : /* Indicates whether/how a variable is used in the scope of loop/basic
1454 : block. */
1455 : enum vect_relevant {
1456 : vect_unused_in_scope = 0,
1457 :
1458 : /* The def is only used outside the loop. */
1459 : vect_used_only_live,
1460 : /* The def is in the inner loop, and the use is in the outer loop, and the
1461 : use is a reduction stmt. */
1462 : vect_used_in_outer_by_reduction,
1463 : /* The def is in the inner loop, and the use is in the outer loop (and is
1464 : not part of reduction). */
1465 : vect_used_in_outer,
1466 :
1467 : /* defs that feed computations that end up (only) in a reduction. These
1468 : defs may be used by non-reduction stmts, but eventually, any
1469 : computations/values that are affected by these defs are used to compute
1470 : a reduction (i.e. don't get stored to memory, for example). We use this
1471 : to identify computations that we can change the order in which they are
1472 : computed. */
1473 : vect_used_by_reduction,
1474 :
1475 : vect_used_in_scope
1476 : };
1477 :
1478 : /* The type of vectorization. pure_slp means the stmt is covered by the
1479 : SLP graph, not_vect means it is not. This is mostly used by BB
1480 : vectorization. */
1481 : enum slp_vect_type {
1482 : not_vect = 0,
1483 : pure_slp,
1484 : };
1485 :
1486 : /* Says whether a statement is a load, a store of a vectorized statement
1487 : result, or a store of an invariant value. */
1488 : enum vec_load_store_type {
1489 : VLS_LOAD,
1490 : VLS_STORE,
1491 : VLS_STORE_INVARIANT
1492 : };
1493 :
1494 : class dr_vec_info {
1495 : public:
1496 : /* The data reference itself. */
1497 : data_reference *dr;
1498 : /* The statement that contains the data reference. */
1499 : stmt_vec_info stmt;
1500 : /* The analysis group this DR belongs to when doing BB vectorization.
1501 : DRs of the same group belong to the same conditional execution context. */
1502 : unsigned group;
1503 : /* The misalignment in bytes of the reference, or -1 if not known. */
1504 : int misalignment;
1505 : /* The byte alignment that we'd ideally like the reference to have,
1506 : and the value that misalignment is measured against. */
1507 : poly_uint64 target_alignment;
1508 : /* If true the alignment of base_decl needs to be increased. */
1509 : bool base_misaligned;
1510 :
1511 : /* Set by early break vectorization when this DR needs peeling for alignment
1512 : for correctness. */
1513 : bool safe_speculative_read_required;
1514 :
1515 : /* Set by early break vectorization when this DR's scalar accesses are known
1516 : to be inbounds of a known bounds loop. */
1517 : bool scalar_access_known_in_bounds;
1518 :
1519 : tree base_decl;
1520 :
1521 : /* Stores current vectorized loop's offset. To be added to the DR's
1522 : offset to calculate current offset of data reference. */
1523 : tree offset;
1524 : };
1525 :
1526 : typedef struct data_reference *dr_p;
1527 :
1528 : class _stmt_vec_info {
1529 : public:
1530 :
1531 : /* Indicates whether this stmts is part of a computation whose result is
1532 : used outside the loop. */
1533 : bool live;
1534 :
1535 : /* Stmt is part of some pattern (computation idiom) */
1536 : bool in_pattern_p;
1537 :
1538 : /* True if the statement was created during pattern recognition as
1539 : part of the replacement for RELATED_STMT. This implies that the
1540 : statement isn't part of any basic block, although for convenience
1541 : its gimple_bb is the same as for RELATED_STMT. */
1542 : bool pattern_stmt_p;
1543 :
1544 : /* Is this statement vectorizable or should it be skipped in (partial)
1545 : vectorization. */
1546 : bool vectorizable;
1547 :
1548 : /* The stmt to which this info struct refers to. */
1549 : gimple *stmt;
1550 :
1551 : /* The vector type to be used for the LHS of this statement. */
1552 : tree vectype;
1553 :
1554 : /* The following is relevant only for stmts that contain a non-scalar
1555 : data-ref (array/pointer/struct access). A GIMPLE stmt is expected to have
1556 : at most one such data-ref. */
1557 :
1558 : dr_vec_info dr_aux;
1559 :
1560 : /* Information about the data-ref relative to this loop
1561 : nest (the loop that is being considered for vectorization). */
1562 : innermost_loop_behavior dr_wrt_vec_loop;
1563 :
1564 : /* For loop PHI nodes, the base and evolution part of it. This makes sure
1565 : this information is still available in vect_update_ivs_after_vectorizer
1566 : where we may not be able to re-analyze the PHI nodes evolution as
1567 : peeling for the prologue loop can make it unanalyzable. The evolution
1568 : part is still correct after peeling, but the base may have changed from
1569 : the version here. */
1570 : tree loop_phi_evolution_base_unchanged;
1571 : tree loop_phi_evolution_part;
1572 : enum vect_induction_op_type loop_phi_evolution_type;
1573 :
1574 : /* Used for various bookkeeping purposes, generally holding a pointer to
1575 : some other stmt S that is in some way "related" to this stmt.
1576 : Current use of this field is:
1577 : If this stmt is part of a pattern (i.e. the field 'in_pattern_p' is
1578 : true): S is the "pattern stmt" that represents (and replaces) the
1579 : sequence of stmts that constitutes the pattern. Similarly, the
1580 : related_stmt of the "pattern stmt" points back to this stmt (which is
1581 : the last stmt in the original sequence of stmts that constitutes the
1582 : pattern). */
1583 : stmt_vec_info related_stmt;
1584 :
1585 : /* Used to keep a sequence of def stmts of a pattern stmt if such exists.
1586 : The sequence is attached to the original statement rather than the
1587 : pattern statement. */
1588 : gimple_seq pattern_def_seq;
1589 :
1590 : /* Classify the def of this stmt. */
1591 : enum vect_def_type def_type;
1592 :
1593 : /* Whether the stmt is SLPed, loop-based vectorized, or both. */
1594 : enum slp_vect_type slp_type;
1595 :
1596 : /* Interleaving chains info. */
1597 : /* First element in the group. */
1598 : stmt_vec_info first_element;
1599 : /* Pointer to the next element in the group. */
1600 : stmt_vec_info next_element;
1601 : /* The size of the group. */
1602 : unsigned int size;
1603 : /* For loads only, the gap from the previous load. For consecutive loads, GAP
1604 : is 1. */
1605 : unsigned int gap;
1606 :
1607 : /* The minimum negative dependence distance this stmt participates in
1608 : or zero if none. */
1609 : unsigned int min_neg_dist;
1610 :
1611 : /* Not all stmts in the loop need to be vectorized. e.g, the increment
1612 : of the loop induction variable and computation of array indexes. relevant
1613 : indicates whether the stmt needs to be vectorized. */
1614 : enum vect_relevant relevant;
1615 :
1616 : /* For loads if this is a gather, for stores if this is a scatter. */
1617 : bool gather_scatter_p;
1618 :
1619 : /* True if this is an access with loop-invariant stride. */
1620 : bool strided_p;
1621 :
1622 : /* For both loads and stores. */
1623 : unsigned simd_lane_access_p : 3;
1624 :
1625 : /* On a reduction PHI the reduction type as detected by
1626 : vect_is_simple_reduction. */
1627 : enum vect_reduction_type reduc_type;
1628 :
1629 : /* On a reduction PHI, the original reduction code as detected by
1630 : vect_is_simple_reduction. */
1631 : code_helper reduc_code;
1632 :
1633 : /* On a stmt participating in a reduction the index of the operand
1634 : on the reduction SSA cycle. */
1635 : int reduc_idx;
1636 :
1637 : /* On a reduction PHI the def returned by vect_is_simple_reduction.
1638 : On the def returned by vect_is_simple_reduction the corresponding PHI. */
1639 : stmt_vec_info reduc_def;
1640 :
1641 : /* If nonzero, the lhs of the statement could be truncated to this
1642 : many bits without affecting any users of the result. */
1643 : unsigned int min_output_precision;
1644 :
1645 : /* If nonzero, all non-boolean input operands have the same precision,
1646 : and they could each be truncated to this many bits without changing
1647 : the result. */
1648 : unsigned int min_input_precision;
1649 :
1650 : /* If OPERATION_BITS is nonzero, the statement could be performed on
1651 : an integer with the sign and number of bits given by OPERATION_SIGN
1652 : and OPERATION_BITS without changing the result. */
1653 : unsigned int operation_precision;
1654 : signop operation_sign;
1655 :
1656 : /* If the statement produces a boolean result, this value describes
1657 : how we should choose the associated vector type. The possible
1658 : values are:
1659 :
1660 : - an integer precision N if we should use the vector mask type
1661 : associated with N-bit integers. This is only used if all relevant
1662 : input booleans also want the vector mask type for N-bit integers,
1663 : or if we can convert them into that form by pattern-matching.
1664 :
1665 : - ~0U if we considered choosing a vector mask type but decided
1666 : to treat the boolean as a normal integer type instead.
1667 :
1668 : - 0 otherwise. This means either that the operation isn't one that
1669 : could have a vector mask type (and so should have a normal vector
1670 : type instead) or that we simply haven't made a choice either way. */
1671 : unsigned int mask_precision;
1672 :
1673 : /* True if this is only suitable for SLP vectorization. */
1674 : bool slp_vect_only_p;
1675 : };
1676 :
1677 : /* Information about a gather/scatter call. */
1678 : struct gather_scatter_info {
1679 : /* The internal function to use for the gather/scatter operation,
1680 : or IFN_LAST if a built-in function should be used instead. */
1681 : internal_fn ifn;
1682 :
1683 : /* The FUNCTION_DECL for the built-in gather/scatter function,
1684 : or null if an internal function should be used instead. */
1685 : tree decl;
1686 :
1687 : /* The loop-invariant base value. */
1688 : tree base;
1689 :
1690 : /* The TBBA alias pointer the value of which determines the alignment
1691 : of the scalar accesses. */
1692 : tree alias_ptr;
1693 :
1694 : /* The original scalar offset, which is a non-loop-invariant SSA_NAME. */
1695 : tree offset;
1696 :
1697 : /* Each offset element should be multiplied by this amount before
1698 : being added to the base. */
1699 : int scale;
1700 :
1701 : /* The type of the vectorized offset. */
1702 : tree offset_vectype;
1703 :
1704 : /* The type of the scalar elements after loading or before storing. */
1705 : tree element_type;
1706 :
1707 : /* The type of the scalar elements being loaded or stored. */
1708 : tree memory_type;
1709 : };
1710 :
1711 : /* Access Functions. */
1712 : #define STMT_VINFO_STMT(S) (S)->stmt
1713 : #define STMT_VINFO_RELEVANT(S) (S)->relevant
1714 : #define STMT_VINFO_LIVE_P(S) (S)->live
1715 : #define STMT_VINFO_VECTYPE(S) (S)->vectype
1716 : #define STMT_VINFO_VECTORIZABLE(S) (S)->vectorizable
1717 : #define STMT_VINFO_DATA_REF(S) ((S)->dr_aux.dr + 0)
1718 : #define STMT_VINFO_GATHER_SCATTER_P(S) (S)->gather_scatter_p
1719 : #define STMT_VINFO_STRIDED_P(S) (S)->strided_p
1720 : #define STMT_VINFO_SIMD_LANE_ACCESS_P(S) (S)->simd_lane_access_p
1721 : #define STMT_VINFO_REDUC_IDX(S) (S)->reduc_idx
1722 :
1723 : #define STMT_VINFO_DR_WRT_VEC_LOOP(S) (S)->dr_wrt_vec_loop
1724 : #define STMT_VINFO_DR_BASE_ADDRESS(S) (S)->dr_wrt_vec_loop.base_address
1725 : #define STMT_VINFO_DR_INIT(S) (S)->dr_wrt_vec_loop.init
1726 : #define STMT_VINFO_DR_OFFSET(S) (S)->dr_wrt_vec_loop.offset
1727 : #define STMT_VINFO_DR_STEP(S) (S)->dr_wrt_vec_loop.step
1728 : #define STMT_VINFO_DR_BASE_ALIGNMENT(S) (S)->dr_wrt_vec_loop.base_alignment
1729 : #define STMT_VINFO_DR_BASE_MISALIGNMENT(S) \
1730 : (S)->dr_wrt_vec_loop.base_misalignment
1731 : #define STMT_VINFO_DR_OFFSET_ALIGNMENT(S) \
1732 : (S)->dr_wrt_vec_loop.offset_alignment
1733 : #define STMT_VINFO_DR_STEP_ALIGNMENT(S) \
1734 : (S)->dr_wrt_vec_loop.step_alignment
1735 :
1736 : #define STMT_VINFO_DR_INFO(S) \
1737 : (gcc_checking_assert ((S)->dr_aux.stmt == (S)), &(S)->dr_aux)
1738 :
1739 : #define STMT_VINFO_IN_PATTERN_P(S) (S)->in_pattern_p
1740 : #define STMT_VINFO_RELATED_STMT(S) (S)->related_stmt
1741 : #define STMT_VINFO_PATTERN_DEF_SEQ(S) (S)->pattern_def_seq
1742 : #define STMT_VINFO_DEF_TYPE(S) (S)->def_type
1743 : #define STMT_VINFO_GROUPED_ACCESS(S) \
1744 : ((S)->dr_aux.dr && DR_GROUP_FIRST_ELEMENT(S))
1745 : #define STMT_VINFO_LOOP_PHI_EVOLUTION_BASE_UNCHANGED(S) (S)->loop_phi_evolution_base_unchanged
1746 : #define STMT_VINFO_LOOP_PHI_EVOLUTION_PART(S) (S)->loop_phi_evolution_part
1747 : #define STMT_VINFO_LOOP_PHI_EVOLUTION_TYPE(S) (S)->loop_phi_evolution_type
1748 : #define STMT_VINFO_MIN_NEG_DIST(S) (S)->min_neg_dist
1749 : #define STMT_VINFO_REDUC_TYPE(S) (S)->reduc_type
1750 : #define STMT_VINFO_REDUC_CODE(S) (S)->reduc_code
1751 : #define STMT_VINFO_REDUC_DEF(S) (S)->reduc_def
1752 : #define STMT_VINFO_SLP_VECT_ONLY(S) (S)->slp_vect_only_p
1753 : #define STMT_VINFO_REDUC_VECTYPE_IN(S) (S)->reduc_vectype_in
1754 :
1755 : #define DR_GROUP_FIRST_ELEMENT(S) \
1756 : (gcc_checking_assert ((S)->dr_aux.dr), (S)->first_element)
1757 : #define DR_GROUP_NEXT_ELEMENT(S) \
1758 : (gcc_checking_assert ((S)->dr_aux.dr), (S)->next_element)
1759 : #define DR_GROUP_SIZE(S) \
1760 : (gcc_checking_assert ((S)->dr_aux.dr), (S)->size)
1761 : #define DR_GROUP_GAP(S) \
1762 : (gcc_checking_assert ((S)->dr_aux.dr), (S)->gap)
1763 :
1764 : #define STMT_VINFO_RELEVANT_P(S) ((S)->relevant != vect_unused_in_scope)
1765 :
1766 : #define PURE_SLP_STMT(S) ((S)->slp_type == pure_slp)
1767 : #define STMT_SLP_TYPE(S) (S)->slp_type
1768 :
1769 :
1770 : /* Contains the scalar or vector costs for a vec_info. */
1771 : class vector_costs
1772 : {
1773 : public:
1774 : vector_costs (vec_info *, bool);
1775 0 : virtual ~vector_costs () {}
1776 :
1777 : /* Update the costs in response to adding COUNT copies of a statement.
1778 :
1779 : - WHERE specifies whether the cost occurs in the loop prologue,
1780 : the loop body, or the loop epilogue.
1781 : - KIND is the kind of statement, which is always meaningful.
1782 : - STMT_INFO or NODE, if nonnull, describe the statement that will be
1783 : vectorized.
1784 : - VECTYPE, if nonnull, is the vector type that the vectorized
1785 : statement will operate on. Note that this should be used in
1786 : preference to STMT_VINFO_VECTYPE (STMT_INFO) since the latter
1787 : is not correct for SLP.
1788 : - for unaligned_load and unaligned_store statements, MISALIGN is
1789 : the byte misalignment of the load or store relative to the target's
1790 : preferred alignment for VECTYPE, or DR_MISALIGNMENT_UNKNOWN
1791 : if the misalignment is not known.
1792 :
1793 : Return the calculated cost as well as recording it. The return
1794 : value is used for dumping purposes. */
1795 : virtual unsigned int add_stmt_cost (int count, vect_cost_for_stmt kind,
1796 : stmt_vec_info stmt_info,
1797 : slp_tree node,
1798 : tree vectype, int misalign,
1799 : vect_cost_model_location where);
1800 :
1801 : /* Update the costs in response to adding costs in V which are all from
1802 : vectorizing NODE to the respective part. */
1803 : virtual unsigned int add_slp_cost (slp_tree node,
1804 : const array_slice<stmt_info_for_cost> &v);
1805 :
1806 : /* Finish calculating the cost of the code. The results can be
1807 : read back using the functions below.
1808 :
1809 : If the costs describe vector code, SCALAR_COSTS gives the costs
1810 : of the corresponding scalar code, otherwise it is null. */
1811 : virtual void finish_cost (const vector_costs *scalar_costs);
1812 :
1813 : /* The costs in THIS and OTHER both describe ways of vectorizing
1814 : a main loop. Return true if the costs described by THIS are
1815 : cheaper than the costs described by OTHER. Return false if any
1816 : of the following are true:
1817 :
1818 : - THIS and OTHER are of equal cost
1819 : - OTHER is better than THIS
1820 : - we can't be sure about the relative costs of THIS and OTHER. */
1821 : virtual bool better_main_loop_than_p (const vector_costs *other) const;
1822 :
1823 : /* Likewise, but the costs in THIS and OTHER both describe ways of
1824 : vectorizing an epilogue loop of MAIN_LOOP. */
1825 : virtual bool better_epilogue_loop_than_p (const vector_costs *other,
1826 : loop_vec_info main_loop) const;
1827 :
1828 : unsigned int prologue_cost () const;
1829 : unsigned int body_cost () const;
1830 : unsigned int epilogue_cost () const;
1831 : unsigned int outside_cost () const;
1832 : unsigned int total_cost () const;
1833 :
1834 : unsigned int suggested_unroll_factor () const;
1835 : machine_mode suggested_epilogue_mode (int &masked) const;
1836 :
1837 32564 : vec_info *vinfo () const { return m_vinfo; }
1838 7902386 : bool costing_for_scalar () const { return m_costing_for_scalar; }
1839 :
1840 : protected:
1841 : unsigned int record_stmt_cost (stmt_vec_info, vect_cost_model_location,
1842 : unsigned int);
1843 : unsigned int adjust_cost_for_freq (stmt_vec_info, vect_cost_model_location,
1844 : unsigned int);
1845 : int compare_inside_loop_cost (const vector_costs *) const;
1846 : int compare_outside_loop_cost (const vector_costs *) const;
1847 :
1848 : /* The region of code that we're considering vectorizing. */
1849 : vec_info *m_vinfo;
1850 :
1851 : /* True if we're costing the scalar code, false if we're costing
1852 : the vector code. */
1853 : bool m_costing_for_scalar;
1854 :
1855 : /* The costs of the three regions, indexed by vect_cost_model_location. */
1856 : unsigned int m_costs[3];
1857 :
1858 : /* The suggested unrolling factor determined at finish_cost. */
1859 : unsigned int m_suggested_unroll_factor;
1860 :
1861 : /* The suggested mode to be used for a vectorized epilogue or VOIDmode,
1862 : determined at finish_cost. m_masked_epilogue specifies whether the
1863 : epilogue should use masked vectorization, regardless of the
1864 : --param vect-partial-vector-usage default. If -1 then the
1865 : --param setting takes precedence. If the user explicitly specified
1866 : --param vect-partial-vector-usage then that takes precedence. */
1867 : machine_mode m_suggested_epilogue_mode;
1868 : int m_masked_epilogue;
1869 :
1870 : /* True if finish_cost has been called. */
1871 : bool m_finished;
1872 : };
1873 :
1874 : /* Create costs for VINFO. COSTING_FOR_SCALAR is true if the costs
1875 : are for scalar code, false if they are for vector code. */
1876 :
1877 : inline
1878 2206735 : vector_costs::vector_costs (vec_info *vinfo, bool costing_for_scalar)
1879 2206735 : : m_vinfo (vinfo),
1880 2206735 : m_costing_for_scalar (costing_for_scalar),
1881 2206735 : m_costs (),
1882 2206735 : m_suggested_unroll_factor(1),
1883 2206735 : m_suggested_epilogue_mode(VOIDmode),
1884 2206735 : m_masked_epilogue (-1),
1885 2206735 : m_finished (false)
1886 : {
1887 : }
1888 :
1889 : /* Return the cost of the prologue code (in abstract units). */
1890 :
1891 : inline unsigned int
1892 1335102 : vector_costs::prologue_cost () const
1893 : {
1894 1335102 : gcc_checking_assert (m_finished);
1895 1335102 : return m_costs[vect_prologue];
1896 : }
1897 :
1898 : /* Return the cost of the body code (in abstract units). */
1899 :
1900 : inline unsigned int
1901 2096520 : vector_costs::body_cost () const
1902 : {
1903 2096520 : gcc_checking_assert (m_finished);
1904 2096520 : return m_costs[vect_body];
1905 : }
1906 :
1907 : /* Return the cost of the epilogue code (in abstract units). */
1908 :
1909 : inline unsigned int
1910 1335102 : vector_costs::epilogue_cost () const
1911 : {
1912 1335102 : gcc_checking_assert (m_finished);
1913 1335102 : return m_costs[vect_epilogue];
1914 : }
1915 :
1916 : /* Return the cost of the prologue and epilogue code (in abstract units). */
1917 :
1918 : inline unsigned int
1919 509609 : vector_costs::outside_cost () const
1920 : {
1921 509609 : return prologue_cost () + epilogue_cost ();
1922 : }
1923 :
1924 : /* Return the cost of the prologue, body and epilogue code
1925 : (in abstract units). */
1926 :
1927 : inline unsigned int
1928 125835 : vector_costs::total_cost () const
1929 : {
1930 125835 : return body_cost () + outside_cost ();
1931 : }
1932 :
1933 : /* Return the suggested unroll factor. */
1934 :
1935 : inline unsigned int
1936 125445 : vector_costs::suggested_unroll_factor () const
1937 : {
1938 125445 : gcc_checking_assert (m_finished);
1939 125445 : return m_suggested_unroll_factor;
1940 : }
1941 :
1942 : /* Return the suggested epilogue mode. */
1943 :
1944 : inline machine_mode
1945 14428 : vector_costs::suggested_epilogue_mode (int &masked_p) const
1946 : {
1947 14428 : gcc_checking_assert (m_finished);
1948 14428 : masked_p = m_masked_epilogue;
1949 14428 : return m_suggested_epilogue_mode;
1950 : }
1951 :
1952 : #define VECT_MAX_COST 1000
1953 :
1954 : /* The maximum number of intermediate steps required in multi-step type
1955 : conversion. */
1956 : #define MAX_INTERM_CVT_STEPS 3
1957 :
1958 : #define MAX_VECTORIZATION_FACTOR INT_MAX
1959 :
1960 : /* Nonzero if TYPE represents a (scalar) boolean type or type
1961 : in the middle-end compatible with it (unsigned precision 1 integral
1962 : types). Used to determine which types should be vectorized as
1963 : VECTOR_BOOLEAN_TYPE_P. */
1964 :
1965 : #define VECT_SCALAR_BOOLEAN_TYPE_P(TYPE) \
1966 : (TREE_CODE (TYPE) == BOOLEAN_TYPE \
1967 : || ((TREE_CODE (TYPE) == INTEGER_TYPE \
1968 : || TREE_CODE (TYPE) == ENUMERAL_TYPE) \
1969 : && TYPE_PRECISION (TYPE) == 1 \
1970 : && TYPE_UNSIGNED (TYPE)))
1971 :
1972 : inline bool
1973 11892456 : nested_in_vect_loop_p (class loop *loop, stmt_vec_info stmt_info)
1974 : {
1975 11892456 : return (loop->inner
1976 9567925 : && (loop->inner == (gimple_bb (stmt_info->stmt))->loop_father));
1977 : }
1978 :
1979 : /* PHI is either a scalar reduction phi or a scalar induction phi.
1980 : Return the initial value of the variable on entry to the containing
1981 : loop. */
1982 :
1983 : inline tree
1984 34501 : vect_phi_initial_value (gphi *phi)
1985 : {
1986 34501 : basic_block bb = gimple_bb (phi);
1987 34501 : edge pe = loop_preheader_edge (bb->loop_father);
1988 34501 : gcc_assert (pe->dest == bb);
1989 34501 : return PHI_ARG_DEF_FROM_EDGE (phi, pe);
1990 : }
1991 :
1992 : /* Return true if STMT_INFO should produce a vector mask type rather than
1993 : a normal nonmask type. */
1994 :
1995 : inline bool
1996 7634529 : vect_use_mask_type_p (stmt_vec_info stmt_info)
1997 : {
1998 7634529 : return stmt_info->mask_precision && stmt_info->mask_precision != ~0U;
1999 : }
2000 :
2001 : /* Return TRUE if a statement represented by STMT_INFO is a part of a
2002 : pattern. */
2003 :
2004 : inline bool
2005 132422457 : is_pattern_stmt_p (stmt_vec_info stmt_info)
2006 : {
2007 85226072 : return stmt_info->pattern_stmt_p;
2008 : }
2009 :
2010 : /* If STMT_INFO is a pattern statement, return the statement that it
2011 : replaces, otherwise return STMT_INFO itself. */
2012 :
2013 : inline stmt_vec_info
2014 51673854 : vect_orig_stmt (stmt_vec_info stmt_info)
2015 : {
2016 39205707 : if (is_pattern_stmt_p (stmt_info))
2017 3836558 : return STMT_VINFO_RELATED_STMT (stmt_info);
2018 : return stmt_info;
2019 : }
2020 :
2021 : /* Return the later statement between STMT1_INFO and STMT2_INFO. */
2022 :
2023 : inline stmt_vec_info
2024 6075740 : get_later_stmt (stmt_vec_info stmt1_info, stmt_vec_info stmt2_info)
2025 : {
2026 6075740 : gimple *stmt1 = vect_orig_stmt (stmt1_info)->stmt;
2027 6075740 : gimple *stmt2 = vect_orig_stmt (stmt2_info)->stmt;
2028 6075740 : if (gimple_bb (stmt1) == gimple_bb (stmt2))
2029 : {
2030 6058712 : if (gimple_uid (stmt1) > gimple_uid (stmt2))
2031 : return stmt1_info;
2032 : else
2033 1252883 : return stmt2_info;
2034 : }
2035 : /* ??? We should be really calling this function only with stmts
2036 : in the same BB but we can recover if there's a domination
2037 : relationship between them. */
2038 17028 : else if (dominated_by_p (CDI_DOMINATORS,
2039 17028 : gimple_bb (stmt1), gimple_bb (stmt2)))
2040 : return stmt1_info;
2041 6065 : else if (dominated_by_p (CDI_DOMINATORS,
2042 6065 : gimple_bb (stmt2), gimple_bb (stmt1)))
2043 : return stmt2_info;
2044 0 : gcc_unreachable ();
2045 : }
2046 :
2047 : /* If STMT_INFO has been replaced by a pattern statement, return the
2048 : replacement statement, otherwise return STMT_INFO itself. */
2049 :
2050 : inline stmt_vec_info
2051 54017138 : vect_stmt_to_vectorize (stmt_vec_info stmt_info)
2052 : {
2053 54017138 : if (STMT_VINFO_IN_PATTERN_P (stmt_info))
2054 1649916 : return STMT_VINFO_RELATED_STMT (stmt_info);
2055 : return stmt_info;
2056 : }
2057 :
2058 : /* Return true if BB is a loop header. */
2059 :
2060 : inline bool
2061 1560677 : is_loop_header_bb_p (basic_block bb)
2062 : {
2063 1560677 : if (bb == (bb->loop_father)->header)
2064 1549872 : return true;
2065 :
2066 : return false;
2067 : }
2068 :
2069 : /* Return pow2 (X). */
2070 :
2071 : inline int
2072 : vect_pow2 (int x)
2073 : {
2074 : int i, res = 1;
2075 :
2076 : for (i = 0; i < x; i++)
2077 : res *= 2;
2078 :
2079 : return res;
2080 : }
2081 :
2082 : /* Alias targetm.vectorize.builtin_vectorization_cost. */
2083 :
2084 : inline int
2085 9368190 : builtin_vectorization_cost (enum vect_cost_for_stmt type_of_cost,
2086 : tree vectype, int misalign)
2087 : {
2088 9300252 : return targetm.vectorize.builtin_vectorization_cost (type_of_cost,
2089 : vectype, misalign);
2090 : }
2091 :
2092 : /* Get cost by calling cost target builtin. */
2093 :
2094 : inline
2095 155 : int vect_get_stmt_cost (enum vect_cost_for_stmt type_of_cost)
2096 : {
2097 67783 : return builtin_vectorization_cost (type_of_cost, NULL, 0);
2098 : }
2099 :
2100 : /* Alias targetm.vectorize.init_cost. */
2101 :
2102 : inline vector_costs *
2103 2206735 : init_cost (vec_info *vinfo, bool costing_for_scalar)
2104 : {
2105 2206735 : return targetm.vectorize.create_costs (vinfo, costing_for_scalar);
2106 : }
2107 :
2108 : extern void dump_stmt_cost (FILE *, int, enum vect_cost_for_stmt,
2109 : stmt_vec_info, slp_tree, tree, int, unsigned,
2110 : enum vect_cost_model_location);
2111 :
2112 : /* Dump and add costs. */
2113 :
2114 : inline unsigned
2115 7902386 : add_stmt_cost (vector_costs *costs, int count,
2116 : enum vect_cost_for_stmt kind,
2117 : stmt_vec_info stmt_info, slp_tree node,
2118 : tree vectype, int misalign,
2119 : enum vect_cost_model_location where)
2120 : {
2121 : /* Even though a vector type might be set on stmt do not pass that on when
2122 : costing the scalar IL. A SLP node shouldn't have been recorded. */
2123 7902386 : if (costs->costing_for_scalar ())
2124 : {
2125 4056091 : vectype = NULL_TREE;
2126 4056091 : gcc_checking_assert (node == NULL);
2127 : }
2128 7902386 : unsigned cost = costs->add_stmt_cost (count, kind, stmt_info, node, vectype,
2129 : misalign, where);
2130 7902386 : if (dump_file && (dump_flags & TDF_DETAILS))
2131 220535 : dump_stmt_cost (dump_file, count, kind, stmt_info, node, vectype, misalign,
2132 : cost, where);
2133 7902386 : return cost;
2134 : }
2135 :
2136 : inline unsigned
2137 83082 : add_stmt_cost (vector_costs *costs, int count, enum vect_cost_for_stmt kind,
2138 : enum vect_cost_model_location where)
2139 : {
2140 83082 : gcc_assert (kind == cond_branch_taken || kind == cond_branch_not_taken
2141 : || kind == scalar_stmt);
2142 83082 : return add_stmt_cost (costs, count, kind, NULL, NULL, NULL_TREE, 0, where);
2143 : }
2144 :
2145 : inline unsigned
2146 2282516 : add_stmt_cost (vector_costs *costs, stmt_info_for_cost *i)
2147 : {
2148 2282516 : return add_stmt_cost (costs, i->count, i->kind, i->stmt_info, i->node,
2149 2282516 : i->vectype, i->misalign, i->where);
2150 : }
2151 :
2152 : inline void
2153 373736 : add_stmt_costs (vector_costs *costs, stmt_vector_for_cost *cost_vec)
2154 : {
2155 373736 : stmt_info_for_cost *cost;
2156 373736 : unsigned i;
2157 2147311 : FOR_EACH_VEC_ELT (*cost_vec, i, cost)
2158 1773575 : add_stmt_cost (costs, cost->count, cost->kind, cost->stmt_info,
2159 : cost->node, cost->vectype, cost->misalign, cost->where);
2160 373736 : }
2161 :
2162 : /*-----------------------------------------------------------------*/
2163 : /* Info on data references alignment. */
2164 : /*-----------------------------------------------------------------*/
2165 : #define DR_MISALIGNMENT_UNKNOWN (-1)
2166 : #define DR_MISALIGNMENT_UNINITIALIZED (-2)
2167 :
2168 : inline void
2169 2663701 : set_dr_misalignment (dr_vec_info *dr_info, int val)
2170 : {
2171 2663701 : dr_info->misalignment = val;
2172 : }
2173 :
2174 : extern int dr_misalignment (dr_vec_info *dr_info, tree vectype,
2175 : poly_int64 offset = 0);
2176 :
2177 : #define SET_DR_MISALIGNMENT(DR, VAL) set_dr_misalignment (DR, VAL)
2178 :
2179 : /* Only defined once DR_MISALIGNMENT is defined. */
2180 : inline const poly_uint64
2181 8168601 : dr_target_alignment (dr_vec_info *dr_info)
2182 : {
2183 8168601 : if (STMT_VINFO_GROUPED_ACCESS (dr_info->stmt))
2184 6006088 : dr_info = STMT_VINFO_DR_INFO (DR_GROUP_FIRST_ELEMENT (dr_info->stmt));
2185 8168601 : return dr_info->target_alignment;
2186 : }
2187 : #define DR_TARGET_ALIGNMENT(DR) dr_target_alignment (DR)
2188 : #define DR_SCALAR_KNOWN_BOUNDS(DR) (DR)->scalar_access_known_in_bounds
2189 :
2190 : /* Return if the stmt_vec_info requires peeling for alignment. */
2191 : inline bool
2192 4637984 : dr_safe_speculative_read_required (stmt_vec_info stmt_info)
2193 : {
2194 4637984 : dr_vec_info *dr_info;
2195 4637984 : if (STMT_VINFO_GROUPED_ACCESS (stmt_info))
2196 1710474 : dr_info = STMT_VINFO_DR_INFO (DR_GROUP_FIRST_ELEMENT (stmt_info));
2197 : else
2198 2927510 : dr_info = STMT_VINFO_DR_INFO (stmt_info);
2199 :
2200 4637984 : return dr_info->safe_speculative_read_required;
2201 : }
2202 :
2203 : /* Set the safe_speculative_read_required for the stmt_vec_info, if group
2204 : access then set on the fist element otherwise set on DR directly. */
2205 : inline void
2206 236385 : dr_set_safe_speculative_read_required (stmt_vec_info stmt_info,
2207 : bool requires_alignment)
2208 : {
2209 236385 : dr_vec_info *dr_info;
2210 236385 : if (STMT_VINFO_GROUPED_ACCESS (stmt_info))
2211 68165 : dr_info = STMT_VINFO_DR_INFO (DR_GROUP_FIRST_ELEMENT (stmt_info));
2212 : else
2213 168220 : dr_info = STMT_VINFO_DR_INFO (stmt_info);
2214 :
2215 236385 : dr_info->safe_speculative_read_required = requires_alignment;
2216 236385 : }
2217 :
2218 : inline void
2219 1639899 : set_dr_target_alignment (dr_vec_info *dr_info, poly_uint64 val)
2220 : {
2221 1639899 : dr_info->target_alignment = val;
2222 : }
2223 : #define SET_DR_TARGET_ALIGNMENT(DR, VAL) set_dr_target_alignment (DR, VAL)
2224 :
2225 : /* Return true if data access DR_INFO is aligned to the targets
2226 : preferred alignment for VECTYPE (which may be less than a full vector). */
2227 :
2228 : inline bool
2229 394514 : aligned_access_p (dr_vec_info *dr_info, tree vectype)
2230 : {
2231 394514 : return (dr_misalignment (dr_info, vectype) == 0);
2232 : }
2233 :
2234 : /* Return TRUE if the (mis-)alignment of the data access is known with
2235 : respect to the targets preferred alignment for VECTYPE, and FALSE
2236 : otherwise. */
2237 :
2238 : inline bool
2239 2383883 : known_alignment_for_access_p (dr_vec_info *dr_info, tree vectype)
2240 : {
2241 2134812 : return (dr_misalignment (dr_info, vectype) != DR_MISALIGNMENT_UNKNOWN);
2242 : }
2243 :
2244 : /* Return the minimum alignment in bytes that the vectorized version
2245 : of DR_INFO is guaranteed to have. */
2246 :
2247 : inline unsigned int
2248 278730 : vect_known_alignment_in_bytes (dr_vec_info *dr_info, tree vectype,
2249 : poly_int64 offset = 0)
2250 : {
2251 278730 : int misalignment = dr_misalignment (dr_info, vectype, offset);
2252 278730 : if (misalignment == DR_MISALIGNMENT_UNKNOWN)
2253 136463 : return TYPE_ALIGN_UNIT (TREE_TYPE (DR_REF (dr_info->dr)));
2254 142267 : else if (misalignment == 0)
2255 98711 : return known_alignment (DR_TARGET_ALIGNMENT (dr_info));
2256 43556 : return misalignment & -misalignment;
2257 : }
2258 :
2259 : /* Return the behavior of DR_INFO with respect to the vectorization context
2260 : (which for outer loop vectorization might not be the behavior recorded
2261 : in DR_INFO itself). */
2262 :
2263 : inline innermost_loop_behavior *
2264 5893092 : vect_dr_behavior (vec_info *vinfo, dr_vec_info *dr_info)
2265 : {
2266 5893092 : stmt_vec_info stmt_info = dr_info->stmt;
2267 5893092 : loop_vec_info loop_vinfo = dyn_cast<loop_vec_info> (vinfo);
2268 2332774 : if (loop_vinfo == NULL
2269 2332774 : || !nested_in_vect_loop_p (LOOP_VINFO_LOOP (loop_vinfo), stmt_info))
2270 5888790 : return &DR_INNERMOST (dr_info->dr);
2271 : else
2272 4302 : return &STMT_VINFO_DR_WRT_VEC_LOOP (stmt_info);
2273 : }
2274 :
2275 : /* Return the offset calculated by adding the offset of this DR_INFO to the
2276 : corresponding data_reference's offset. If CHECK_OUTER then use
2277 : vect_dr_behavior to select the appropriate data_reference to use. */
2278 :
2279 : inline tree
2280 732648 : get_dr_vinfo_offset (vec_info *vinfo,
2281 : dr_vec_info *dr_info, bool check_outer = false)
2282 : {
2283 732648 : innermost_loop_behavior *base;
2284 732648 : if (check_outer)
2285 692684 : base = vect_dr_behavior (vinfo, dr_info);
2286 : else
2287 39964 : base = &dr_info->dr->innermost;
2288 :
2289 732648 : tree offset = base->offset;
2290 :
2291 732648 : if (!dr_info->offset)
2292 : return offset;
2293 :
2294 1149 : offset = fold_convert (sizetype, offset);
2295 1149 : return fold_build2 (PLUS_EXPR, TREE_TYPE (dr_info->offset), offset,
2296 : dr_info->offset);
2297 : }
2298 :
2299 :
2300 : /* Return the vect cost model for LOOP. */
2301 : inline enum vect_cost_model
2302 2489328 : loop_cost_model (loop_p loop)
2303 : {
2304 2489328 : if (loop != NULL
2305 1799331 : && loop->force_vectorize
2306 79925 : && flag_simd_cost_model != VECT_COST_MODEL_DEFAULT)
2307 : return flag_simd_cost_model;
2308 2409403 : return flag_vect_cost_model;
2309 : }
2310 :
2311 : /* Return true if the vect cost model is unlimited. */
2312 : inline bool
2313 1731927 : unlimited_cost_model (loop_p loop)
2314 : {
2315 1731927 : return loop_cost_model (loop) == VECT_COST_MODEL_UNLIMITED;
2316 : }
2317 :
2318 : /* Return true if the loop described by LOOP_VINFO is fully-masked and
2319 : if the first iteration should use a partial mask in order to achieve
2320 : alignment. */
2321 :
2322 : inline bool
2323 267502 : vect_use_loop_mask_for_alignment_p (loop_vec_info loop_vinfo)
2324 : {
2325 : /* With early break vectorization we don't know whether the accesses will stay
2326 : inside the loop or not. TODO: The early break adjustment code can be
2327 : implemented the same way as vectorizable_linear_induction. However we
2328 : can't test this today so reject it. */
2329 98 : return (LOOP_VINFO_FULLY_MASKED_P (loop_vinfo)
2330 98 : && LOOP_VINFO_PEELING_FOR_ALIGNMENT (loop_vinfo)
2331 267510 : && !(LOOP_VINFO_NON_LINEAR_IV (loop_vinfo)
2332 0 : && LOOP_VINFO_EARLY_BREAKS (loop_vinfo)));
2333 : }
2334 :
2335 : /* Return the number of vectors of type VECTYPE that are needed to get
2336 : NUNITS elements. NUNITS should be based on the vectorization factor,
2337 : so it is always a known multiple of the number of elements in VECTYPE. */
2338 :
2339 : inline unsigned int
2340 5623155 : vect_get_num_vectors (poly_uint64 nunits, tree vectype)
2341 : {
2342 5623155 : return exact_div (nunits, TYPE_VECTOR_SUBPARTS (vectype)).to_constant ();
2343 : }
2344 :
2345 : /* Return the number of vectors in the context of vectorization region VINFO,
2346 : needed for a group of statements and a vector type as specified by NODE. */
2347 :
2348 : inline unsigned int
2349 5622320 : vect_get_num_copies (vec_info *vinfo, slp_tree node)
2350 : {
2351 5622320 : poly_uint64 vf;
2352 :
2353 5622320 : if (loop_vec_info loop_vinfo = dyn_cast <loop_vec_info> (vinfo))
2354 3034706 : vf = LOOP_VINFO_VECT_FACTOR (loop_vinfo);
2355 : else
2356 : vf = 1;
2357 :
2358 5622320 : vf *= SLP_TREE_LANES (node);
2359 5622320 : tree vectype = SLP_TREE_VECTYPE (node);
2360 :
2361 5622320 : return vect_get_num_vectors (vf, vectype);
2362 : }
2363 :
2364 : /* Return the vectorization factor that should be used for costing
2365 : purposes while vectorizing the loop described by LOOP_VINFO.
2366 : Pick a reasonable estimate if the vectorization factor isn't
2367 : known at compile time. */
2368 :
2369 : inline unsigned int
2370 1443473 : vect_vf_for_cost (loop_vec_info loop_vinfo)
2371 : {
2372 1443473 : return estimated_poly_value (LOOP_VINFO_VECT_FACTOR (loop_vinfo));
2373 : }
2374 :
2375 : /* Estimate the number of elements in VEC_TYPE for costing purposes.
2376 : Pick a reasonable estimate if the exact number isn't known at
2377 : compile time. */
2378 :
2379 : inline unsigned int
2380 56784 : vect_nunits_for_cost (tree vec_type)
2381 : {
2382 56784 : return estimated_poly_value (TYPE_VECTOR_SUBPARTS (vec_type));
2383 : }
2384 :
2385 : /* Return the maximum possible vectorization factor for LOOP_VINFO. */
2386 :
2387 : inline unsigned HOST_WIDE_INT
2388 105856 : vect_max_vf (loop_vec_info loop_vinfo)
2389 : {
2390 105856 : unsigned HOST_WIDE_INT vf;
2391 105856 : if (LOOP_VINFO_VECT_FACTOR (loop_vinfo).is_constant (&vf))
2392 105856 : return vf;
2393 : return MAX_VECTORIZATION_FACTOR;
2394 : }
2395 :
2396 : /* Return the size of the value accessed by unvectorized data reference
2397 : DR_INFO. This is only valid once STMT_VINFO_VECTYPE has been calculated
2398 : for the associated gimple statement, since that guarantees that DR_INFO
2399 : accesses either a scalar or a scalar equivalent. ("Scalar equivalent"
2400 : here includes things like V1SI, which can be vectorized in the same way
2401 : as a plain SI.) */
2402 :
2403 : inline unsigned int
2404 2025313 : vect_get_scalar_dr_size (dr_vec_info *dr_info)
2405 : {
2406 2025313 : return tree_to_uhwi (TYPE_SIZE_UNIT (TREE_TYPE (DR_REF (dr_info->dr))));
2407 : }
2408 :
2409 : /* Return true if LOOP_VINFO requires a runtime check for whether the
2410 : vector loop is profitable. */
2411 :
2412 : inline bool
2413 71772 : vect_apply_runtime_profitability_check_p (loop_vec_info loop_vinfo)
2414 : {
2415 71772 : unsigned int th = LOOP_VINFO_COST_MODEL_THRESHOLD (loop_vinfo);
2416 37888 : return (!LOOP_VINFO_NITERS_KNOWN_P (loop_vinfo)
2417 71772 : && th >= vect_vf_for_cost (loop_vinfo));
2418 : }
2419 :
2420 : /* Return true if CODE is a lane-reducing opcode. */
2421 :
2422 : inline bool
2423 395048 : lane_reducing_op_p (code_helper code)
2424 : {
2425 395048 : return code == DOT_PROD_EXPR || code == WIDEN_SUM_EXPR || code == SAD_EXPR;
2426 : }
2427 :
2428 : /* Return true if STMT is a lane-reducing statement. */
2429 :
2430 : inline bool
2431 483529 : lane_reducing_stmt_p (gimple *stmt)
2432 : {
2433 483529 : if (auto *assign = dyn_cast <gassign *> (stmt))
2434 337050 : return lane_reducing_op_p (gimple_assign_rhs_code (assign));
2435 : return false;
2436 : }
2437 :
2438 : /* Source location + hotness information. */
2439 : extern dump_user_location_t vect_location;
2440 :
2441 : /* A macro for calling:
2442 : dump_begin_scope (MSG, vect_location);
2443 : via an RAII object, thus printing "=== MSG ===\n" to the dumpfile etc,
2444 : and then calling
2445 : dump_end_scope ();
2446 : once the object goes out of scope, thus capturing the nesting of
2447 : the scopes.
2448 :
2449 : These scopes affect dump messages within them: dump messages at the
2450 : top level implicitly default to MSG_PRIORITY_USER_FACING, whereas those
2451 : in a nested scope implicitly default to MSG_PRIORITY_INTERNALS. */
2452 :
2453 : #define DUMP_VECT_SCOPE(MSG) \
2454 : AUTO_DUMP_SCOPE (MSG, vect_location)
2455 :
2456 : /* A sentinel class for ensuring that the "vect_location" global gets
2457 : reset at the end of a scope.
2458 :
2459 : The "vect_location" global is used during dumping and contains a
2460 : location_t, which could contain references to a tree block via the
2461 : ad-hoc data. This data is used for tracking inlining information,
2462 : but it's not a GC root; it's simply assumed that such locations never
2463 : get accessed if the blocks are optimized away.
2464 :
2465 : Hence we need to ensure that such locations are purged at the end
2466 : of any operations using them (e.g. via this class). */
2467 :
2468 : class auto_purge_vect_location
2469 : {
2470 : public:
2471 : ~auto_purge_vect_location ();
2472 : };
2473 :
2474 : /*-----------------------------------------------------------------*/
2475 : /* Function prototypes. */
2476 : /*-----------------------------------------------------------------*/
2477 :
2478 : /* Simple loop peeling and versioning utilities for vectorizer's purposes -
2479 : in tree-vect-loop-manip.cc. */
2480 : extern void vect_set_loop_condition (class loop *, edge, loop_vec_info,
2481 : tree, tree, tree, bool);
2482 : extern bool slpeel_can_duplicate_loop_p (const class loop *, const_edge,
2483 : const_edge);
2484 : class loop *slpeel_tree_duplicate_loop_to_edge_cfg (class loop *, edge,
2485 : class loop *, edge,
2486 : edge, edge *, bool = true,
2487 : vec<basic_block> * = NULL,
2488 : bool = false, bool = false,
2489 : bool = true);
2490 : class loop *vect_loop_versioning (loop_vec_info, gimple *);
2491 : extern class loop *vect_do_peeling (loop_vec_info, tree, tree,
2492 : tree *, tree *, tree *, int, bool, bool,
2493 : tree *);
2494 : extern tree vect_get_main_loop_result (loop_vec_info, tree, tree);
2495 : extern void vect_prepare_for_masked_peels (loop_vec_info);
2496 : extern dump_user_location_t find_loop_location (class loop *);
2497 : extern bool vect_can_advance_ivs_p (loop_vec_info);
2498 : extern void vect_update_inits_of_drs (loop_vec_info, tree, tree_code);
2499 : extern edge vec_init_loop_exit_info (class loop *);
2500 : extern void vect_iv_increment_position (edge, gimple_stmt_iterator *, bool *);
2501 :
2502 : /* In tree-vect-stmts.cc. */
2503 : extern tree get_related_vectype_for_scalar_type (machine_mode, tree,
2504 : poly_uint64 = 0);
2505 : extern tree get_vectype_for_scalar_type (vec_info *, tree, unsigned int = 0);
2506 : extern tree get_vectype_for_scalar_type (vec_info *, tree, slp_tree);
2507 : extern tree get_mask_type_for_scalar_type (vec_info *, tree, unsigned int = 0);
2508 : extern tree get_mask_type_for_scalar_type (vec_info *, tree, slp_tree);
2509 : extern tree get_same_sized_vectype (tree, tree);
2510 : extern bool vect_chooses_same_modes_p (vec_info *, machine_mode);
2511 : extern bool vect_chooses_same_modes_p (machine_mode, machine_mode);
2512 : extern bool vect_get_loop_mask_type (loop_vec_info);
2513 : extern bool vect_is_simple_use (tree, vec_info *, enum vect_def_type *,
2514 : stmt_vec_info * = NULL, gimple ** = NULL);
2515 : extern bool vect_is_simple_use (vec_info *, slp_tree,
2516 : unsigned, tree *, slp_tree *,
2517 : enum vect_def_type *,
2518 : tree *, stmt_vec_info * = NULL);
2519 : extern bool vect_is_simple_use (vec_info *, slp_tree,
2520 : unsigned, slp_tree *,
2521 : enum vect_def_type *, tree *);
2522 : extern bool vect_maybe_update_slp_op_vectype (slp_tree, tree);
2523 : extern tree perm_mask_for_reverse (tree);
2524 : extern bool supportable_widening_operation (code_helper, tree, tree, bool,
2525 : code_helper*, code_helper*,
2526 : int*, vec<tree> *);
2527 : extern bool supportable_narrowing_operation (code_helper, tree, tree,
2528 : code_helper *, int *,
2529 : vec<tree> *);
2530 : extern bool supportable_indirect_convert_operation (code_helper,
2531 : tree, tree,
2532 : vec<std::pair<tree, tree_code> > &,
2533 : slp_tree = NULL);
2534 : extern int compare_step_with_zero (vec_info *, stmt_vec_info);
2535 :
2536 : extern unsigned record_stmt_cost (stmt_vector_for_cost *, int,
2537 : enum vect_cost_for_stmt, stmt_vec_info,
2538 : tree, int, enum vect_cost_model_location);
2539 : extern unsigned record_stmt_cost (stmt_vector_for_cost *, int,
2540 : enum vect_cost_for_stmt, slp_tree,
2541 : tree, int, enum vect_cost_model_location);
2542 : extern unsigned record_stmt_cost (stmt_vector_for_cost *, int,
2543 : enum vect_cost_for_stmt,
2544 : enum vect_cost_model_location);
2545 : extern unsigned record_stmt_cost (stmt_vector_for_cost *, int,
2546 : enum vect_cost_for_stmt, stmt_vec_info,
2547 : slp_tree, tree, int,
2548 : enum vect_cost_model_location);
2549 :
2550 : /* Overload of record_stmt_cost with VECTYPE derived from STMT_INFO. */
2551 :
2552 : inline unsigned
2553 1840092 : record_stmt_cost (stmt_vector_for_cost *body_cost_vec, int count,
2554 : enum vect_cost_for_stmt kind, stmt_vec_info stmt_info,
2555 : int misalign, enum vect_cost_model_location where)
2556 : {
2557 1839523 : return record_stmt_cost (body_cost_vec, count, kind, stmt_info,
2558 : STMT_VINFO_VECTYPE (stmt_info), misalign, where);
2559 : }
2560 :
2561 : /* Overload of record_stmt_cost with VECTYPE derived from SLP node. */
2562 :
2563 : inline unsigned
2564 1575313 : record_stmt_cost (stmt_vector_for_cost *body_cost_vec, int count,
2565 : enum vect_cost_for_stmt kind, slp_tree node,
2566 : int misalign, enum vect_cost_model_location where)
2567 : {
2568 1365182 : return record_stmt_cost (body_cost_vec, count, kind, node,
2569 : SLP_TREE_VECTYPE (node), misalign, where);
2570 : }
2571 :
2572 : extern void vect_finish_replace_stmt (vec_info *, stmt_vec_info, gimple *);
2573 : extern void vect_finish_stmt_generation (vec_info *, stmt_vec_info, gimple *,
2574 : gimple_stmt_iterator *);
2575 : extern opt_result vect_mark_stmts_to_be_vectorized (loop_vec_info, bool *);
2576 : extern tree vect_get_store_rhs (stmt_vec_info);
2577 : void vect_get_vec_defs (vec_info *, slp_tree,
2578 : bool, vec<tree> *,
2579 : bool = false, vec<tree> * = NULL,
2580 : bool = false, vec<tree> * = NULL,
2581 : bool = false, vec<tree> * = NULL);
2582 : extern tree vect_init_vector (vec_info *, stmt_vec_info, tree, tree,
2583 : gimple_stmt_iterator *);
2584 : extern tree vect_get_slp_vect_def (slp_tree, unsigned);
2585 : extern void vect_transform_stmt (vec_info *, stmt_vec_info,
2586 : gimple_stmt_iterator *,
2587 : slp_tree, slp_instance);
2588 : extern void vect_remove_stores (vec_info *, stmt_vec_info);
2589 : extern bool vect_nop_conversion_p (stmt_vec_info);
2590 : extern opt_result vect_analyze_stmt (vec_info *, slp_tree,
2591 : slp_instance, stmt_vector_for_cost *);
2592 : extern void vect_get_load_cost (vec_info *, stmt_vec_info, slp_tree, int,
2593 : dr_alignment_support, int, bool,
2594 : unsigned int *, unsigned int *,
2595 : stmt_vector_for_cost *,
2596 : stmt_vector_for_cost *, bool);
2597 : extern void vect_get_store_cost (vec_info *, stmt_vec_info, slp_tree, int,
2598 : dr_alignment_support, int,
2599 : unsigned int *, stmt_vector_for_cost *);
2600 : extern bool vect_supportable_shift (vec_info *, enum tree_code, tree);
2601 : extern tree vect_gen_perm_mask_any (tree, const vec_perm_indices &);
2602 : extern tree vect_gen_perm_mask_checked (tree, const vec_perm_indices &);
2603 : extern void optimize_mask_stores (class loop*);
2604 : extern tree vect_gen_while (gimple_seq *, tree, tree, tree,
2605 : const char * = nullptr);
2606 : extern tree vect_gen_while_not (gimple_seq *, tree, tree, tree);
2607 : extern opt_result vect_get_vector_types_for_stmt (vec_info *,
2608 : stmt_vec_info, tree *,
2609 : tree *, unsigned int = 0);
2610 : extern opt_tree vect_get_mask_type_for_stmt (stmt_vec_info, unsigned int = 0);
2611 :
2612 : /* In tree-if-conv.cc. */
2613 : extern bool ref_within_array_bound (gimple *, tree);
2614 :
2615 : /* In tree-vect-data-refs.cc. */
2616 : extern bool vect_can_force_dr_alignment_p (const_tree, poly_uint64);
2617 : extern enum dr_alignment_support vect_supportable_dr_alignment
2618 : (vec_info *, dr_vec_info *, tree, int,
2619 : bool = false);
2620 : extern tree vect_get_smallest_scalar_type (stmt_vec_info, tree);
2621 : extern opt_result vect_analyze_data_ref_dependences (loop_vec_info, unsigned int *);
2622 : extern bool vect_slp_analyze_instance_dependence (vec_info *, slp_instance);
2623 : extern opt_result vect_enhance_data_refs_alignment (loop_vec_info);
2624 : extern void vect_analyze_data_refs_alignment (loop_vec_info);
2625 : extern bool vect_slp_analyze_instance_alignment (vec_info *, slp_instance);
2626 : extern opt_result vect_analyze_data_ref_accesses (vec_info *, vec<int> *);
2627 : extern opt_result vect_prune_runtime_alias_test_list (loop_vec_info);
2628 : extern bool vect_gather_scatter_fn_p (vec_info *, bool, bool, tree, tree,
2629 : tree, int, int *, internal_fn *, tree *,
2630 : tree *, vec<int> * = nullptr);
2631 : extern bool vect_check_gather_scatter (stmt_vec_info, tree,
2632 : loop_vec_info, gather_scatter_info *,
2633 : vec<int> * = nullptr);
2634 : extern void vect_describe_gather_scatter_call (stmt_vec_info,
2635 : gather_scatter_info *);
2636 : extern opt_result vect_find_stmt_data_reference (loop_p, gimple *,
2637 : vec<data_reference_p> *,
2638 : vec<int> *, int);
2639 : extern opt_result vect_analyze_data_refs (vec_info *, bool *);
2640 : extern void vect_record_base_alignments (vec_info *);
2641 : extern tree vect_create_data_ref_ptr (vec_info *,
2642 : stmt_vec_info, tree, class loop *, tree,
2643 : tree *, gimple_stmt_iterator *,
2644 : gimple **, bool,
2645 : tree = NULL_TREE);
2646 : extern tree bump_vector_ptr (vec_info *, tree, gimple_stmt_iterator *,
2647 : stmt_vec_info, tree);
2648 : extern void vect_copy_ref_info (tree, tree);
2649 : extern tree vect_create_destination_var (tree, tree);
2650 : extern bool vect_grouped_store_supported (tree, unsigned HOST_WIDE_INT);
2651 : extern internal_fn vect_store_lanes_supported (tree, unsigned HOST_WIDE_INT, bool);
2652 : extern bool vect_grouped_load_supported (tree, bool, unsigned HOST_WIDE_INT);
2653 : extern internal_fn vect_load_lanes_supported (tree, unsigned HOST_WIDE_INT,
2654 : bool, vec<int> * = nullptr);
2655 : extern tree vect_setup_realignment (vec_info *,
2656 : stmt_vec_info, tree, gimple_stmt_iterator *,
2657 : tree *, enum dr_alignment_support, tree,
2658 : class loop **);
2659 : extern tree vect_get_new_vect_var (tree, enum vect_var_kind, const char *);
2660 : extern tree vect_get_new_ssa_name (tree, enum vect_var_kind,
2661 : const char * = NULL);
2662 : extern tree vect_create_addr_base_for_vector_ref (vec_info *,
2663 : stmt_vec_info, gimple_seq *,
2664 : tree);
2665 :
2666 : /* In tree-vect-loop.cc. */
2667 : extern tree neutral_op_for_reduction (tree, code_helper, tree, bool = true);
2668 : extern widest_int vect_iv_limit_for_partial_vectors (loop_vec_info loop_vinfo);
2669 : bool vect_rgroup_iv_might_wrap_p (loop_vec_info, rgroup_controls *);
2670 : /* Used in gimple-loop-interchange.c and tree-parloops.cc. */
2671 : extern bool check_reduction_path (dump_user_location_t, loop_p, gphi *, tree,
2672 : enum tree_code);
2673 : extern bool needs_fold_left_reduction_p (tree, code_helper);
2674 : /* Drive for loop analysis stage. */
2675 : extern opt_loop_vec_info vect_analyze_loop (class loop *, gimple *,
2676 : vec_info_shared *);
2677 : extern tree vect_build_loop_niters (loop_vec_info, bool * = NULL);
2678 : extern void vect_gen_vector_loop_niters (loop_vec_info, tree, tree *,
2679 : tree *, bool);
2680 : extern tree vect_get_loop_iv_increment (loop_vec_info);
2681 : extern tree vect_halve_mask_nunits (tree, machine_mode);
2682 : extern tree vect_double_mask_nunits (tree, machine_mode);
2683 : extern void vect_record_loop_mask (loop_vec_info, vec_loop_masks *,
2684 : unsigned int, tree, tree);
2685 : extern tree vect_get_loop_mask (loop_vec_info, gimple_stmt_iterator *,
2686 : vec_loop_masks *,
2687 : unsigned int, tree, unsigned int);
2688 : extern void vect_record_loop_len (loop_vec_info, vec_loop_lens *, unsigned int,
2689 : tree, unsigned int);
2690 : extern tree vect_get_loop_len (loop_vec_info, gimple_stmt_iterator *,
2691 : vec_loop_lens *, unsigned int, tree,
2692 : unsigned int, unsigned int, bool);
2693 : extern tree vect_gen_loop_len_mask (loop_vec_info, gimple_stmt_iterator *,
2694 : gimple_stmt_iterator *, vec_loop_lens *,
2695 : unsigned int, tree, tree, unsigned int,
2696 : unsigned int);
2697 : extern gimple_seq vect_gen_len (tree, tree, tree, tree);
2698 : extern vect_reduc_info info_for_reduction (loop_vec_info, slp_tree);
2699 : extern bool reduction_fn_for_scalar_code (code_helper, internal_fn *);
2700 : extern unsigned vect_min_prec_for_max_niters (loop_vec_info, unsigned int);
2701 : /* Drive for loop transformation stage. */
2702 : extern class loop *vect_transform_loop (loop_vec_info, gimple *);
2703 929630 : struct vect_loop_form_info
2704 : {
2705 : tree number_of_iterations;
2706 : tree number_of_iterationsm1;
2707 : tree assumptions;
2708 : auto_vec<gcond *> conds;
2709 : gcond *inner_loop_cond;
2710 : edge loop_exit;
2711 : };
2712 : extern opt_result vect_analyze_loop_form (class loop *, gimple *,
2713 : vect_loop_form_info *);
2714 : extern loop_vec_info vect_create_loop_vinfo (class loop *, vec_info_shared *,
2715 : const vect_loop_form_info *,
2716 : loop_vec_info = nullptr);
2717 : extern bool vectorizable_live_operation (vec_info *, stmt_vec_info,
2718 : slp_tree, slp_instance, int,
2719 : bool, stmt_vector_for_cost *);
2720 : extern bool vectorizable_lane_reducing (loop_vec_info, stmt_vec_info,
2721 : slp_tree, stmt_vector_for_cost *);
2722 : extern bool vectorizable_reduction (loop_vec_info, stmt_vec_info,
2723 : slp_tree, slp_instance,
2724 : stmt_vector_for_cost *);
2725 : extern bool vectorizable_induction (loop_vec_info, stmt_vec_info,
2726 : slp_tree, stmt_vector_for_cost *);
2727 : extern bool vect_transform_reduction (loop_vec_info, stmt_vec_info,
2728 : gimple_stmt_iterator *,
2729 : slp_tree);
2730 : extern bool vect_transform_cycle_phi (loop_vec_info, stmt_vec_info,
2731 : slp_tree, slp_instance);
2732 : extern bool vectorizable_lc_phi (loop_vec_info, stmt_vec_info, slp_tree);
2733 : extern bool vect_transform_lc_phi (loop_vec_info, stmt_vec_info, slp_tree);
2734 : extern bool vectorizable_phi (bb_vec_info, stmt_vec_info, slp_tree,
2735 : stmt_vector_for_cost *);
2736 : extern bool vectorizable_recurr (loop_vec_info, stmt_vec_info,
2737 : slp_tree, stmt_vector_for_cost *);
2738 : extern bool vectorizable_early_exit (loop_vec_info, stmt_vec_info,
2739 : gimple_stmt_iterator *,
2740 : slp_tree, stmt_vector_for_cost *);
2741 : extern bool vect_emulated_vector_p (tree);
2742 : extern bool vect_can_vectorize_without_simd_p (tree_code);
2743 : extern bool vect_can_vectorize_without_simd_p (code_helper);
2744 : extern int vect_get_known_peeling_cost (loop_vec_info, int);
2745 : extern tree cse_and_gimplify_to_preheader (loop_vec_info, tree);
2746 :
2747 : /* Nonlinear induction. */
2748 : extern tree vect_peel_nonlinear_iv_init (gimple_seq*, tree, tree,
2749 : tree, enum vect_induction_op_type,
2750 : bool);
2751 :
2752 : /* In tree-vect-slp.cc. */
2753 : extern void vect_slp_init (void);
2754 : extern void vect_slp_fini (void);
2755 : extern void vect_free_slp_instance (slp_instance);
2756 : extern bool vect_transform_slp_perm_load (vec_info *, slp_tree, const vec<tree> &,
2757 : gimple_stmt_iterator *, poly_uint64,
2758 : bool, unsigned *,
2759 : unsigned * = nullptr, bool = false);
2760 : extern bool vectorizable_slp_permutation (vec_info *, gimple_stmt_iterator *,
2761 : slp_tree, stmt_vector_for_cost *);
2762 : extern bool vect_slp_analyze_operations (vec_info *);
2763 : extern bool vect_schedule_slp (vec_info *, vec<slp_instance> &, bool);
2764 : extern opt_result vect_analyze_slp (vec_info *, unsigned, bool);
2765 : extern bool vect_make_slp_decision (loop_vec_info);
2766 : extern bool vect_detect_hybrid_slp (loop_vec_info);
2767 : extern void vect_optimize_slp (vec_info *);
2768 : extern void vect_gather_slp_loads (vec_info *);
2769 : extern tree vect_get_slp_scalar_def (slp_tree, unsigned);
2770 : extern void vect_get_slp_defs (slp_tree, vec<tree> *);
2771 : extern void vect_get_slp_defs (vec_info *, slp_tree, vec<vec<tree> > *,
2772 : unsigned n = -1U);
2773 : extern bool vect_slp_if_converted_bb (basic_block bb, loop_p orig_loop);
2774 : extern bool vect_slp_function (function *);
2775 : extern stmt_vec_info vect_find_last_scalar_stmt_in_slp (slp_tree);
2776 : extern stmt_vec_info vect_find_first_scalar_stmt_in_slp (slp_tree);
2777 : extern bool is_simple_and_all_uses_invariant (stmt_vec_info, loop_vec_info);
2778 : extern bool can_duplicate_and_interleave_p (vec_info *, unsigned int, tree,
2779 : unsigned int * = NULL,
2780 : tree * = NULL, tree * = NULL);
2781 : extern void duplicate_and_interleave (vec_info *, gimple_seq *, tree,
2782 : const vec<tree> &, unsigned int, vec<tree> &);
2783 : extern int vect_get_place_in_interleaving_chain (stmt_vec_info, stmt_vec_info);
2784 : extern slp_tree vect_create_new_slp_node (unsigned, tree_code);
2785 : extern void vect_free_slp_tree (slp_tree);
2786 : extern bool compatible_calls_p (gcall *, gcall *, bool);
2787 : extern int vect_slp_child_index_for_operand (const stmt_vec_info, int op);
2788 :
2789 : extern tree prepare_vec_mask (loop_vec_info, tree, tree, tree,
2790 : gimple_stmt_iterator *);
2791 : extern tree vect_get_mask_load_else (int, tree);
2792 : extern bool vect_load_perm_consecutive_p (slp_tree, unsigned = UINT_MAX);
2793 :
2794 : /* In tree-vect-patterns.cc. */
2795 : extern void
2796 : vect_mark_pattern_stmts (vec_info *, stmt_vec_info, gimple *, tree);
2797 : extern bool vect_get_range_info (tree, wide_int*, wide_int*);
2798 :
2799 : /* Pattern recognition functions.
2800 : Additional pattern recognition functions can (and will) be added
2801 : in the future. */
2802 : void vect_pattern_recog (vec_info *);
2803 :
2804 : /* In tree-vectorizer.cc. */
2805 : unsigned vectorize_loops (void);
2806 : void vect_free_loop_info_assumptions (class loop *);
2807 : gimple *vect_loop_vectorized_call (class loop *, gcond **cond = NULL);
2808 : bool vect_stmt_dominates_stmt_p (gimple *, gimple *);
2809 :
2810 : /* SLP Pattern matcher types, tree-vect-slp-patterns.cc. */
2811 :
2812 : /* Forward declaration of possible two operands operation that can be matched
2813 : by the complex numbers pattern matchers. */
2814 : enum _complex_operation : unsigned;
2815 :
2816 : /* All possible load permute values that could result from the partial data-flow
2817 : analysis. */
2818 : typedef enum _complex_perm_kinds {
2819 : PERM_UNKNOWN,
2820 : PERM_EVENODD,
2821 : PERM_ODDEVEN,
2822 : PERM_ODDODD,
2823 : PERM_EVENEVEN,
2824 : /* Can be combined with any other PERM values. */
2825 : PERM_TOP
2826 : } complex_perm_kinds_t;
2827 :
2828 : /* Cache from nodes to the load permutation they represent. */
2829 : typedef hash_map <slp_tree, complex_perm_kinds_t>
2830 : slp_tree_to_load_perm_map_t;
2831 :
2832 : /* Cache from nodes pair to being compatible or not. */
2833 : typedef pair_hash <nofree_ptr_hash <_slp_tree>,
2834 : nofree_ptr_hash <_slp_tree>> slp_node_hash;
2835 : typedef hash_map <slp_node_hash, bool> slp_compat_nodes_map_t;
2836 :
2837 :
2838 : /* Vector pattern matcher base class. All SLP pattern matchers must inherit
2839 : from this type. */
2840 :
2841 : class vect_pattern
2842 : {
2843 : protected:
2844 : /* The number of arguments that the IFN requires. */
2845 : unsigned m_num_args;
2846 :
2847 : /* The internal function that will be used when a pattern is created. */
2848 : internal_fn m_ifn;
2849 :
2850 : /* The current node being inspected. */
2851 : slp_tree *m_node;
2852 :
2853 : /* The list of operands to be the children for the node produced when the
2854 : internal function is created. */
2855 : vec<slp_tree> m_ops;
2856 :
2857 : /* Default constructor where NODE is the root of the tree to inspect. */
2858 1114 : vect_pattern (slp_tree *node, vec<slp_tree> *m_ops, internal_fn ifn)
2859 32 : {
2860 1114 : this->m_ifn = ifn;
2861 1114 : this->m_node = node;
2862 1114 : this->m_ops.create (0);
2863 1114 : if (m_ops)
2864 32 : this->m_ops.safe_splice (*m_ops);
2865 : }
2866 :
2867 : public:
2868 :
2869 : /* Create a new instance of the pattern matcher class of the given type. */
2870 : static vect_pattern* recognize (slp_tree_to_load_perm_map_t *,
2871 : slp_compat_nodes_map_t *, slp_tree *);
2872 :
2873 : /* Build the pattern from the data collected so far. */
2874 : virtual void build (vec_info *) = 0;
2875 :
2876 : /* Default destructor. */
2877 : virtual ~vect_pattern ()
2878 : {
2879 : this->m_ops.release ();
2880 : }
2881 : };
2882 :
2883 : /* Function pointer to create a new pattern matcher from a generic type. */
2884 : typedef vect_pattern* (*vect_pattern_decl_t) (slp_tree_to_load_perm_map_t *,
2885 : slp_compat_nodes_map_t *,
2886 : slp_tree *);
2887 :
2888 : /* List of supported pattern matchers. */
2889 : extern vect_pattern_decl_t slp_patterns[];
2890 :
2891 : /* Number of supported pattern matchers. */
2892 : extern size_t num__slp_patterns;
2893 :
2894 : /* ----------------------------------------------------------------------
2895 : Target support routines
2896 : -----------------------------------------------------------------------
2897 : The following routines are provided to simplify costing decisions in
2898 : target code. Please add more as needed. */
2899 :
2900 : /* Return true if an operation of kind KIND for STMT_INFO represents
2901 : the extraction of an element from a vector in preparation for
2902 : storing the element to memory. */
2903 : inline bool
2904 : vect_is_store_elt_extraction (vect_cost_for_stmt kind, stmt_vec_info stmt_info)
2905 : {
2906 : return (kind == vec_to_scalar
2907 : && STMT_VINFO_DATA_REF (stmt_info)
2908 : && DR_IS_WRITE (STMT_VINFO_DATA_REF (stmt_info)));
2909 : }
2910 :
2911 : /* Return true if STMT_INFO represents part of a reduction. */
2912 : inline bool
2913 50582169 : vect_is_reduction (stmt_vec_info stmt_info)
2914 : {
2915 50582169 : return STMT_VINFO_REDUC_IDX (stmt_info) != -1;
2916 : }
2917 :
2918 : /* Return true if SLP_NODE represents part of a reduction. */
2919 : inline bool
2920 308242 : vect_is_reduction (slp_tree slp_node)
2921 : {
2922 308242 : return SLP_TREE_REDUC_IDX (slp_node) != -1;
2923 : }
2924 :
2925 : /* If STMT_INFO describes a reduction, return the vect_reduction_type
2926 : of the reduction it describes, otherwise return -1. */
2927 : inline int
2928 55752 : vect_reduc_type (vec_info *vinfo, slp_tree node)
2929 : {
2930 55752 : if (loop_vec_info loop_vinfo = dyn_cast<loop_vec_info> (vinfo))
2931 : {
2932 55752 : vect_reduc_info reduc_info = info_for_reduction (loop_vinfo, node);
2933 55752 : if (reduc_info)
2934 52832 : return int (VECT_REDUC_INFO_TYPE (reduc_info));
2935 : }
2936 : return -1;
2937 : }
2938 :
2939 : /* If STMT_INFO is a COND_EXPR that includes an embedded comparison, return the
2940 : scalar type of the values being compared. Return null otherwise. */
2941 : inline tree
2942 0 : vect_embedded_comparison_type (stmt_vec_info stmt_info)
2943 : {
2944 0 : if (auto *assign = dyn_cast<gassign *> (stmt_info->stmt))
2945 0 : if (gimple_assign_rhs_code (assign) == COND_EXPR)
2946 : {
2947 0 : tree cond = gimple_assign_rhs1 (assign);
2948 0 : if (COMPARISON_CLASS_P (cond))
2949 0 : return TREE_TYPE (TREE_OPERAND (cond, 0));
2950 : }
2951 : return NULL_TREE;
2952 : }
2953 :
2954 : /* If STMT_INFO is a comparison or contains an embedded comparison, return the
2955 : scalar type of the values being compared. Return null otherwise. */
2956 : inline tree
2957 67735 : vect_comparison_type (stmt_vec_info stmt_info)
2958 : {
2959 67735 : if (auto *assign = dyn_cast<gassign *> (stmt_info->stmt))
2960 67735 : if (TREE_CODE_CLASS (gimple_assign_rhs_code (assign)) == tcc_comparison)
2961 67735 : return TREE_TYPE (gimple_assign_rhs1 (assign));
2962 0 : return vect_embedded_comparison_type (stmt_info);
2963 : }
2964 :
2965 : /* Return true if STMT_INFO extends the result of a load. */
2966 : inline bool
2967 : vect_is_extending_load (class vec_info *vinfo, stmt_vec_info stmt_info)
2968 : {
2969 : /* Although this is quite large for an inline function, this part
2970 : at least should be inline. */
2971 : gassign *assign = dyn_cast <gassign *> (stmt_info->stmt);
2972 : if (!assign || !CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (assign)))
2973 : return false;
2974 :
2975 : tree rhs = gimple_assign_rhs1 (stmt_info->stmt);
2976 : tree lhs_type = TREE_TYPE (gimple_assign_lhs (assign));
2977 : tree rhs_type = TREE_TYPE (rhs);
2978 : if (!INTEGRAL_TYPE_P (lhs_type)
2979 : || !INTEGRAL_TYPE_P (rhs_type)
2980 : || TYPE_PRECISION (lhs_type) <= TYPE_PRECISION (rhs_type))
2981 : return false;
2982 :
2983 : stmt_vec_info def_stmt_info = vinfo->lookup_def (rhs);
2984 : return (def_stmt_info
2985 : && STMT_VINFO_DATA_REF (def_stmt_info)
2986 : && DR_IS_READ (STMT_VINFO_DATA_REF (def_stmt_info)));
2987 : }
2988 :
2989 : /* Return true if STMT_INFO truncates the input of a store. */
2990 : inline bool
2991 : vect_is_truncating_store (class vec_info *vinfo, stmt_vec_info stmt_info)
2992 : {
2993 : /* Although this is quite large for an inline function, this part
2994 : at least should be inline. */
2995 : gassign *assign = dyn_cast<gassign *> (stmt_info->stmt);
2996 : if (!assign || !CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (assign)))
2997 : return false;
2998 :
2999 : tree rhs = gimple_assign_rhs1 (stmt_info->stmt);
3000 : tree lhs = gimple_assign_lhs (assign);
3001 : tree lhs_type = TREE_TYPE (lhs);
3002 : tree rhs_type = TREE_TYPE (rhs);
3003 : if (!INTEGRAL_TYPE_P (lhs_type) || !INTEGRAL_TYPE_P (rhs_type)
3004 : || TYPE_PRECISION (lhs_type) >= TYPE_PRECISION (rhs_type))
3005 : return false;
3006 :
3007 : gimple *use_stmt;
3008 : use_operand_p use_p;
3009 : if (!single_imm_use (lhs, &use_p, &use_stmt))
3010 : return false;
3011 :
3012 : stmt_vec_info use_stmt_info = vinfo->lookup_stmt (use_stmt);
3013 : return (use_stmt_info && STMT_VINFO_DATA_REF (use_stmt_info)
3014 : && DR_IS_WRITE (STMT_VINFO_DATA_REF (use_stmt_info)));
3015 : }
3016 :
3017 : /* Return true if STMT_INFO is an integer truncation. */
3018 : inline bool
3019 : vect_is_integer_truncation (stmt_vec_info stmt_info)
3020 : {
3021 : gassign *assign = dyn_cast <gassign *> (stmt_info->stmt);
3022 : if (!assign || !CONVERT_EXPR_CODE_P (gimple_assign_rhs_code (assign)))
3023 : return false;
3024 :
3025 : tree lhs_type = TREE_TYPE (gimple_assign_lhs (assign));
3026 : tree rhs_type = TREE_TYPE (gimple_assign_rhs1 (assign));
3027 : return (INTEGRAL_TYPE_P (lhs_type)
3028 : && INTEGRAL_TYPE_P (rhs_type)
3029 : && TYPE_PRECISION (lhs_type) < TYPE_PRECISION (rhs_type));
3030 : }
3031 :
3032 : /* Build a GIMPLE_ASSIGN or GIMPLE_CALL with the tree_code,
3033 : or internal_fn contained in ch, respectively. */
3034 : gimple * vect_gimple_build (tree, code_helper, tree, tree = NULL_TREE);
3035 : #endif /* GCC_TREE_VECTORIZER_H */
|