From: Reshma Roy <[email protected]>

 Transform uniform gather operations into optimized scalar load + broadcast
 instead of emulated gather with N individual scalar loads.

 BEFORE (Emulated Gather - 8 loads for VF=8):
     for each lane i in 0..7:
         offset[i] = BIT_FIELD_REF<offset_vector, i>
         addr[i] = base + offset[i] * scale
         result[i] = MEM[addr[i]]
       result_vector = {result[0], ..., result[7]}

       AFTER (Broadcast - 1 load):
           offset = BIT_FIELD_REF<offset_vector, 0>
             addr = base + offset * scale
             scalar_val = MEM[addr]
             result_vector = VEC_DUPLICATE_EXPR(scalar_val)

v2 changes:
- Keep the full loop (k = 0 ??? const_nunits-1) but only
perform the scalar load once and reuse elt for all lanes
- Since VEC_DUPLICATE_EXPR is for VLA/scalable vector types only
removed that and added constructor
- Removed the SSA copy by refactoring the code

gcc/ChangeLog:

        * tree-vect-stmts.cc (vectorizable_load): Generate broadcast for 
        uniform gather offsets.

gcc/testsuite/ChangeLog:

        * g++.dg/vect/vect-uniform-broadcast-neg.c: New test.
        * g++.dg/vect/vect-uniform-broadcast-op2-var.c: New test.
        * g++.dg/vect/vect-uniform-broadcast.c: New test.

---

Hi Richard,

The comments were addressed and the updated patch is attached.

Thanks,
Reshma Roy

Original Message-----
> From: Richard Biener <[email protected]>
> Sent: Tuesday, June 9, 2026 7:31 PM
> To: Roy, Reshma <[email protected]>
> Cc: [email protected]; Kumar, Venkataramanan
> <[email protected]>
> Subject: Re: [PATCH 2/2] Loop Vectorizer: Generate broadcast for uniform 
> gather
> offsets
> 
> Caution: This message originated from an External Source. Use proper caution
> when opening attachments, clicking links, or responding.
> 
> 
> On Mon, 25 May 2026, [email protected] wrote:
> 
> > From: Reshma Roy <[email protected]>
> >
> >  Transform uniform gather operations into optimized scalar load +
> > broadcast  instead of emulated gather with N individual scalar loads.
> >
> >  BEFORE (Emulated Gather - 8 loads for VF=8):
> >       for each lane i in 0..7:
> >               offset[i] = BIT_FIELD_REF<offset_vector, i>
> >               addr[i] = base + offset[i] * scale
> >               result[i] = MEM[addr[i]]
> >               result_vector = {result[0], ..., result[7]}
> >
> >  AFTER (Broadcast - 1 load):
> >       offset = BIT_FIELD_REF<offset_vector, 0>
> >       addr = base + offset * scale
> >       scalar_val = MEM[addr]
> >       result_vector = VEC_DUPLICATE_EXPR(scalar_val)
> >
> > gcc/ChangeLog:
> >
> >       * tree-vect-stmts.cc (vectorizable_load): Generate broadcast for
> >       uniform gather offsets.
> >
> > gcc/testsuite/ChangeLog:
> >       * g++.dg/vect/vect-uniform-broadcast-neg.c: New test.
> >       * g++.dg/vect/vect-uniform-broadcast-op2-var.c: New test.
> >       * g++.dg/vect/vect-uniform-broadcast.c: New test.
> >
> > ---
> >  .../g++.dg/vect/vect-uniform-broadcast-neg.c  | 29 +++++++++
> >  .../vect/vect-uniform-broadcast-op2-var.c     | 38 ++++++++++++
> >  .../g++.dg/vect/vect-uniform-broadcast.c      | 29 +++++++++
> >  gcc/tree-vect-stmts.cc                        | 59 ++++++++++++++++---
> >  4 files changed, 148 insertions(+), 7 deletions(-)  create mode
> > 100644 gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
> >  create mode 100644
> > gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
> >  create mode 100644 gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
> >
> > diff --git a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
> > b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
> > new file mode 100644
> > index 00000000000..590706d3b51
> > --- /dev/null
> > +++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
> > @@ -0,0 +1,29 @@
> > +/* { dg-do compile } */
> > +/* { dg-additional-options "-O3 -std=c++11 -march=znver5
> > +-fdump-tree-vect-details" } */
> > +/* Negative test: loop i starts at 1 -> SCEV {1, +, 1}.  POC only handles
> > +   {0, +, 1}, so offset is not treated as uniform -> no broadcast
> > +   (vec_duplicate_expr must not appear for this gather).  */ #include
> > +<bitset> #include <vector> class Foo {
> > +  public:
> > +    void fun (unsigned boardsize, bool check);
> > +    std::vector<int> m_mcowner;
> > +};
> > +void Foo::fun (unsigned boardsize, bool check) {
> > +  std::bitset<21*21> blacksq;
> > +  __asm__ ("" : "+g" (blacksq));
> > +  for (int i = 0; i < boardsize; i++) {
> > +    if (i % 2 == 0)
> > +      blacksq[i] = true;
> > +  }
> > +  unsigned int i = 1;
> > +  do {
> > +    if (blacksq[i])
> > +      m_mcowner[i]++;
> > +  } while (++i < blacksq.size ());
> > +}
> > +/* { dg-final { scan-tree-dump-not "\.vec_duplicate_expr" "vect" } }
> > +*/
> > +
> > diff --git
> > a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
> > b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
> > new file mode 100644
> > index 00000000000..b212b3d56be
> > --- /dev/null
> > +++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
> > @@ -0,0 +1,38 @@
> > +/* { dg-do compile } */
> > +/* { dg-additional-options "-O3 -std=c++11 -march=znver5
> > +-fdump-tree-vect-details" } */
> > +/* Gather load with PLUS (hi, lo). The second operand of the offset PLUS 
> > is a
> > +   variable, hence does not emit vec_duplicate_expr. The inner loop still
> > +   vectorizes as a normal gather */
> > +
> > +#include <vector>
> > +
> > +class Foo
> > +{
> > +  public:
> > +    void fun (unsigned boardsize);
> > +    std::vector<int> m_mcowner;
> > +};
> > +
> > +void
> > +Foo::fun (unsigned boardsize)
> > +{
> > +  unsigned nwords = (21 * 21 + 63) / 64 + 2;
> > +  std::vector<unsigned long> words (nwords, 0);
> > +  __asm__ ("" : "+g" (words));
> > +  for (unsigned i = 0; i < boardsize; i++)
> > +    if (i % 2 == 0)
> > +      words[i / 64] |= 1UL << (i % 64);
> > +  unsigned int i = 0;
> > +  do
> > +    {
> > +      unsigned hi = i / 64;
> > +      unsigned lo = i & 1u;
> > +      /* the second operand is ensured to be loop-varying by the SCEV path 
> > */
> > +      if (words[hi + lo])
> > +     m_mcowner[i]++;
> > +    }
> > +  while (++i < 21 * 21);
> > +}
> > +
> > +/* { dg-final { scan-tree-dump-not "\.vec_duplicate_expr" "vect" } }
> > +*/
> > +/* { dg-final { scan-tree-dump "gs_offset_uniform_p is set to: 0"
> > +"vect" } } */
> > diff --git a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
> > b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
> > new file mode 100644
> > index 00000000000..818b7a63c00
> > --- /dev/null
> > +++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
> > @@ -0,0 +1,29 @@
> > +/* { dg-do compile } */
> > +/* { dg-additional-options "-O3 -std=c++11 -march=znver5
> > +-fdump-tree-vect-details" } */
> > +/* Test case for generating uniform broadcast instead of gather
> > +   when there is uniform value accross the vector lane */ #include
> > +<bitset> #include <vector> class Foo {
> > +  public:
> > +    void fun (unsigned boardsize, bool check);
> > +    std::vector<int> m_mcowner;
> > +};
> > +void Foo::fun( unsigned boardsize, bool check) {
> > +  std::bitset<21*21> blacksq;
> > +  __asm__ ("" : "+g" (blacksq));
> > +  for (int i = 0; i < boardsize; i++) {
> > +    if(i%2 == 0)
> > +      blacksq[i] = true;
> > +  }
> > +  unsigned int i = 0;
> > +  do{
> > +    if (blacksq[i])
> > +      m_mcowner[i]++;
> > +  }while(++i < blacksq.size());
> > +
> > +}
> > +/* { dg-final { scan-tree-dump "\.vec_duplicate_expr" "vect" } } */
> > +
> > diff --git a/gcc/tree-vect-stmts.cc b/gcc/tree-vect-stmts.cc index
> > b68fc5072af..71ff138c069 100644
> > --- a/gcc/tree-vect-stmts.cc
> > +++ b/gcc/tree-vect-stmts.cc
> > @@ -11198,6 +11198,25 @@ vectorizable_load (vec_info *vinfo,
> >             unsigned HOST_WIDE_INT const_nunits = nunits.to_constant ();
> >             if (costing_p)
> >               {
> > +               if (slp_node->gs_offset_uniform_p)
> > +                 {
> > +                   if (dump_enabled_p ())
> > +                     dump_printf_loc (MSG_NOTE, vect_location,
> > +                                      "computing cost for broadcast \n");
> > +                   /* Broadcast optimization: 1 extract + 1 load
> > +                      + 1 broadcast.  */
> > +                   inside_cost = record_stmt_cost (cost_vec, 1,
> > +                                                   vec_to_scalar, slp_node,
> > +                                                   0, vect_body);
> > +                   inside_cost += record_stmt_cost (cost_vec, 1, 
> > scalar_load,
> > +                                                    slp_node, 0, 
> > vect_body);
> > +                   inside_cost += record_stmt_cost (cost_vec, 1,
> > +                                                    scalar_to_vec, 
> > slp_node,
> > +                                                    0, vect_body);
> > +                 }
> > +               else
> > +                 {
> > +
> >                 /* For emulated gathers N offset vector element
> >                    offset add is consumed by the load).  */
> >                 inside_cost = record_stmt_cost (cost_vec,
> > const_nunits, @@ -11211,6 +11230,7 @@ vectorizable_load (vec_info *vinfo,
> >                 inside_cost
> >                   = record_stmt_cost (cost_vec, 1, vec_construct,
> >                                       slp_node, 0, vect_body);
> > +                 }
> >                 continue;
> >               }
> >             tree offset_vectype = TREE_TYPE (vec_offsets[0]); @@
> > -11228,7 +11248,12 @@ vectorizable_load (vec_info *vinfo,
> >             tree idx_type = TREE_TYPE (TREE_TYPE (vec_offset));
> >             tree scale = size_int (SLP_TREE_GS_SCALE (slp_node));
> >             tree ltype = build_aligned_type (TREE_TYPE (vectype), align);
> > -           for (unsigned k = 0; k < const_nunits; ++k)
> > +           tree broadcast_vec = NULL_TREE;
> > +           /* If all lanes are uniform then generate broadcast otherwise
> > +              emulated gathers.  */
> > +           unsigned HOST_WIDE_INT vec_lane_loads
> > +             = slp_node->gs_offset_uniform_p ? 1 : const_nunits;
> 
> instead of this
> 
> > +           for (unsigned k = 0; k < vec_lane_loads; ++k)
> >               {
> >                 tree boff = size_binop (MULT_EXPR, TYPE_SIZE (idx_type),
> >                                         bitsize_int (k + elt_offset));
> > @@ -11247,13 +11272,33 @@ vectorizable_load (vec_info *vinfo,
> >                 new_stmt = gimple_build_assign (elt, ref);
> >                 gimple_set_vuse (new_stmt, gimple_vuse (gsi_stmt (*gsi)));
> >                 gimple_seq_add_stmt (&stmts, new_stmt);
> > -               CONSTRUCTOR_APPEND_ELT (ctor_elts, NULL_TREE, elt);
> > +               if (!slp_node->gs_offset_uniform_p)
> > +                     CONSTRUCTOR_APPEND_ELT (ctor_elts, NULL_TREE, elt);
> > +               else
> > +                 /* Broadcast to all lanes.  */
> > +                 broadcast_vec = gimple_build (&stmts,
> > +                                               VEC_DUPLICATE_EXPR,
> > +                                               vectype, elt);
> 
> break from the loop here.  Also use gimple_build_vector_from_val, a
> VEC_DUPLICATE_EXPR is only valid for VLA vectors, a broadcast for non-VLA
> vectors should use a uniform CONSTRUCTOR.
Done.
> 
> > +             }
> > +           if (!slp_node->gs_offset_uniform_p)
> > +             {
> > +               gsi_insert_seq_before (gsi, stmts, GSI_SAME_STMT);
> > +               new_stmt = gimple_build_assign (NULL_TREE,
> > +                                               build_constructor (vectype,
> > +                                                                  
> > ctor_elts)
> > +                                               );
> > +               data_ref = NULL_TREE;
> > +             }
> > +           else
> > +             {
> > +               gsi_insert_seq_before (gsi, stmts, GSI_SAME_STMT);
> > +               new_stmt = gimple_build_assign (NULL_TREE,
> > + broadcast_vec);
> 
> you get a SSA copy from this, some refactoring should avoid that.  Or simply 
> build
> that uniform vector above by keeping the loop but only load 'elt' once (but 
> add it
> nunits time).
Refactored the code such that SSA copy is removed.
> 
> > +               data_ref = NULL_TREE;
> > +               if (dump_enabled_p ())
> > +                 dump_printf_loc (MSG_OPTIMIZED_LOCATIONS, vect_location,
> > +                                  "generating broadcast instead of gather "
> > +                                  "loads (uniform offset lanes)\n");
> >               }
> > -           gsi_insert_seq_before (gsi, stmts, GSI_SAME_STMT);
> > -           new_stmt = gimple_build_assign (NULL_TREE,
> > -                                           build_constructor (vectype,
> > -                                                              ctor_elts));
> > -           data_ref = NULL_TREE;
> >           }
> >
> >         vec_dest = vect_create_destination_var (scalar_dest, vectype);
> >
> 
> --
> Richard Biener <[email protected]>
> SUSE Software Solutions Germany GmbH,
> Frankenstrasse 146, 90461 Nuernberg, Germany;
> GF: Jochen Jaser, Andrew McDonald, Werner Knoblich; (HRB 36809, AG
> Nuernberg)


 .../g++.dg/vect/vect-uniform-broadcast-neg.c  | 29 ++++++++
 .../vect/vect-uniform-broadcast-op2-var.c     | 38 ++++++++++
 .../g++.dg/vect/vect-uniform-broadcast.c      | 29 ++++++++
 gcc/tree-vect-stmts.cc                        | 71 ++++++++++++++-----
 4 files changed, 149 insertions(+), 18 deletions(-)
 create mode 100644 gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
 create mode 100644 gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
 create mode 100644 gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c

diff --git a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c 
b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
new file mode 100644
index 00000000000..590706d3b51
--- /dev/null
+++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-neg.c
@@ -0,0 +1,29 @@
+/* { dg-do compile } */
+/* { dg-additional-options "-O3 -std=c++11 -march=znver5 
-fdump-tree-vect-details" } */
+/* Negative test: loop i starts at 1 -> SCEV {1, +, 1}.  POC only handles
+   {0, +, 1}, so offset is not treated as uniform -> no broadcast
+   (vec_duplicate_expr must not appear for this gather).  */
+#include <bitset>
+#include <vector>
+class Foo
+{
+  public:
+    void fun (unsigned boardsize, bool check);
+    std::vector<int> m_mcowner;
+};
+void Foo::fun (unsigned boardsize, bool check)
+{
+  std::bitset<21*21> blacksq;
+  __asm__ ("" : "+g" (blacksq));
+  for (int i = 0; i < boardsize; i++) {
+    if (i % 2 == 0)
+      blacksq[i] = true;
+  }
+  unsigned int i = 1;
+  do {
+    if (blacksq[i])
+      m_mcowner[i]++;
+  } while (++i < blacksq.size ());
+}
+/* { dg-final { scan-tree-dump-not "\.vec_duplicate_expr" "vect" } } */
+
diff --git a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c 
b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
new file mode 100644
index 00000000000..b212b3d56be
--- /dev/null
+++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast-op2-var.c
@@ -0,0 +1,38 @@
+/* { dg-do compile } */
+/* { dg-additional-options "-O3 -std=c++11 -march=znver5 
-fdump-tree-vect-details" } */
+/* Gather load with PLUS (hi, lo). The second operand of the offset PLUS is a 
+   variable, hence does not emit vec_duplicate_expr. The inner loop still 
+   vectorizes as a normal gather */
+
+#include <vector>
+
+class Foo
+{
+  public:
+    void fun (unsigned boardsize);
+    std::vector<int> m_mcowner;
+};
+
+void
+Foo::fun (unsigned boardsize)
+{
+  unsigned nwords = (21 * 21 + 63) / 64 + 2;
+  std::vector<unsigned long> words (nwords, 0);
+  __asm__ ("" : "+g" (words));
+  for (unsigned i = 0; i < boardsize; i++)
+    if (i % 2 == 0)
+      words[i / 64] |= 1UL << (i % 64);
+  unsigned int i = 0;
+  do
+    {
+      unsigned hi = i / 64;
+      unsigned lo = i & 1u;
+      /* the second operand is ensured to be loop-varying by the SCEV path */
+      if (words[hi + lo])
+       m_mcowner[i]++;
+    }
+  while (++i < 21 * 21);
+}
+
+/* { dg-final { scan-tree-dump-not "\.vec_duplicate_expr" "vect" } } */
+/* { dg-final { scan-tree-dump "gs_offset_uniform_p is set to: 0" "vect" } } */
diff --git a/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c 
b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
new file mode 100644
index 00000000000..818b7a63c00
--- /dev/null
+++ b/gcc/testsuite/g++.dg/vect/vect-uniform-broadcast.c
@@ -0,0 +1,29 @@
+/* { dg-do compile } */
+/* { dg-additional-options "-O3 -std=c++11 -march=znver5 
-fdump-tree-vect-details" } */
+/* Test case for generating uniform broadcast instead of gather
+   when there is uniform value accross the vector lane */
+#include <bitset>
+#include <vector>
+class Foo 
+{
+  public:
+    void fun (unsigned boardsize, bool check);
+    std::vector<int> m_mcowner;
+};
+void Foo::fun( unsigned boardsize, bool check)
+{
+  std::bitset<21*21> blacksq;
+  __asm__ ("" : "+g" (blacksq)); 
+  for (int i = 0; i < boardsize; i++) {
+    if(i%2 == 0)
+      blacksq[i] = true;
+  }
+  unsigned int i = 0;
+  do{
+    if (blacksq[i])
+      m_mcowner[i]++;
+  }while(++i < blacksq.size());
+
+}
+/* { dg-final { scan-tree-dump "\.vec_duplicate_expr" "vect" } } */
+
diff --git a/gcc/tree-vect-stmts.cc b/gcc/tree-vect-stmts.cc
index afd5109c9f7..44282f20873 100644
--- a/gcc/tree-vect-stmts.cc
+++ b/gcc/tree-vect-stmts.cc
@@ -11207,6 +11207,25 @@ vectorizable_load (vec_info *vinfo,
              unsigned HOST_WIDE_INT const_nunits = nunits.to_constant ();
              if (costing_p)
                {
+                 if (slp_node->gs_offset_uniform_p)
+                   {
+                     if (dump_enabled_p ())
+                       dump_printf_loc (MSG_NOTE, vect_location,
+                                        "computing cost for broadcast \n");
+                     /* Broadcast optimization: 1 extract + 1 load
+                        + 1 broadcast.  */
+                     inside_cost = record_stmt_cost (cost_vec, 1,
+                                                     vec_to_scalar, slp_node,
+                                                     0, vect_body);
+                     inside_cost += record_stmt_cost (cost_vec, 1, scalar_load,
+                                                      slp_node, 0, vect_body);
+                     inside_cost += record_stmt_cost (cost_vec, 1,
+                                                      scalar_to_vec, slp_node,
+                                                      0, vect_body);
+                   }
+                 else
+                   {
+
                  /* For emulated gathers N offset vector element
                     offset add is consumed by the load).  */
                  inside_cost = record_stmt_cost (cost_vec, 1, vec_deconstruct,
@@ -11219,6 +11238,7 @@ vectorizable_load (vec_info *vinfo,
                  inside_cost
                    = record_stmt_cost (cost_vec, 1, vec_construct,
                                        slp_node, 0, vect_body);
+                   }
                  continue;
                }
              tree offset_vectype = TREE_TYPE (vec_offsets[0]);
@@ -11236,25 +11256,37 @@ vectorizable_load (vec_info *vinfo,
              tree idx_type = TREE_TYPE (TREE_TYPE (vec_offset));
              tree scale = size_int (SLP_TREE_GS_SCALE (slp_node));
              tree ltype = build_aligned_type (TREE_TYPE (vectype), align);
+             tree first_elt = NULL_TREE;
              for (unsigned k = 0; k < const_nunits; ++k)
                {
-                 tree boff = size_binop (MULT_EXPR, TYPE_SIZE (idx_type),
-                                         bitsize_int (k + elt_offset));
-                 tree idx = gimple_build (&stmts, BIT_FIELD_REF, idx_type,
-                                          vec_offset, TYPE_SIZE (idx_type),
-                                          boff);
-                 idx = gimple_convert (&stmts, sizetype, idx);
-                 idx = gimple_build (&stmts, MULT_EXPR, sizetype, idx, scale);
-                 tree ptr = gimple_build (&stmts, PLUS_EXPR,
-                                          TREE_TYPE (dataref_ptr),
-                                          dataref_ptr, idx);
-                 ptr = gimple_convert (&stmts, ptr_type_node, ptr);
-                 tree elt = make_ssa_name (TREE_TYPE (vectype));
-                 tree ref = build2 (MEM_REF, ltype, ptr,
-                                    build_int_cst (ref_type, 0));
-                 new_stmt = gimple_build_assign (elt, ref);
-                 gimple_set_vuse (new_stmt, gimple_vuse (gsi_stmt (*gsi)));
-                 gimple_seq_add_stmt (&stmts, new_stmt);
+                 tree elt;
+                 /* If all lanes are uniform then generate broadcast otherwise
+                emulated gathers.  */
+                 if (slp_node->gs_offset_uniform_p && k>0)
+                   elt = first_elt;
+                 else
+                   {
+
+                     tree boff = size_binop (MULT_EXPR, TYPE_SIZE (idx_type),
+                                             bitsize_int (k + elt_offset));
+                     tree idx = gimple_build (&stmts, BIT_FIELD_REF, idx_type,
+                                              vec_offset, TYPE_SIZE (idx_type),
+                                              boff);
+                     idx = gimple_convert (&stmts, sizetype, idx);
+                     idx = gimple_build (&stmts, MULT_EXPR, sizetype, idx,
+                                         scale);
+                     tree ptr = gimple_build (&stmts, PLUS_EXPR,
+                                              TREE_TYPE (dataref_ptr),
+                                              dataref_ptr, idx);
+                     ptr = gimple_convert (&stmts, ptr_type_node, ptr);
+                     elt = make_ssa_name (TREE_TYPE (vectype));
+                     tree ref = build2 (MEM_REF, ltype, ptr,
+                                        build_int_cst (ref_type, 0));
+                     new_stmt = gimple_build_assign (elt, ref);
+                     gimple_set_vuse (new_stmt, gimple_vuse (gsi_stmt (*gsi)));
+                     gimple_seq_add_stmt (&stmts, new_stmt);
+                     first_elt = elt;
+                   }
                  CONSTRUCTOR_APPEND_ELT (ctor_elts, NULL_TREE, elt);
                }
              gsi_insert_seq_before (gsi, stmts, GSI_SAME_STMT);
@@ -11262,8 +11294,11 @@ vectorizable_load (vec_info *vinfo,
                                              build_constructor (vectype,
                                                                 ctor_elts));
              data_ref = NULL_TREE;
+             if (slp_node->gs_offset_uniform_p && dump_enabled_p ())
+               dump_printf_loc (MSG_OPTIMIZED_LOCATIONS, vect_location,
+                                "generating broadcast instead of gather "
+                                "loads (uniform offset lanes)\n");
            }
-
          vec_dest = vect_create_destination_var (scalar_dest, vectype);
          /* DATA_REF is null if we've already built the statement.  */
          if (data_ref)
-- 
2.34.1

Reply via email to