@@ -1456,10 +1456,13 @@ gimple_fold_builtin_bzero (gimple_stmt_iterator *gsi)
}
/* Fold function call to builtin memset or bzero at *GSI setting the
- memory of size LEN to VAL. Return whether a simplification was made. */
+ memory of size LEN to VAL. Return whether a simplification was made.
+ If FOLD_ARBITRARY_N is true, also try folding supported constant
+ lengths greater than one through arbitrary pointer destinations. */
-static bool
-gimple_fold_builtin_memset (gimple_stmt_iterator *gsi, tree c, tree len)
+bool
+gimple_fold_builtin_memset (gimple_stmt_iterator *gsi, tree c, tree len,
+ bool fold_arbitrary_n)
{
gimple *stmt = gsi_stmt (*gsi);
tree etype;
@@ -1481,12 +1484,18 @@ gimple_fold_builtin_memset (gimple_stmt_iterator *gsi, tree c, tree len)
length = tree_to_uhwi (len);
tree dest = gimple_call_arg (stmt, 0);
- if (length == 1
- && POINTER_TYPE_P (TREE_TYPE (dest)))
+ if (POINTER_TYPE_P (TREE_TYPE (dest))
+ && (length == 1 || fold_arbitrary_n))
{
- /* Keep the original call until object-size analysis has inspected it. */
+ /* Keep the arbitrary-pointer fold until object-size analysis has
+ inspected the original call. Preserve the old early ADDR_EXPR fold
+ for multi-byte memsets. */
if (!(cfun->curr_properties & PROP_objsz))
- return false;
+ {
+ if (length == 1)
+ return false;
+ goto normal_memset;
+ }
/* Detect out-of-bounds accesses without issuing warnings.
Avoid folding out-of-bounds accesses but to avoid false
@@ -1499,19 +1508,89 @@ gimple_fold_builtin_memset (gimple_stmt_iterator *gsi, tree c, tree len)
if (warning != OPT_Wrestrict)
return false;
- etype = unsigned_char_type_node;
+ unsigned int dest_align = get_pointer_alignment (dest);
+ machine_mode mode = VOIDmode;
+
+ if (length == 1)
+ etype = unsigned_char_type_node;
+ else
+ {
+ /* Use one scalar store only when it is no larger than the target's
+ normal move size and its byte-replication constant fits in a host
+ integer. */
+ if (length > MOVE_MAX
+ || length > HOST_BITS_PER_WIDE_INT / BITS_PER_UNIT)
+ goto normal_memset;
+
+ unsigned HOST_WIDE_INT bits = length * BITS_PER_UNIT;
+ scalar_int_mode imode;
+ if (!int_mode_for_size (bits, 0).exists (&imode)
+ || !bitwise_mode_for_size (bits).exists (&mode)
+ || !known_eq (GET_MODE_BITSIZE (mode), bits)
+ /* If DEST is not naturally aligned, require an efficient
+ unaligned store or a movmisalign pattern. */
+ || (dest_align < GET_MODE_ALIGNMENT (mode)
+ && targetm.slow_unaligned_access (mode, dest_align)
+ && (optab_handler (movmisalign_optab, mode)
+ == CODE_FOR_nothing)))
+ goto normal_memset;
+
+ etype = bitwise_type_for_mode (mode);
+ if (!etype || !INTEGRAL_TYPE_P (etype))
+ goto normal_memset;
+
+ if (dest_align < GET_MODE_ALIGNMENT (mode))
+ etype = build_aligned_type (etype, dest_align);
+ }
+
+ tree value_type = TYPE_MAIN_VARIANT (etype);
tree ptype = TREE_TYPE (TREE_TYPE (dest));
if (TYPE_VOLATILE (ptype))
etype = build_qualified_type (etype, TYPE_QUAL_VOLATILE);
location_t loc = gimple_location (stmt);
tree cval_tree;
- if (TREE_CODE (c) == INTEGER_CST)
- cval_tree = fold_convert (etype, c);
+ if (length == 1)
+ {
+ if (TREE_CODE (c) == INTEGER_CST)
+ cval_tree = fold_convert (etype, c);
+ else
+ cval_tree = gimple_convert (gsi, true, GSI_SAME_STMT, loc,
+ etype, c);
+ }
else
- cval_tree = gimple_convert (gsi, true, GSI_SAME_STMT, loc, etype, c);
+ {
+ tree byte;
+ if (TREE_CODE (c) == INTEGER_CST)
+ byte = fold_convert (unsigned_char_type_node, c);
+ else
+ byte = gimple_convert (gsi, true, GSI_SAME_STMT, loc,
+ unsigned_char_type_node, c);
- /* Build accesses at offset zero with a ref-all character type. */
+ tree byte_value;
+ if (TREE_CODE (byte) == INTEGER_CST)
+ byte_value = fold_convert (value_type, byte);
+ else
+ byte_value = gimple_convert (gsi, true, GSI_SAME_STMT, loc,
+ value_type, byte);
+
+ unsigned HOST_WIDE_INT multiplier = 1;
+ for (unsigned HOST_WIDE_INT i = 1; i < length; ++i)
+ multiplier = (multiplier << BITS_PER_UNIT) | 1;
+ tree multiplier_tree = build_int_cst_type (value_type, multiplier);
+ if (TREE_CODE (byte_value) == INTEGER_CST)
+ cval_tree = fold_build2_loc (loc, MULT_EXPR, value_type,
+ byte_value, multiplier_tree);
+ else
+ cval_tree = gimple_build (gsi, true, GSI_SAME_STMT, loc,
+ MULT_EXPR, value_type, byte_value,
+ multiplier_tree);
+ if (!useless_type_conversion_p (etype, TREE_TYPE (cval_tree)))
+ cval_tree = gimple_convert (gsi, true, GSI_SAME_STMT, loc,
+ etype, cval_tree);
+ }
+
+ /* Build the store at offset zero with the selected scalar type. */
tree off0
= build_int_cst (build_pointer_type_for_mode (char_type_node,
ptr_mode, true), 0);
@@ -1539,6 +1618,8 @@ gimple_fold_builtin_memset (gimple_stmt_iterator *gsi, tree c, tree len)
return true;
}
+normal_memset:
+
if (TREE_CODE (c) != INTEGER_CST)
return false;
@@ -5527,7 +5608,8 @@ gimple_fold_builtin (gimple_stmt_iterator *gsi)
case BUILT_IN_MEMSET:
return gimple_fold_builtin_memset (gsi,
gimple_call_arg (stmt, 1),
- gimple_call_arg (stmt, 2));
+ gimple_call_arg (stmt, 2),
+ false);
case BUILT_IN_MEMPCPY:
if (gimple_fold_builtin_memory_op (gsi, gimple_call_arg (stmt, 0),
gimple_call_arg (stmt, 1), fcode))
@@ -58,6 +58,8 @@ extern tree gimple_get_virt_method_for_vtable (HOST_WIDE_INT, tree,
unsigned HOST_WIDE_INT,
bool *can_refer = NULL);
extern tree gimple_fold_indirect_ref (tree);
+extern bool gimple_fold_builtin_memset (gimple_stmt_iterator *, tree, tree,
+ bool = false);
extern bool gimple_fold_builtin_sprintf (gimple_stmt_iterator *);
extern bool gimple_fold_builtin_snprintf (gimple_stmt_iterator *);
extern bool arith_code_with_undefined_signed_overflow (tree_code);
new file mode 100644
@@ -0,0 +1,35 @@
+/* PR tree-optimization/102202 */
+/* { dg-do compile } */
+/* { dg-require-effective-target int32plus } */
+/* { dg-options "-O2 -fdump-tree-cdce-details -fdump-tree-optimized" } */
+
+/* CDCE turns the exact range {0, 2} into a zero-length bypass and a call
+ with constant length two. The latter should then fold to one scalar
+ store even though the destination is an arbitrary pointer. */
+
+void
+g1 (unsigned int n, int c, unsigned short *d)
+{
+ __SIZE_TYPE__ len = (n & 1) ? 2 : 0;
+ __builtin_memset (d, c, len);
+}
+
+void *
+g2 (unsigned int n, int c, unsigned short *d)
+{
+ __SIZE_TYPE__ len = (n & 1) ? 2 : 0;
+ return __builtin_memset (d, c, len);
+}
+
+void
+g3 (unsigned int n, unsigned short *d)
+{
+ __SIZE_TYPE__ len = (n & 1) ? 2 : 0;
+ __builtin_memset (d, 7, len);
+}
+
+/* The three calls have exact {0, 2} lengths. */
+/* { dg-final { scan-tree-dump-times "function call is shrink-wrapped into error conditions" 3 "cdce" } } */
+
+/* All three nonzero paths are scalarized. */
+/* { dg-final { scan-tree-dump-not "__builtin_memset" "optimized" } } */
@@ -1369,6 +1369,9 @@ shrink_wrap_len_call (gcall *call, unsigned len_arg, tree zero_len_result,
update_stmt (call);
gimple_stmt_iterator gsi = gsi_for_stmt (call);
fold_stmt (&gsi);
+ if (gsi_stmt (gsi) == call)
+ gimple_fold_builtin_memset (&gsi, gimple_call_arg (call, 1),
+ nonzero_len, true);
}
/* The top level function for conditional dead code shrink