tcg/optimize: Lower unsupported deposit during optimize
The expansions that we chose in tcg-op.c may be less than optimal. Delay lowering until optimize, so that we have propagated constants and have computed known zero/one masks. Reviewed-by: Jim MacArthur <jim.macarthur@linaro.org> Signed-off-by: Richard Henderson <richard.henderson@linaro.org> Message-ID: <20260303010833.1115741-3-richard.henderson@linaro.org> Signed-off-by: Philippe Mathieu-Daudé <philmd@linaro.org>
Richard Henderson committed
Oct 24, 2023 at 00:31 UTC
5f747705a45e808f159b40cd1ca59ef7c6c0a249
2 files changed
+163
-99
tcg/optimize.c
+158
-21
@@ -1652,12 +1652,17 @@ static bool fold_ctpop(OptContext *ctx, TCGOp *op)
1652
1653
static bool fold_deposit(OptContext *ctx, TCGOp *op)
1654
{
1655
- TempOptInfo *t1 = arg_info(op->args[1]);
1656
- TempOptInfo *t2 = arg_info(op->args[2]);
1655
+ TCGArg ret = op->args[0];
1656
+ TCGArg arg1 = op->args[1];
1657
+ TCGArg arg2 = op->args[2];
1658
int ofs = op->args[3];
1659
int len = op->args[4];
1659
- int width = 8 * tcg_type_size(ctx->type);
1660
- uint64_t z_mask, o_mask, s_mask;
1660
+ TempOptInfo *t1 = arg_info(arg1);
1661
+ TempOptInfo *t2 = arg_info(arg2);
1662
+ int width;
1663
+ uint64_t z_mask, o_mask, s_mask, type_mask, len_mask;
1664
+ TCGOp *op2;
1665
+ bool valid;
1666
1667
if (ti_is_const(t1) && ti_is_const(t2)) {
1668
return tcg_opt_gen_movi(ctx, op, op->args[0],
@@ -1665,35 +1670,167 @@ static bool fold_deposit(OptContext *ctx, TCGOp *op)
1670
ti_const_val(t2)));
1671
}
1672
1668
- /* Inserting a value into zero at offset 0. */
1669
- if (ti_is_const_val(t1, 0) && ofs == 0) {
1670
- uint64_t mask = MAKE_64BIT_MASK(0, len);
1673
+ width = 8 * tcg_type_size(ctx->type);
1674
+ type_mask = MAKE_64BIT_MASK(0, width);
1675
+ len_mask = MAKE_64BIT_MASK(0, len);
1676
1677
+ /* Inserting all-zero into a value. */
1678
+ if ((t2->z_mask & len_mask) == 0) {
1679
op->opc = INDEX_op_and;
1673
- op->args[1] = op->args[2];
1674
- op->args[2] = arg_new_constant(ctx, mask);
1680
+ op->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1681
return fold_and(ctx, op);
1682
}
1683
1678
- /* Inserting zero into a value. */
1679
- if (ti_is_const_val(t2, 0)) {
1680
- uint64_t mask = deposit64(-1, ofs, len, 0);
1681
-
1682
- op->opc = INDEX_op_and;
1683
- op->args[2] = arg_new_constant(ctx, mask);
1684
- return fold_and(ctx, op);
1684
+ /* Inserting all-one into a value. */
1685
+ if ((t2->o_mask & len_mask) == len_mask) {
1686
+ op->opc = INDEX_op_or;
1687
+ op->args[2] = arg_new_constant(ctx, len_mask << ofs);
1688
+ return fold_or(ctx, op);
1689
}
1690
1687
- /* The s_mask from the top portion of the deposit is still valid. */
1688
- if (ofs + len == width) {
1689
- s_mask = t2->s_mask << ofs;
1690
- } else {
1691
- s_mask = t1->s_mask & ~MAKE_64BIT_MASK(0, ofs + len);
1691
+ valid = TCG_TARGET_deposit_valid(ctx->type, ofs, len);
1692
+
1693
+ /* Lower invalid deposit of constant as AND + OR. */
1694
+ if (!valid && ti_is_const(t2)) {
1695
+ uint64_t ins_val = (ti_const_val(t2) & len_mask) << ofs;
1696
+
1697
+ op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1698
+ op2->args[0] = ret;
1699
+ op2->args[1] = arg1;
1700
+ op2->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1701
+ fold_and(ctx, op2);
1702
+
1703
+ op->opc = INDEX_op_or;
1704
+ op->args[1] = ret;
1705
+ op->args[2] = arg_new_constant(ctx, ins_val);
1706
+ return fold_or(ctx, op);
1707
}
1708
1709
+ /*
1710
+ * Compute result masks before calling other fold_* subroutines
1711
+ * which could modify the masks of our inputs.
1712
+ */
1713
z_mask = deposit64(t1->z_mask, ofs, len, t2->z_mask);
1714
o_mask = deposit64(t1->o_mask, ofs, len, t2->o_mask);
1715
+ if (ofs + len < width) {
1716
+ s_mask = t1->s_mask & ~MAKE_64BIT_MASK(0, ofs + len);
1717
+ } else {
1718
+ s_mask = t2->s_mask << ofs;
1719
+ }
1720
+
1721
+ /* Inserting a value into zero. */
1722
+ if (ti_is_const_val(t1, 0)) {
1723
+ uint64_t need_mask;
1724
+
1725
+ /* Always lower deposit into zero at 0 as AND. */
1726
+ if (ofs == 0) {
1727
+ op->opc = INDEX_op_and;
1728
+ op->args[1] = arg2;
1729
+ op->args[2] = arg_new_constant(ctx, len_mask);
1730
+ return fold_and(ctx, op);
1731
+ }
1732
+
1733
+ /*
1734
+ * If the portion of the value outside len that remains after
1735
+ * shifting is zero, we can elide the mask and just shift.
1736
+ */
1737
+ need_mask = t2->z_mask & ~len_mask;
1738
+ need_mask = (need_mask << ofs) & type_mask;
1739
+ if (!need_mask) {
1740
+ op->opc = INDEX_op_shl;
1741
+ op->args[1] = arg2;
1742
+ op->args[2] = arg_new_constant(ctx, ofs);
1743
+ goto done;
1744
+ }
1745
+
1746
+ /* Lower invalid deposit into zero as AND + SHL. */
1747
+ if (!valid) {
1748
+ op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1749
+ op2->args[0] = ret;
1750
+ op2->args[1] = arg2;
1751
+ op2->args[2] = arg_new_constant(ctx, len_mask);
1752
+ fold_and(ctx, op2);
1753
+
1754
+ op->opc = INDEX_op_shl;
1755
+ op->args[1] = ret;
1756
+ op->args[2] = arg_new_constant(ctx, ofs);
1757
+ goto done;
1758
+ }
1759
+ }
1760
+
1761
+ /* After special cases, lower invalid deposit. */
1762
+ if (!valid) {
1763
+ TCGArg tmp;
1764
+
1765
+ if (tcg_op_supported(INDEX_op_extract2, ctx->type, 0)) {
1766
+ if (ofs == 0 && tcg_op_supported(INDEX_op_rotl, ctx->type, 0)) {
1767
+ /*
1768
+ * ret = arg2:arg1 >> len
1769
+ * ret = rotl(ret, len)
1770
+ */
1771
+ op2 = opt_insert_before(ctx, op, INDEX_op_extract2, 4);
1772
+ op2->args[0] = ret;
1773
+ op2->args[1] = arg1;
1774
+ op2->args[2] = arg2;
1775
+ op2->args[3] = len;
1776
+
1777
+ op->opc = INDEX_op_rotl;
1778
+ op->args[1] = ret;
1779
+ op->args[2] = arg_new_constant(ctx, len);
1780
+ goto done;
1781
+ }
1782
+ if (ofs + len == width) {
1783
+ /*
1784
+ * tmp = arg1 << len
1785
+ * ret = arg2:tmp >> len
1786
+ */
1787
+ tmp = ret == arg2 ? arg_new_temp(ctx) : ret;
1788
+
1789
+ op2 = opt_insert_before(ctx, op, INDEX_op_shl, 4);
1790
+ op2->args[0] = tmp;
1791
+ op2->args[1] = arg1;
1792
+ op2->args[2] = arg_new_constant(ctx, len);
1793
+
1794
+ op->opc = INDEX_op_extract2;
1795
+ op->args[0] = ret;
1796
+ op->args[1] = tmp;
1797
+ op->args[2] = arg2;
1798
+ op->args[3] = len;
1799
+ goto done;
1800
+ }
1801
+ }
1802
+
1803
+ /*
1804
+ * tmp = arg2 & mask
1805
+ * ret = arg1 & ~(mask << ofs)
1806
+ * tmp = tmp << ofs
1807
+ * ret = ret | tmp
1808
+ */
1809
+ tmp = arg_new_temp(ctx);
1810
+
1811
+ op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1812
+ op2->args[0] = tmp;
1813
+ op2->args[1] = arg2;
1814
+ op2->args[2] = arg_new_constant(ctx, len_mask);
1815
+ fold_and(ctx, op2);
1816
+
1817
+ op2 = opt_insert_before(ctx, op, INDEX_op_shl, 3);
1818
+ op2->args[0] = tmp;
1819
+ op2->args[1] = tmp;
1820
+ op2->args[2] = arg_new_constant(ctx, ofs);
1821
+
1822
+ op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1823
+ op2->args[0] = ret;
1824
+ op2->args[1] = arg1;
1825
+ op2->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1826
+ fold_and(ctx, op2);
1827
+
1828
+ op->opc = INDEX_op_or;
1829
+ op->args[1] = ret;
1830
+ op->args[2] = tmp;
1831
+ }
1832
1833
+ done:
1834
return fold_masks_zos(ctx, op, z_mask, o_mask, s_mask);
1835
}
1836
tcg/tcg-op.c
+5
-78
@@ -876,9 +876,6 @@ void tcg_gen_rotri_i32(TCGv_i32 ret, TCGv_i32 arg1, int32_t arg2)
876
void tcg_gen_deposit_i32(TCGv_i32 ret, TCGv_i32 arg1, TCGv_i32 arg2,
877
unsigned int ofs, unsigned int len)
878
{
879
- uint32_t mask;
880
- TCGv_i32 t1;
881
-
879
tcg_debug_assert(ofs < 32);
880
tcg_debug_assert(len > 0);
881
tcg_debug_assert(len <= 32);
@@ -886,39 +883,9 @@ void tcg_gen_deposit_i32(TCGv_i32 ret, TCGv_i32 arg1, TCGv_i32 arg2,
883
884
if (len == 32) {
885
tcg_gen_mov_i32(ret, arg2);
889
- return;
890
- }
891
- if (TCG_TARGET_deposit_valid(TCG_TYPE_I32, ofs, len)) {
892
- tcg_gen_op5ii_i32(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
893
- return;
894
- }
895
-
896
- t1 = tcg_temp_ebb_new_i32();
897
-
898
- if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I32, 0)) {
899
- if (ofs + len == 32) {
900
- tcg_gen_shli_i32(t1, arg1, len);
901
- tcg_gen_extract2_i32(ret, t1, arg2, len);
902
- goto done;
903
- }
904
- if (ofs == 0) {
905
- tcg_gen_extract2_i32(ret, arg1, arg2, len);
906
- tcg_gen_rotli_i32(ret, ret, len);
907
- goto done;
908
- }
909
- }
910
-
911
- mask = (1u << len) - 1;
912
- if (ofs + len < 32) {
913
- tcg_gen_andi_i32(t1, arg2, mask);
914
- tcg_gen_shli_i32(t1, t1, ofs);
886
} else {
916
- tcg_gen_shli_i32(t1, arg2, ofs);
887
+ tcg_gen_op5ii_i32(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
888
}
918
- tcg_gen_andi_i32(ret, arg1, ~(mask << ofs));
919
- tcg_gen_or_i32(ret, ret, t1);
920
- done:
921
- tcg_temp_free_i32(t1);
889
}
890
891
void tcg_gen_deposit_z_i32(TCGv_i32 ret, TCGv_i32 arg,
@@ -932,13 +899,10 @@ void tcg_gen_deposit_z_i32(TCGv_i32 ret, TCGv_i32 arg,
899
if (ofs + len == 32) {
900
tcg_gen_shli_i32(ret, arg, ofs);
901
} else if (ofs == 0) {
935
- tcg_gen_andi_i32(ret, arg, (1u << len) - 1);
936
- } else if (TCG_TARGET_deposit_valid(TCG_TYPE_I32, ofs, len)) {
902
+ tcg_gen_extract_i32(ret, arg, 0, len);
903
+ } else {
904
TCGv_i32 zero = tcg_constant_i32(0);
905
tcg_gen_op5ii_i32(INDEX_op_deposit, ret, zero, arg, ofs, len);
939
- } else {
940
- tcg_gen_andi_i32(ret, arg, (1u << len) - 1);
941
- tcg_gen_shli_i32(ret, ret, ofs);
906
}
907
}
908
@@ -2133,9 +2097,6 @@ void tcg_gen_rotri_i64(TCGv_i64 ret, TCGv_i64 arg1, int64_t arg2)
2097
void tcg_gen_deposit_i64(TCGv_i64 ret, TCGv_i64 arg1, TCGv_i64 arg2,
2098
unsigned int ofs, unsigned int len)
2099
{
2136
- uint64_t mask;
2137
- TCGv_i64 t1;
2138
-
2100
tcg_debug_assert(ofs < 64);
2101
tcg_debug_assert(len > 0);
2102
tcg_debug_assert(len <= 64);
@@ -2143,40 +2104,9 @@ void tcg_gen_deposit_i64(TCGv_i64 ret, TCGv_i64 arg1, TCGv_i64 arg2,
2104
2105
if (len == 64) {
2106
tcg_gen_mov_i64(ret, arg2);
2146
- return;
2147
- }
2148
-
2149
- if (TCG_TARGET_deposit_valid(TCG_TYPE_I64, ofs, len)) {
2150
- tcg_gen_op5ii_i64(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
2151
- return;
2152
- }
2153
-
2154
- t1 = tcg_temp_ebb_new_i64();
2155
-
2156
- if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I64, 0)) {
2157
- if (ofs + len == 64) {
2158
- tcg_gen_shli_i64(t1, arg1, len);
2159
- tcg_gen_extract2_i64(ret, t1, arg2, len);
2160
- goto done;
2161
- }
2162
- if (ofs == 0) {
2163
- tcg_gen_extract2_i64(ret, arg1, arg2, len);
2164
- tcg_gen_rotli_i64(ret, ret, len);
2165
- goto done;
2166
- }
2167
- }
2168
-
2169
- mask = (1ull << len) - 1;
2170
- if (ofs + len < 64) {
2171
- tcg_gen_andi_i64(t1, arg2, mask);
2172
- tcg_gen_shli_i64(t1, t1, ofs);
2107
} else {
2174
- tcg_gen_shli_i64(t1, arg2, ofs);
2108
+ tcg_gen_op5ii_i64(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
2109
}
2176
- tcg_gen_andi_i64(ret, arg1, ~(mask << ofs));
2177
- tcg_gen_or_i64(ret, ret, t1);
2178
- done:
2179
- tcg_temp_free_i64(t1);
2110
}
2111
2112
void tcg_gen_deposit_z_i64(TCGv_i64 ret, TCGv_i64 arg,
@@ -2191,12 +2121,9 @@ void tcg_gen_deposit_z_i64(TCGv_i64 ret, TCGv_i64 arg,
2121
tcg_gen_shli_i64(ret, arg, ofs);
2122
} else if (ofs == 0) {
2123
tcg_gen_andi_i64(ret, arg, (1ull << len) - 1);
2194
- } else if (TCG_TARGET_deposit_valid(TCG_TYPE_I64, ofs, len)) {
2124
+ } else {
2125
TCGv_i64 zero = tcg_constant_i64(0);
2126
tcg_gen_op5ii_i64(INDEX_op_deposit, ret, zero, arg, ofs, len);
2197
- } else {
2198
- tcg_gen_andi_i64(ret, arg, (1ull << len) - 1);
2199
- tcg_gen_shli_i64(ret, ret, ofs);
2127
}
2128
}
2129