@samitouri / QOSamiQemu / commits / 5f747705a4

tcg/optimize: Lower unsupported deposit during optimize

The expansions that we chose in tcg-op.c may be less than optimal. Delay lowering until optimize, so that we have propagated constants and have computed known zero/one masks. Reviewed-by: Jim MacArthur <jim.macarthur@linaro.org> Signed-off-by: Richard Henderson <richard.henderson@linaro.org> Message-ID: <20260303010833.1115741-3-richard.henderson@linaro.org> Signed-off-by: Philippe Mathieu-Daudé <philmd@linaro.org>

Richard Henderson committed Oct 24, 2023 at 00:31 UTC 5f747705a45e808f159b40cd1ca59ef7c6c0a249
2 files changed +163 -99
tcg/optimize.c
+158 -21
@@ -1652,12 +1652,17 @@ static bool fold_ctpop(OptContext *ctx, TCGOp *op)
1652
1653 static bool fold_deposit(OptContext *ctx, TCGOp *op)
1654 {
1655 - TempOptInfo *t1 = arg_info(op->args[1]);
1656 - TempOptInfo *t2 = arg_info(op->args[2]);
1655 + TCGArg ret = op->args[0];
1656 + TCGArg arg1 = op->args[1];
1657 + TCGArg arg2 = op->args[2];
1658 int ofs = op->args[3];
1659 int len = op->args[4];
1659 - int width = 8 * tcg_type_size(ctx->type);
1660 - uint64_t z_mask, o_mask, s_mask;
1660 + TempOptInfo *t1 = arg_info(arg1);
1661 + TempOptInfo *t2 = arg_info(arg2);
1662 + int width;
1663 + uint64_t z_mask, o_mask, s_mask, type_mask, len_mask;
1664 + TCGOp *op2;
1665 + bool valid;
1666
1667 if (ti_is_const(t1) && ti_is_const(t2)) {
1668 return tcg_opt_gen_movi(ctx, op, op->args[0],
@@ -1665,35 +1670,167 @@ static bool fold_deposit(OptContext *ctx, TCGOp *op)
1670 ti_const_val(t2)));
1671 }
1672
1668 - /* Inserting a value into zero at offset 0. */
1669 - if (ti_is_const_val(t1, 0) && ofs == 0) {
1670 - uint64_t mask = MAKE_64BIT_MASK(0, len);
1673 + width = 8 * tcg_type_size(ctx->type);
1674 + type_mask = MAKE_64BIT_MASK(0, width);
1675 + len_mask = MAKE_64BIT_MASK(0, len);
1676
1677 + /* Inserting all-zero into a value. */
1678 + if ((t2->z_mask & len_mask) == 0) {
1679 op->opc = INDEX_op_and;
1673 - op->args[1] = op->args[2];
1674 - op->args[2] = arg_new_constant(ctx, mask);
1680 + op->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1681 return fold_and(ctx, op);
1682 }
1683
1678 - /* Inserting zero into a value. */
1679 - if (ti_is_const_val(t2, 0)) {
1680 - uint64_t mask = deposit64(-1, ofs, len, 0);
1681 -
1682 - op->opc = INDEX_op_and;
1683 - op->args[2] = arg_new_constant(ctx, mask);
1684 - return fold_and(ctx, op);
1684 + /* Inserting all-one into a value. */
1685 + if ((t2->o_mask & len_mask) == len_mask) {
1686 + op->opc = INDEX_op_or;
1687 + op->args[2] = arg_new_constant(ctx, len_mask << ofs);
1688 + return fold_or(ctx, op);
1689 }
1690
1687 - /* The s_mask from the top portion of the deposit is still valid. */
1688 - if (ofs + len == width) {
1689 - s_mask = t2->s_mask << ofs;
1690 - } else {
1691 - s_mask = t1->s_mask & ~MAKE_64BIT_MASK(0, ofs + len);
1691 + valid = TCG_TARGET_deposit_valid(ctx->type, ofs, len);
1692 +
1693 + /* Lower invalid deposit of constant as AND + OR. */
1694 + if (!valid && ti_is_const(t2)) {
1695 + uint64_t ins_val = (ti_const_val(t2) & len_mask) << ofs;
1696 +
1697 + op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1698 + op2->args[0] = ret;
1699 + op2->args[1] = arg1;
1700 + op2->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1701 + fold_and(ctx, op2);
1702 +
1703 + op->opc = INDEX_op_or;
1704 + op->args[1] = ret;
1705 + op->args[2] = arg_new_constant(ctx, ins_val);
1706 + return fold_or(ctx, op);
1707 }
1708
1709 + /*
1710 + * Compute result masks before calling other fold_* subroutines
1711 + * which could modify the masks of our inputs.
1712 + */
1713 z_mask = deposit64(t1->z_mask, ofs, len, t2->z_mask);
1714 o_mask = deposit64(t1->o_mask, ofs, len, t2->o_mask);
1715 + if (ofs + len < width) {
1716 + s_mask = t1->s_mask & ~MAKE_64BIT_MASK(0, ofs + len);
1717 + } else {
1718 + s_mask = t2->s_mask << ofs;
1719 + }
1720 +
1721 + /* Inserting a value into zero. */
1722 + if (ti_is_const_val(t1, 0)) {
1723 + uint64_t need_mask;
1724 +
1725 + /* Always lower deposit into zero at 0 as AND. */
1726 + if (ofs == 0) {
1727 + op->opc = INDEX_op_and;
1728 + op->args[1] = arg2;
1729 + op->args[2] = arg_new_constant(ctx, len_mask);
1730 + return fold_and(ctx, op);
1731 + }
1732 +
1733 + /*
1734 + * If the portion of the value outside len that remains after
1735 + * shifting is zero, we can elide the mask and just shift.
1736 + */
1737 + need_mask = t2->z_mask & ~len_mask;
1738 + need_mask = (need_mask << ofs) & type_mask;
1739 + if (!need_mask) {
1740 + op->opc = INDEX_op_shl;
1741 + op->args[1] = arg2;
1742 + op->args[2] = arg_new_constant(ctx, ofs);
1743 + goto done;
1744 + }
1745 +
1746 + /* Lower invalid deposit into zero as AND + SHL. */
1747 + if (!valid) {
1748 + op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1749 + op2->args[0] = ret;
1750 + op2->args[1] = arg2;
1751 + op2->args[2] = arg_new_constant(ctx, len_mask);
1752 + fold_and(ctx, op2);
1753 +
1754 + op->opc = INDEX_op_shl;
1755 + op->args[1] = ret;
1756 + op->args[2] = arg_new_constant(ctx, ofs);
1757 + goto done;
1758 + }
1759 + }
1760 +
1761 + /* After special cases, lower invalid deposit. */
1762 + if (!valid) {
1763 + TCGArg tmp;
1764 +
1765 + if (tcg_op_supported(INDEX_op_extract2, ctx->type, 0)) {
1766 + if (ofs == 0 && tcg_op_supported(INDEX_op_rotl, ctx->type, 0)) {
1767 + /*
1768 + * ret = arg2:arg1 >> len
1769 + * ret = rotl(ret, len)
1770 + */
1771 + op2 = opt_insert_before(ctx, op, INDEX_op_extract2, 4);
1772 + op2->args[0] = ret;
1773 + op2->args[1] = arg1;
1774 + op2->args[2] = arg2;
1775 + op2->args[3] = len;
1776 +
1777 + op->opc = INDEX_op_rotl;
1778 + op->args[1] = ret;
1779 + op->args[2] = arg_new_constant(ctx, len);
1780 + goto done;
1781 + }
1782 + if (ofs + len == width) {
1783 + /*
1784 + * tmp = arg1 << len
1785 + * ret = arg2:tmp >> len
1786 + */
1787 + tmp = ret == arg2 ? arg_new_temp(ctx) : ret;
1788 +
1789 + op2 = opt_insert_before(ctx, op, INDEX_op_shl, 4);
1790 + op2->args[0] = tmp;
1791 + op2->args[1] = arg1;
1792 + op2->args[2] = arg_new_constant(ctx, len);
1793 +
1794 + op->opc = INDEX_op_extract2;
1795 + op->args[0] = ret;
1796 + op->args[1] = tmp;
1797 + op->args[2] = arg2;
1798 + op->args[3] = len;
1799 + goto done;
1800 + }
1801 + }
1802 +
1803 + /*
1804 + * tmp = arg2 & mask
1805 + * ret = arg1 & ~(mask << ofs)
1806 + * tmp = tmp << ofs
1807 + * ret = ret | tmp
1808 + */
1809 + tmp = arg_new_temp(ctx);
1810 +
1811 + op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1812 + op2->args[0] = tmp;
1813 + op2->args[1] = arg2;
1814 + op2->args[2] = arg_new_constant(ctx, len_mask);
1815 + fold_and(ctx, op2);
1816 +
1817 + op2 = opt_insert_before(ctx, op, INDEX_op_shl, 3);
1818 + op2->args[0] = tmp;
1819 + op2->args[1] = tmp;
1820 + op2->args[2] = arg_new_constant(ctx, ofs);
1821 +
1822 + op2 = opt_insert_before(ctx, op, INDEX_op_and, 3);
1823 + op2->args[0] = ret;
1824 + op2->args[1] = arg1;
1825 + op2->args[2] = arg_new_constant(ctx, ~(len_mask << ofs));
1826 + fold_and(ctx, op2);
1827 +
1828 + op->opc = INDEX_op_or;
1829 + op->args[1] = ret;
1830 + op->args[2] = tmp;
1831 + }
1832
1833 + done:
1834 return fold_masks_zos(ctx, op, z_mask, o_mask, s_mask);
1835 }
1836
tcg/tcg-op.c
+5 -78
@@ -876,9 +876,6 @@ void tcg_gen_rotri_i32(TCGv_i32 ret, TCGv_i32 arg1, int32_t arg2)
876 void tcg_gen_deposit_i32(TCGv_i32 ret, TCGv_i32 arg1, TCGv_i32 arg2,
877 unsigned int ofs, unsigned int len)
878 {
879 - uint32_t mask;
880 - TCGv_i32 t1;
881 -
879 tcg_debug_assert(ofs < 32);
880 tcg_debug_assert(len > 0);
881 tcg_debug_assert(len <= 32);
@@ -886,39 +883,9 @@ void tcg_gen_deposit_i32(TCGv_i32 ret, TCGv_i32 arg1, TCGv_i32 arg2,
883
884 if (len == 32) {
885 tcg_gen_mov_i32(ret, arg2);
889 - return;
890 - }
891 - if (TCG_TARGET_deposit_valid(TCG_TYPE_I32, ofs, len)) {
892 - tcg_gen_op5ii_i32(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
893 - return;
894 - }
895 -
896 - t1 = tcg_temp_ebb_new_i32();
897 -
898 - if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I32, 0)) {
899 - if (ofs + len == 32) {
900 - tcg_gen_shli_i32(t1, arg1, len);
901 - tcg_gen_extract2_i32(ret, t1, arg2, len);
902 - goto done;
903 - }
904 - if (ofs == 0) {
905 - tcg_gen_extract2_i32(ret, arg1, arg2, len);
906 - tcg_gen_rotli_i32(ret, ret, len);
907 - goto done;
908 - }
909 - }
910 -
911 - mask = (1u << len) - 1;
912 - if (ofs + len < 32) {
913 - tcg_gen_andi_i32(t1, arg2, mask);
914 - tcg_gen_shli_i32(t1, t1, ofs);
886 } else {
916 - tcg_gen_shli_i32(t1, arg2, ofs);
887 + tcg_gen_op5ii_i32(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
888 }
918 - tcg_gen_andi_i32(ret, arg1, ~(mask << ofs));
919 - tcg_gen_or_i32(ret, ret, t1);
920 - done:
921 - tcg_temp_free_i32(t1);
889 }
890
891 void tcg_gen_deposit_z_i32(TCGv_i32 ret, TCGv_i32 arg,
@@ -932,13 +899,10 @@ void tcg_gen_deposit_z_i32(TCGv_i32 ret, TCGv_i32 arg,
899 if (ofs + len == 32) {
900 tcg_gen_shli_i32(ret, arg, ofs);
901 } else if (ofs == 0) {
935 - tcg_gen_andi_i32(ret, arg, (1u << len) - 1);
936 - } else if (TCG_TARGET_deposit_valid(TCG_TYPE_I32, ofs, len)) {
902 + tcg_gen_extract_i32(ret, arg, 0, len);
903 + } else {
904 TCGv_i32 zero = tcg_constant_i32(0);
905 tcg_gen_op5ii_i32(INDEX_op_deposit, ret, zero, arg, ofs, len);
939 - } else {
940 - tcg_gen_andi_i32(ret, arg, (1u << len) - 1);
941 - tcg_gen_shli_i32(ret, ret, ofs);
906 }
907 }
908
@@ -2133,9 +2097,6 @@ void tcg_gen_rotri_i64(TCGv_i64 ret, TCGv_i64 arg1, int64_t arg2)
2097 void tcg_gen_deposit_i64(TCGv_i64 ret, TCGv_i64 arg1, TCGv_i64 arg2,
2098 unsigned int ofs, unsigned int len)
2099 {
2136 - uint64_t mask;
2137 - TCGv_i64 t1;
2138 -
2100 tcg_debug_assert(ofs < 64);
2101 tcg_debug_assert(len > 0);
2102 tcg_debug_assert(len <= 64);
@@ -2143,40 +2104,9 @@ void tcg_gen_deposit_i64(TCGv_i64 ret, TCGv_i64 arg1, TCGv_i64 arg2,
2104
2105 if (len == 64) {
2106 tcg_gen_mov_i64(ret, arg2);
2146 - return;
2147 - }
2148 -
2149 - if (TCG_TARGET_deposit_valid(TCG_TYPE_I64, ofs, len)) {
2150 - tcg_gen_op5ii_i64(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
2151 - return;
2152 - }
2153 -
2154 - t1 = tcg_temp_ebb_new_i64();
2155 -
2156 - if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I64, 0)) {
2157 - if (ofs + len == 64) {
2158 - tcg_gen_shli_i64(t1, arg1, len);
2159 - tcg_gen_extract2_i64(ret, t1, arg2, len);
2160 - goto done;
2161 - }
2162 - if (ofs == 0) {
2163 - tcg_gen_extract2_i64(ret, arg1, arg2, len);
2164 - tcg_gen_rotli_i64(ret, ret, len);
2165 - goto done;
2166 - }
2167 - }
2168 -
2169 - mask = (1ull << len) - 1;
2170 - if (ofs + len < 64) {
2171 - tcg_gen_andi_i64(t1, arg2, mask);
2172 - tcg_gen_shli_i64(t1, t1, ofs);
2107 } else {
2174 - tcg_gen_shli_i64(t1, arg2, ofs);
2108 + tcg_gen_op5ii_i64(INDEX_op_deposit, ret, arg1, arg2, ofs, len);
2109 }
2176 - tcg_gen_andi_i64(ret, arg1, ~(mask << ofs));
2177 - tcg_gen_or_i64(ret, ret, t1);
2178 - done:
2179 - tcg_temp_free_i64(t1);
2110 }
2111
2112 void tcg_gen_deposit_z_i64(TCGv_i64 ret, TCGv_i64 arg,
@@ -2191,12 +2121,9 @@ void tcg_gen_deposit_z_i64(TCGv_i64 ret, TCGv_i64 arg,
2121 tcg_gen_shli_i64(ret, arg, ofs);
2122 } else if (ofs == 0) {
2123 tcg_gen_andi_i64(ret, arg, (1ull << len) - 1);
2194 - } else if (TCG_TARGET_deposit_valid(TCG_TYPE_I64, ofs, len)) {
2124 + } else {
2125 TCGv_i64 zero = tcg_constant_i64(0);
2126 tcg_gen_op5ii_i64(INDEX_op_deposit, ret, zero, arg, ofs, len);
2197 - } else {
2198 - tcg_gen_andi_i64(ret, arg, (1ull << len) - 1);
2199 - tcg_gen_shli_i64(ret, ret, ofs);
2127 }
2128 }
2129