@samitouri / QOSamiQemu / commits / e4cebfc664

tcg/optimize: Lower unsupported extract2 during optimize

The expansions that we chose in tcg-op.c may be less than optimal. Delay lowering until optimize, so that we have propagated constants and have computed known zero/one masks. Reviewed-by: Jim MacArthur <jim.macarthur@linaro.org> Reviewed-by: Manos Pitsidianakis <manos.pitsidianakis@linaro.org> Signed-off-by: Richard Henderson <richard.henderson@linaro.org> Message-ID: <20260303010833.1115741-4-richard.henderson@linaro.org> Signed-off-by: Philippe Mathieu-Daudé <philmd@linaro.org>

Richard Henderson committed Jan 1, 2025 at 20:55 UTC e4cebfc664a4b617969b1ceb911e7b65b5a846ac
2 files changed +60 -17
tcg/optimize.c
+58 -5
@@ -1918,21 +1918,74 @@ static bool fold_extract2(OptContext *ctx, TCGOp *op)
1918 uint64_t z2 = t2->z_mask;
1919 uint64_t o1 = t1->o_mask;
1920 uint64_t o2 = t2->o_mask;
1921 + uint64_t zr, or;
1922 int shr = op->args[3];
1923 + int shl;
1924
1925 if (ctx->type == TCG_TYPE_I32) {
1926 z1 = (uint32_t)z1 >> shr;
1927 o1 = (uint32_t)o1 >> shr;
1926 - z2 = (uint64_t)((int32_t)z2 << (32 - shr));
1927 - o2 = (uint64_t)((int32_t)o2 << (32 - shr));
1928 + shl = 32 - shr;
1929 + z2 = (uint64_t)((int32_t)z2 << shl);
1930 + o2 = (uint64_t)((int32_t)o2 << shl);
1931 } else {
1932 z1 >>= shr;
1933 o1 >>= shr;
1931 - z2 <<= 64 - shr;
1932 - o2 <<= 64 - shr;
1934 + shl = 64 - shr;
1935 + z2 <<= shl;
1936 + o2 <<= shl;
1937 }
1938 + zr = z1 | z2;
1939 + or = o1 | o2;
1940
1935 - return fold_masks_zo(ctx, op, z1 | z2, o1 | o2);
1941 + if (zr == or) {
1942 + return tcg_opt_gen_movi(ctx, op, op->args[0], zr);
1943 + }
1944 +
1945 + if (z2 == 0) {
1946 + /* High part zeros folds to simple right shift. */
1947 + op->opc = INDEX_op_shr;
1948 + op->args[2] = arg_new_constant(ctx, shr);
1949 + } else if (z1 == 0) {
1950 + /* Low part zeros folds to simple left shift. */
1951 + op->opc = INDEX_op_shl;
1952 + op->args[1] = op->args[2];
1953 + op->args[2] = arg_new_constant(ctx, shl);
1954 + } else if (!tcg_op_supported(INDEX_op_extract2, ctx->type, 0)) {
1955 + TCGArg tmp = arg_new_temp(ctx);
1956 + TCGOp *op2 = opt_insert_before(ctx, op, INDEX_op_shr, 3);
1957 +
1958 + op2->args[0] = tmp;
1959 + op2->args[1] = op->args[1];
1960 + op2->args[2] = arg_new_constant(ctx, shr);
1961 +
1962 + if (TCG_TARGET_deposit_valid(ctx->type, shl, shr)) {
1963 + /*
1964 + * Deposit has more arguments than extract2,
1965 + * so we need to create a new TCGOp.
1966 + */
1967 + op2 = opt_insert_before(ctx, op, INDEX_op_deposit, 5);
1968 + op2->args[0] = op->args[0];
1969 + op2->args[1] = tmp;
1970 + op2->args[2] = op->args[2];
1971 + op2->args[3] = shl;
1972 + op2->args[4] = shr;
1973 +
1974 + tcg_op_remove(ctx->tcg, op);
1975 + op = op2;
1976 + } else {
1977 + op2 = opt_insert_before(ctx, op, INDEX_op_shl, 3);
1978 + op2->args[0] = op->args[0];
1979 + op2->args[1] = op->args[2];
1980 + op2->args[2] = arg_new_constant(ctx, shl);
1981 +
1982 + op->opc = INDEX_op_or;
1983 + op->args[1] = op->args[0];
1984 + op->args[2] = tmp;
1985 + }
1986 + }
1987 +
1988 + return fold_masks_zo(ctx, op, zr, or);
1989 }
1990
1991 static bool fold_exts(OptContext *ctx, TCGOp *op)
tcg/tcg-op.c
+2 -12
@@ -1000,13 +1000,8 @@ void tcg_gen_extract2_i32(TCGv_i32 ret, TCGv_i32 al, TCGv_i32 ah,
1000 tcg_gen_mov_i32(ret, ah);
1001 } else if (al == ah) {
1002 tcg_gen_rotri_i32(ret, al, ofs);
1003 - } else if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I32, 0)) {
1004 - tcg_gen_op4i_i32(INDEX_op_extract2, ret, al, ah, ofs);
1003 } else {
1006 - TCGv_i32 t0 = tcg_temp_ebb_new_i32();
1007 - tcg_gen_shri_i32(t0, al, ofs);
1008 - tcg_gen_deposit_i32(ret, t0, ah, 32 - ofs, ofs);
1009 - tcg_temp_free_i32(t0);
1004 + tcg_gen_op4i_i32(INDEX_op_extract2, ret, al, ah, ofs);
1005 }
1006 }
1007
@@ -2221,13 +2216,8 @@ void tcg_gen_extract2_i64(TCGv_i64 ret, TCGv_i64 al, TCGv_i64 ah,
2216 tcg_gen_mov_i64(ret, ah);
2217 } else if (al == ah) {
2218 tcg_gen_rotri_i64(ret, al, ofs);
2224 - } else if (tcg_op_supported(INDEX_op_extract2, TCG_TYPE_I64, 0)) {
2225 - tcg_gen_op4i_i64(INDEX_op_extract2, ret, al, ah, ofs);
2219 } else {
2227 - TCGv_i64 t0 = tcg_temp_ebb_new_i64();
2228 - tcg_gen_shri_i64(t0, al, ofs);
2229 - tcg_gen_deposit_i64(ret, t0, ah, 64 - ofs, ofs);
2230 - tcg_temp_free_i64(t0);
2220 + tcg_gen_op4i_i64(INDEX_op_extract2, ret, al, ah, ofs);
2221 }
2222 }
2223