| 1 | /* |
| 2 | * RISC-V translation routines for the RV64M Standard Extension. |
| 3 | * |
| 4 | * Copyright (c) 2016-2017 Sagar Karandikar, sagark@eecs.berkeley.edu |
| 5 | * Copyright (c) 2018 Peer Adelt, peer.adelt@hni.uni-paderborn.de |
| 6 | * Bastian Koppelmann, kbastian@mail.uni-paderborn.de |
| 7 | * |
| 8 | * This program is free software; you can redistribute it and/or modify it |
| 9 | * under the terms and conditions of the GNU General Public License, |
| 10 | * version 2 or later, as published by the Free Software Foundation. |
| 11 | * |
| 12 | * This program is distributed in the hope it will be useful, but WITHOUT |
| 13 | * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or |
| 14 | * FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for |
| 15 | * more details. |
| 16 | * |
| 17 | * You should have received a copy of the GNU General Public License along with |
| 18 | * this program. If not, see <http://www.gnu.org/licenses/>. |
| 19 | */ |
| 20 | |
| 21 | #define REQUIRE_M_OR_ZMMUL(ctx) do { \ |
| 22 | if (!ctx->cfg_ptr->ext_zmmul && !has_ext(ctx, RVM)) { \ |
| 23 | return false; \ |
| 24 | } \ |
| 25 | } while (0) |
| 26 | |
| 27 | static void gen_mulhu_i128(TCGv r2, TCGv r3, TCGv al, TCGv ah, TCGv bl, TCGv bh) |
| 28 | { |
| 29 | TCGv tmpl = tcg_temp_new(); |
| 30 | TCGv tmph = tcg_temp_new(); |
| 31 | TCGv r0 = tcg_temp_new(); |
| 32 | TCGv r1 = tcg_temp_new(); |
| 33 | TCGv zero = tcg_constant_tl(0); |
| 34 | |
| 35 | tcg_gen_mulu2_tl(r0, r1, al, bl); |
| 36 | |
| 37 | tcg_gen_mulu2_tl(tmpl, tmph, al, bh); |
| 38 | tcg_gen_add2_tl(r1, r2, r1, zero, tmpl, tmph); |
| 39 | tcg_gen_mulu2_tl(tmpl, tmph, ah, bl); |
| 40 | tcg_gen_add2_tl(r1, tmph, r1, r2, tmpl, tmph); |
| 41 | /* Overflow detection into r3 */ |
| 42 | tcg_gen_setcond_tl(TCG_COND_LTU, r3, tmph, r2); |
| 43 | |
| 44 | tcg_gen_mov_tl(r2, tmph); |
| 45 | |
| 46 | tcg_gen_mulu2_tl(tmpl, tmph, ah, bh); |
| 47 | tcg_gen_add2_tl(r2, r3, r2, r3, tmpl, tmph); |
| 48 | } |
| 49 | |
| 50 | static void gen_mul_i128(TCGv rl, TCGv rh, |
| 51 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 52 | { |
| 53 | TCGv tmpl = tcg_temp_new(); |
| 54 | TCGv tmph = tcg_temp_new(); |
| 55 | TCGv tmpx = tcg_temp_new(); |
| 56 | TCGv zero = tcg_constant_tl(0); |
| 57 | |
| 58 | tcg_gen_mulu2_tl(rl, rh, rs1l, rs2l); |
| 59 | tcg_gen_mulu2_tl(tmpl, tmph, rs1l, rs2h); |
| 60 | tcg_gen_add2_tl(rh, tmpx, rh, zero, tmpl, tmph); |
| 61 | tcg_gen_mulu2_tl(tmpl, tmph, rs1h, rs2l); |
| 62 | tcg_gen_add2_tl(rh, tmph, rh, tmpx, tmpl, tmph); |
| 63 | } |
| 64 | |
| 65 | static bool trans_mul(DisasContext *ctx, arg_mul *a) |
| 66 | { |
| 67 | REQUIRE_M_OR_ZMMUL(ctx); |
| 68 | return gen_arith(ctx, a, EXT_NONE, tcg_gen_mul_tl, gen_mul_i128); |
| 69 | } |
| 70 | |
| 71 | static void gen_mulh_i128(TCGv rl, TCGv rh, |
| 72 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 73 | { |
| 74 | TCGv t0l = tcg_temp_new(); |
| 75 | TCGv t0h = tcg_temp_new(); |
| 76 | TCGv t1l = tcg_temp_new(); |
| 77 | TCGv t1h = tcg_temp_new(); |
| 78 | |
| 79 | gen_mulhu_i128(rl, rh, rs1l, rs1h, rs2l, rs2h); |
| 80 | tcg_gen_sari_tl(t0h, rs1h, 63); |
| 81 | tcg_gen_and_tl(t0l, t0h, rs2l); |
| 82 | tcg_gen_and_tl(t0h, t0h, rs2h); |
| 83 | tcg_gen_sari_tl(t1h, rs2h, 63); |
| 84 | tcg_gen_and_tl(t1l, t1h, rs1l); |
| 85 | tcg_gen_and_tl(t1h, t1h, rs1h); |
| 86 | tcg_gen_sub2_tl(t0l, t0h, rl, rh, t0l, t0h); |
| 87 | tcg_gen_sub2_tl(rl, rh, t0l, t0h, t1l, t1h); |
| 88 | } |
| 89 | |
| 90 | static void gen_mulh(TCGv ret, TCGv s1, TCGv s2) |
| 91 | { |
| 92 | TCGv discard = tcg_temp_new(); |
| 93 | |
| 94 | tcg_gen_muls2_tl(discard, ret, s1, s2); |
| 95 | } |
| 96 | |
| 97 | static void gen_mulh_w(TCGv ret, TCGv s1, TCGv s2) |
| 98 | { |
| 99 | tcg_gen_mul_tl(ret, s1, s2); |
| 100 | tcg_gen_sari_tl(ret, ret, 32); |
| 101 | } |
| 102 | |
| 103 | static bool trans_mulh(DisasContext *ctx, arg_mulh *a) |
| 104 | { |
| 105 | REQUIRE_M_OR_ZMMUL(ctx); |
| 106 | return gen_arith_per_ol(ctx, a, EXT_SIGN, gen_mulh, gen_mulh_w, |
| 107 | gen_mulh_i128); |
| 108 | } |
| 109 | |
| 110 | static void gen_mulhsu_i128(TCGv rl, TCGv rh, |
| 111 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 112 | { |
| 113 | |
| 114 | TCGv t0l = tcg_temp_new(); |
| 115 | TCGv t0h = tcg_temp_new(); |
| 116 | |
| 117 | gen_mulhu_i128(rl, rh, rs1l, rs1h, rs2l, rs2h); |
| 118 | tcg_gen_sari_tl(t0h, rs1h, 63); |
| 119 | tcg_gen_and_tl(t0l, t0h, rs2l); |
| 120 | tcg_gen_and_tl(t0h, t0h, rs2h); |
| 121 | tcg_gen_sub2_tl(rl, rh, rl, rh, t0l, t0h); |
| 122 | } |
| 123 | |
| 124 | static void gen_mulhsu(TCGv ret, TCGv arg1, TCGv arg2) |
| 125 | { |
| 126 | TCGv rl = tcg_temp_new(); |
| 127 | TCGv rh = tcg_temp_new(); |
| 128 | |
| 129 | tcg_gen_mulu2_tl(rl, rh, arg1, arg2); |
| 130 | /* fix up for one negative */ |
| 131 | tcg_gen_sari_tl(rl, arg1, TARGET_LONG_BITS - 1); |
| 132 | tcg_gen_and_tl(rl, rl, arg2); |
| 133 | tcg_gen_sub_tl(ret, rh, rl); |
| 134 | } |
| 135 | |
| 136 | static void gen_mulhsu_w(TCGv ret, TCGv arg1, TCGv arg2) |
| 137 | { |
| 138 | TCGv t1 = tcg_temp_new(); |
| 139 | TCGv t2 = tcg_temp_new(); |
| 140 | |
| 141 | tcg_gen_ext32s_tl(t1, arg1); |
| 142 | tcg_gen_ext32u_tl(t2, arg2); |
| 143 | tcg_gen_mul_tl(ret, t1, t2); |
| 144 | tcg_gen_sari_tl(ret, ret, 32); |
| 145 | } |
| 146 | |
| 147 | static bool trans_mulhsu(DisasContext *ctx, arg_mulhsu *a) |
| 148 | { |
| 149 | REQUIRE_M_OR_ZMMUL(ctx); |
| 150 | return gen_arith_per_ol(ctx, a, EXT_NONE, gen_mulhsu, gen_mulhsu_w, |
| 151 | gen_mulhsu_i128); |
| 152 | } |
| 153 | |
| 154 | static void gen_mulhu(TCGv ret, TCGv s1, TCGv s2) |
| 155 | { |
| 156 | TCGv discard = tcg_temp_new(); |
| 157 | |
| 158 | tcg_gen_mulu2_tl(discard, ret, s1, s2); |
| 159 | } |
| 160 | |
| 161 | static bool trans_mulhu(DisasContext *ctx, arg_mulhu *a) |
| 162 | { |
| 163 | REQUIRE_M_OR_ZMMUL(ctx); |
| 164 | /* gen_mulh_w works for either sign as input. */ |
| 165 | return gen_arith_per_ol(ctx, a, EXT_ZERO, gen_mulhu, gen_mulh_w, |
| 166 | gen_mulhu_i128); |
| 167 | } |
| 168 | |
| 169 | static void gen_div_i128(TCGv rdl, TCGv rdh, |
| 170 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 171 | { |
| 172 | TCGv_i64 wide_rdh = tcg_temp_new_i64(); |
| 173 | gen_helper_divs_i128(rdl, tcg_env, rs1l, rs1h, rs2l, rs2h); |
| 174 | tcg_gen_ld_i64(wide_rdh, tcg_env, offsetof(CPURISCVState, retxh)); |
| 175 | tcg_gen_trunc_i64_tl(rdh, wide_rdh); |
| 176 | } |
| 177 | |
| 178 | static void gen_div(TCGv ret, TCGv source1, TCGv source2) |
| 179 | { |
| 180 | TCGv temp1, temp2, zero, one, mone, min; |
| 181 | |
| 182 | temp1 = tcg_temp_new(); |
| 183 | temp2 = tcg_temp_new(); |
| 184 | zero = tcg_constant_tl(0); |
| 185 | one = tcg_constant_tl(1); |
| 186 | mone = tcg_constant_tl(-1); |
| 187 | min = tcg_constant_tl(1ull << (TARGET_LONG_BITS - 1)); |
| 188 | |
| 189 | /* |
| 190 | * If overflow, set temp2 to 1, else source2. |
| 191 | * This produces the required result of min. |
| 192 | */ |
| 193 | tcg_gen_setcond_tl(TCG_COND_EQ, temp1, source1, min); |
| 194 | tcg_gen_setcond_tl(TCG_COND_EQ, temp2, source2, mone); |
| 195 | tcg_gen_and_tl(temp1, temp1, temp2); |
| 196 | tcg_gen_movcond_tl(TCG_COND_NE, temp2, temp1, zero, one, source2); |
| 197 | |
| 198 | /* |
| 199 | * If div by zero, set temp1 to -1 and temp2 to 1 to |
| 200 | * produce the required result of -1. |
| 201 | */ |
| 202 | tcg_gen_movcond_tl(TCG_COND_EQ, temp1, source2, zero, mone, source1); |
| 203 | tcg_gen_movcond_tl(TCG_COND_EQ, temp2, source2, zero, one, temp2); |
| 204 | |
| 205 | tcg_gen_div_tl(ret, temp1, temp2); |
| 206 | } |
| 207 | |
| 208 | static bool trans_div(DisasContext *ctx, arg_div *a) |
| 209 | { |
| 210 | REQUIRE_EXT(ctx, RVM); |
| 211 | return gen_arith(ctx, a, EXT_SIGN, gen_div, gen_div_i128); |
| 212 | } |
| 213 | |
| 214 | static void gen_divu_i128(TCGv rdl, TCGv rdh, |
| 215 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 216 | { |
| 217 | TCGv_i64 wide_rdh = tcg_temp_new_i64(); |
| 218 | gen_helper_divu_i128(rdl, tcg_env, rs1l, rs1h, rs2l, rs2h); |
| 219 | tcg_gen_ld_i64(wide_rdh, tcg_env, offsetof(CPURISCVState, retxh)); |
| 220 | tcg_gen_trunc_i64_tl(rdh, wide_rdh); |
| 221 | } |
| 222 | |
| 223 | static void gen_divu(TCGv ret, TCGv source1, TCGv source2) |
| 224 | { |
| 225 | TCGv temp1, temp2, zero, one, max; |
| 226 | |
| 227 | temp1 = tcg_temp_new(); |
| 228 | temp2 = tcg_temp_new(); |
| 229 | zero = tcg_constant_tl(0); |
| 230 | one = tcg_constant_tl(1); |
| 231 | max = tcg_constant_tl(~0); |
| 232 | |
| 233 | /* |
| 234 | * If div by zero, set temp1 to max and temp2 to 1 to |
| 235 | * produce the required result of max. |
| 236 | */ |
| 237 | tcg_gen_movcond_tl(TCG_COND_EQ, temp1, source2, zero, max, source1); |
| 238 | tcg_gen_movcond_tl(TCG_COND_EQ, temp2, source2, zero, one, source2); |
| 239 | tcg_gen_divu_tl(ret, temp1, temp2); |
| 240 | } |
| 241 | |
| 242 | static bool trans_divu(DisasContext *ctx, arg_divu *a) |
| 243 | { |
| 244 | REQUIRE_EXT(ctx, RVM); |
| 245 | return gen_arith(ctx, a, EXT_ZERO, gen_divu, gen_divu_i128); |
| 246 | } |
| 247 | |
| 248 | static void gen_rem_i128(TCGv rdl, TCGv rdh, |
| 249 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 250 | { |
| 251 | TCGv_i64 wide_rdh = tcg_temp_new_i64(); |
| 252 | gen_helper_rems_i128(rdl, tcg_env, rs1l, rs1h, rs2l, rs2h); |
| 253 | tcg_gen_ld_i64(wide_rdh, tcg_env, offsetof(CPURISCVState, retxh)); |
| 254 | tcg_gen_trunc_i64_tl(rdh, wide_rdh); |
| 255 | } |
| 256 | |
| 257 | static void gen_rem(TCGv ret, TCGv source1, TCGv source2) |
| 258 | { |
| 259 | TCGv temp1, temp2, zero, one, mone, min; |
| 260 | |
| 261 | temp1 = tcg_temp_new(); |
| 262 | temp2 = tcg_temp_new(); |
| 263 | zero = tcg_constant_tl(0); |
| 264 | one = tcg_constant_tl(1); |
| 265 | mone = tcg_constant_tl(-1); |
| 266 | min = tcg_constant_tl(1ull << (TARGET_LONG_BITS - 1)); |
| 267 | |
| 268 | /* |
| 269 | * If overflow, set temp1 to 0, else source1. |
| 270 | * This avoids a possible host trap, and produces the required result of 0. |
| 271 | */ |
| 272 | tcg_gen_setcond_tl(TCG_COND_EQ, temp1, source1, min); |
| 273 | tcg_gen_setcond_tl(TCG_COND_EQ, temp2, source2, mone); |
| 274 | tcg_gen_and_tl(temp1, temp1, temp2); |
| 275 | tcg_gen_movcond_tl(TCG_COND_NE, temp1, temp1, zero, zero, source1); |
| 276 | |
| 277 | /* |
| 278 | * If div by zero, set temp2 to 1, else source2. |
| 279 | * This avoids a possible host trap, but produces an incorrect result. |
| 280 | */ |
| 281 | tcg_gen_movcond_tl(TCG_COND_EQ, temp2, source2, zero, one, source2); |
| 282 | |
| 283 | tcg_gen_rem_tl(temp1, temp1, temp2); |
| 284 | |
| 285 | /* If div by zero, the required result is the original dividend. */ |
| 286 | tcg_gen_movcond_tl(TCG_COND_EQ, ret, source2, zero, source1, temp1); |
| 287 | } |
| 288 | |
| 289 | static bool trans_rem(DisasContext *ctx, arg_rem *a) |
| 290 | { |
| 291 | REQUIRE_EXT(ctx, RVM); |
| 292 | return gen_arith(ctx, a, EXT_SIGN, gen_rem, gen_rem_i128); |
| 293 | } |
| 294 | |
| 295 | static void gen_remu_i128(TCGv rdl, TCGv rdh, |
| 296 | TCGv rs1l, TCGv rs1h, TCGv rs2l, TCGv rs2h) |
| 297 | { |
| 298 | TCGv_i64 wide_rdh = tcg_temp_new_i64(); |
| 299 | gen_helper_remu_i128(rdl, tcg_env, rs1l, rs1h, rs2l, rs2h); |
| 300 | tcg_gen_ld_i64(wide_rdh, tcg_env, offsetof(CPURISCVState, retxh)); |
| 301 | tcg_gen_trunc_i64_tl(rdh, wide_rdh); |
| 302 | } |
| 303 | |
| 304 | static void gen_remu(TCGv ret, TCGv source1, TCGv source2) |
| 305 | { |
| 306 | TCGv temp, zero, one; |
| 307 | |
| 308 | temp = tcg_temp_new(); |
| 309 | zero = tcg_constant_tl(0); |
| 310 | one = tcg_constant_tl(1); |
| 311 | |
| 312 | /* |
| 313 | * If div by zero, set temp to 1, else source2. |
| 314 | * This avoids a possible host trap, but produces an incorrect result. |
| 315 | */ |
| 316 | tcg_gen_movcond_tl(TCG_COND_EQ, temp, source2, zero, one, source2); |
| 317 | |
| 318 | tcg_gen_remu_tl(temp, source1, temp); |
| 319 | |
| 320 | /* If div by zero, the required result is the original dividend. */ |
| 321 | tcg_gen_movcond_tl(TCG_COND_EQ, ret, source2, zero, source1, temp); |
| 322 | } |
| 323 | |
| 324 | static bool trans_remu(DisasContext *ctx, arg_remu *a) |
| 325 | { |
| 326 | REQUIRE_EXT(ctx, RVM); |
| 327 | return gen_arith(ctx, a, EXT_ZERO, gen_remu, gen_remu_i128); |
| 328 | } |
| 329 | |
| 330 | static bool trans_mulw(DisasContext *ctx, arg_mulw *a) |
| 331 | { |
| 332 | REQUIRE_64_OR_128BIT(ctx); |
| 333 | REQUIRE_M_OR_ZMMUL(ctx); |
| 334 | ctx->ol = MXL_RV32; |
| 335 | return gen_arith(ctx, a, EXT_NONE, tcg_gen_mul_tl, NULL); |
| 336 | } |
| 337 | |
| 338 | static bool trans_divw(DisasContext *ctx, arg_divw *a) |
| 339 | { |
| 340 | REQUIRE_64_OR_128BIT(ctx); |
| 341 | REQUIRE_EXT(ctx, RVM); |
| 342 | ctx->ol = MXL_RV32; |
| 343 | return gen_arith(ctx, a, EXT_SIGN, gen_div, NULL); |
| 344 | } |
| 345 | |
| 346 | static bool trans_divuw(DisasContext *ctx, arg_divuw *a) |
| 347 | { |
| 348 | REQUIRE_64_OR_128BIT(ctx); |
| 349 | REQUIRE_EXT(ctx, RVM); |
| 350 | ctx->ol = MXL_RV32; |
| 351 | return gen_arith(ctx, a, EXT_ZERO, gen_divu, NULL); |
| 352 | } |
| 353 | |
| 354 | static bool trans_remw(DisasContext *ctx, arg_remw *a) |
| 355 | { |
| 356 | REQUIRE_64_OR_128BIT(ctx); |
| 357 | REQUIRE_EXT(ctx, RVM); |
| 358 | ctx->ol = MXL_RV32; |
| 359 | return gen_arith(ctx, a, EXT_SIGN, gen_rem, NULL); |
| 360 | } |
| 361 | |
| 362 | static bool trans_remuw(DisasContext *ctx, arg_remuw *a) |
| 363 | { |
| 364 | REQUIRE_64_OR_128BIT(ctx); |
| 365 | REQUIRE_EXT(ctx, RVM); |
| 366 | ctx->ol = MXL_RV32; |
| 367 | return gen_arith(ctx, a, EXT_ZERO, gen_remu, NULL); |
| 368 | } |
| 369 | |
| 370 | static bool trans_muld(DisasContext *ctx, arg_muld *a) |
| 371 | { |
| 372 | REQUIRE_128BIT(ctx); |
| 373 | REQUIRE_M_OR_ZMMUL(ctx); |
| 374 | ctx->ol = MXL_RV64; |
| 375 | return gen_arith(ctx, a, EXT_SIGN, tcg_gen_mul_tl, NULL); |
| 376 | } |
| 377 | |
| 378 | static bool trans_divd(DisasContext *ctx, arg_divd *a) |
| 379 | { |
| 380 | REQUIRE_128BIT(ctx); |
| 381 | REQUIRE_EXT(ctx, RVM); |
| 382 | ctx->ol = MXL_RV64; |
| 383 | return gen_arith(ctx, a, EXT_SIGN, gen_div, NULL); |
| 384 | } |
| 385 | |
| 386 | static bool trans_divud(DisasContext *ctx, arg_divud *a) |
| 387 | { |
| 388 | REQUIRE_128BIT(ctx); |
| 389 | REQUIRE_EXT(ctx, RVM); |
| 390 | ctx->ol = MXL_RV64; |
| 391 | return gen_arith(ctx, a, EXT_ZERO, gen_divu, NULL); |
| 392 | } |
| 393 | |
| 394 | static bool trans_remd(DisasContext *ctx, arg_remd *a) |
| 395 | { |
| 396 | REQUIRE_128BIT(ctx); |
| 397 | REQUIRE_EXT(ctx, RVM); |
| 398 | ctx->ol = MXL_RV64; |
| 399 | return gen_arith(ctx, a, EXT_SIGN, gen_rem, NULL); |
| 400 | } |
| 401 | |
| 402 | static bool trans_remud(DisasContext *ctx, arg_remud *a) |
| 403 | { |
| 404 | REQUIRE_128BIT(ctx); |
| 405 | REQUIRE_EXT(ctx, RVM); |
| 406 | ctx->ol = MXL_RV64; |
| 407 | return gen_arith(ctx, a, EXT_ZERO, gen_remu, NULL); |
| 408 | } |