@samitouri / QOSamiQemu / commits / 0182fcd813

fpu: Add saturate parameter to parts_uncanon

The OCP FP8 conversion operations have a parameter to control saturate vs overflow. Add a parameter, currently always false. Reviewed-by: Chao Liu <chao.liu.zevorn@gmail.com> Signed-off-by: Max Chou <max.chou@sifive.com> [rth: Split out of a larger patch] Signed-off-by: Richard Henderson <richard.henderson@linaro.org>

Max Chou committed Feb 5, 2026 at 16:29 UTC 0182fcd813244def266e9ca4a274d565791b7fe1
2 files changed +32 -29
fpu/softfloat-parts.c.inc
+9 -6
@@ -258,9 +258,12 @@ static void partsN(canonicalize)(FloatPartsN *p, float_status *status,
258 * are FRAC_SHIFT bits that may require rounding at the bottom of the
259 * fraction; these bits will be removed. The exponent will be biased
260 * by EXP_BIAS and must be bounded by [EXP_MAX-1, 0].
261 + *
262 + * The saturate parameter controls saturation behavior for formats that
263 + * support it -- when true, overflow produces max normal instead of infinity.
264 */
265 static void partsN(uncanon_normal)(FloatPartsN *p, float_status *s,
263 - const FloatFmt *fmt)
266 + const FloatFmt *fmt, bool saturate)
267 {
268 const int exp_max = fmt->exp_max;
269 const int frac_shift = fmt->frac_shift;
@@ -269,7 +272,7 @@ static void partsN(uncanon_normal)(FloatPartsN *p, float_status *s,
272 const uint64_t frac_lsbm1 = round_mask ^ (round_mask >> 1);
273 const uint64_t roundeven_mask = round_mask | frac_lsb;
274 uint64_t inc;
272 - bool overflow_norm = false;
275 + bool overflow_norm = saturate;
276 int exp, flags = 0;
277
278 switch (s->float_rounding_mode) {
@@ -294,11 +297,11 @@ static void partsN(uncanon_normal)(FloatPartsN *p, float_status *s,
297 break;
298 case float_round_up:
299 inc = p->sign ? 0 : round_mask;
297 - overflow_norm = p->sign;
300 + overflow_norm |= p->sign;
301 break;
302 case float_round_down:
303 inc = p->sign ? round_mask : 0;
301 - overflow_norm = !p->sign;
304 + overflow_norm |= !p->sign;
305 break;
306 case float_round_to_odd:
307 overflow_norm = true;
@@ -445,10 +448,10 @@ static void partsN(uncanon_normal)(FloatPartsN *p, float_status *s,
448 }
449
450 static void partsN(uncanon)(FloatPartsN *p, float_status *s,
448 - const FloatFmt *fmt)
451 + const FloatFmt *fmt, bool saturate)
452 {
453 if (likely(is_anynorm(p->cls))) {
451 - parts_uncanon_normal(p, s, fmt);
454 + parts_uncanon_normal(p, s, fmt, saturate);
455 } else {
456 switch (p->cls) {
457 case float_class_zero:
fpu/softfloat.c
+23 -23
@@ -771,20 +771,20 @@ static void parts128_canonicalize(FloatParts128 *p, float_status *status,
771 PARTS_GENERIC_64_128(canonicalize, A)(A, S, F)
772
773 static void parts64_uncanon_normal(FloatParts64 *p, float_status *status,
774 - const FloatFmt *fmt);
774 + const FloatFmt *fmt, bool saturate);
775 static void parts128_uncanon_normal(FloatParts128 *p, float_status *status,
776 - const FloatFmt *fmt);
776 + const FloatFmt *fmt, bool saturate);
777
778 -#define parts_uncanon_normal(A, S, F) \
779 - PARTS_GENERIC_64_128(uncanon_normal, A)(A, S, F)
778 +#define parts_uncanon_normal(A, S, F, X) \
779 + PARTS_GENERIC_64_128(uncanon_normal, A)(A, S, F, X)
780
781 static void parts64_uncanon(FloatParts64 *p, float_status *status,
782 - const FloatFmt *fmt);
782 + const FloatFmt *fmt, bool saturate);
783 static void parts128_uncanon(FloatParts128 *p, float_status *status,
784 - const FloatFmt *fmt);
784 + const FloatFmt *fmt, bool saturate);
785
786 -#define parts_uncanon(A, S, F) \
787 - PARTS_GENERIC_64_128(uncanon, A)(A, S, F)
786 +#define parts_uncanon(A, S, F, X) \
787 + PARTS_GENERIC_64_128(uncanon, A)(A, S, F, X)
788
789 static void parts64_add_normal(FloatParts64 *a, FloatParts64 *b);
790 static void parts128_add_normal(FloatParts128 *a, FloatParts128 *b);
@@ -1699,7 +1699,7 @@ static float16 float16a_round_pack_canonical(FloatParts64 *p,
1699 float_status *s,
1700 const FloatFmt *params)
1701 {
1702 - parts_uncanon(p, s, params);
1702 + parts_uncanon(p, s, params, false);
1703 return float16_pack_raw(p);
1704 }
1705
@@ -1712,7 +1712,7 @@ static float16 float16_round_pack_canonical(FloatParts64 *p,
1712 static bfloat16 bfloat16_round_pack_canonical(FloatParts64 *p,
1713 float_status *s)
1714 {
1715 - parts_uncanon(p, s, &bfloat16_params);
1715 + parts_uncanon(p, s, &bfloat16_params, false);
1716 return bfloat16_pack_raw(p);
1717 }
1718
@@ -1726,7 +1726,7 @@ static void float32_unpack_canonical(FloatParts64 *p, float32 f,
1726 static float32 float32_round_pack_canonical(FloatParts64 *p,
1727 float_status *s)
1728 {
1729 - parts_uncanon(p, s, &float32_params);
1729 + parts_uncanon(p, s, &float32_params, false);
1730 return float32_pack_raw(p);
1731 }
1732
@@ -1740,7 +1740,7 @@ static void float64_unpack_canonical(FloatParts64 *p, float64 f,
1740 static float64 float64_round_pack_canonical(FloatParts64 *p,
1741 float_status *s)
1742 {
1743 - parts_uncanon(p, s, &float64_params);
1743 + parts_uncanon(p, s, &float64_params, false);
1744 return float64_pack_raw(p);
1745 }
1746
@@ -1789,7 +1789,7 @@ static float64 float64r32_pack_raw(FloatParts64 *p)
1789 static float64 float64r32_round_pack_canonical(FloatParts64 *p,
1790 float_status *s)
1791 {
1792 - parts_uncanon(p, s, &float32_params);
1792 + parts_uncanon(p, s, &float32_params, false);
1793 return float64r32_pack_raw(p);
1794 }
1795
@@ -1803,7 +1803,7 @@ static void float128_unpack_canonical(FloatParts128 *p, float128 f,
1803 static float128 float128_round_pack_canonical(FloatParts128 *p,
1804 float_status *s)
1805 {
1806 - parts_uncanon(p, s, &float128_params);
1806 + parts_uncanon(p, s, &float128_params, false);
1807 return float128_pack_raw(p);
1808 }
1809
@@ -1851,7 +1851,7 @@ static floatx80 floatx80_round_pack_canonical(FloatParts128 *p,
1851 case float_class_normal:
1852 case float_class_denormal:
1853 if (s->floatx80_rounding_precision == floatx80_precision_x) {
1854 - parts_uncanon_normal(p, s, fmt);
1854 + parts_uncanon_normal(p, s, fmt, false);
1855 frac = p->frac_hi;
1856 exp = p->exp;
1857 } else {
@@ -1860,7 +1860,7 @@ static floatx80 floatx80_round_pack_canonical(FloatParts128 *p,
1860 p64.sign = p->sign;
1861 p64.exp = p->exp;
1862 frac_truncjam(&p64, p);
1863 - parts_uncanon_normal(&p64, s, fmt);
1863 + parts_uncanon_normal(&p64, s, fmt, false);
1864 frac = p64.frac;
1865 exp = p64.exp;
1866 }
@@ -2258,7 +2258,7 @@ float16_muladd_scalbn(float16 a, float16 b, float16 c,
2258 pr = parts_muladd_scalbn(&pa, &pb, &pc, scale, flags, status);
2259
2260 /* Round before applying negate result. */
2261 - parts_uncanon(pr, status, &float16_params);
2261 + parts_uncanon(pr, status, &float16_params, false);
2262 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2263 pr->sign ^= 1;
2264 }
@@ -2283,7 +2283,7 @@ float32_muladd_scalbn(float32 a, float32 b, float32 c,
2283 pr = parts_muladd_scalbn(&pa, &pb, &pc, scale, flags, status);
2284
2285 /* Round before applying negate result. */
2286 - parts_uncanon(pr, status, &float32_params);
2286 + parts_uncanon(pr, status, &float32_params, false);
2287 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2288 pr->sign ^= 1;
2289 }
@@ -2302,7 +2302,7 @@ float64_muladd_scalbn(float64 a, float64 b, float64 c,
2302 pr = parts_muladd_scalbn(&pa, &pb, &pc, scale, flags, status);
2303
2304 /* Round before applying negate result. */
2305 - parts_uncanon(pr, status, &float64_params);
2305 + parts_uncanon(pr, status, &float64_params, false);
2306 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2307 pr->sign ^= 1;
2308 }
@@ -2461,7 +2461,7 @@ float64 float64r32_muladd(float64 a, float64 b, float64 c,
2461 pr = parts_muladd_scalbn(&pa, &pb, &pc, 0, flags, status);
2462
2463 /* Round before applying negate result. */
2464 - parts_uncanon(pr, status, &float32_params);
2464 + parts_uncanon(pr, status, &float32_params, false);
2465 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2466 pr->sign ^= 1;
2467 }
@@ -2479,7 +2479,7 @@ bfloat16 QEMU_FLATTEN bfloat16_muladd(bfloat16 a, bfloat16 b, bfloat16 c,
2479 pr = parts_muladd_scalbn(&pa, &pb, &pc, 0, flags, status);
2480
2481 /* Round before applying negate result. */
2482 - parts_uncanon(pr, status, &bfloat16_params);
2482 + parts_uncanon(pr, status, &bfloat16_params, false);
2483 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2484 pr->sign ^= 1;
2485 }
@@ -2497,7 +2497,7 @@ float128 QEMU_FLATTEN float128_muladd(float128 a, float128 b, float128 c,
2497 pr = parts_muladd_scalbn(&pa, &pb, &pc, 0, flags, status);
2498
2499 /* Round before applying negate result. */
2500 - parts_uncanon(pr, status, &float128_params);
2500 + parts_uncanon(pr, status, &float128_params, false);
2501 if ((flags & float_muladd_negate_result) && !is_nan(pr->cls)) {
2502 pr->sign ^= 1;
2503 }
@@ -5435,7 +5435,7 @@ static void parts_s390_divide_to_integer(FloatParts64 *a, FloatParts64 *b,
5435 /* Round remainder to the target format */
5436 *r = *r_precise;
5437 status->float_exception_flags = 0;
5438 - parts_uncanon(r, status, fmt);
5438 + parts_uncanon(r, status, fmt, false);
5439 r_flags = status->float_exception_flags;
5440 r->frac &= (1ULL << fmt->frac_size) - 1;
5441 parts_canonicalize(r, status, fmt);