|
|
1.1 ! root 1: ! 2: /*============================================================================ ! 3: ! 4: This C source file is part of the SoftFloat IEC/IEEE Floating-point Arithmetic ! 5: Package, Release 2b. ! 6: ! 7: Written by John R. Hauser. This work was made possible in part by the ! 8: International Computer Science Institute, located at Suite 600, 1947 Center ! 9: Street, Berkeley, California 94704. Funding was partially provided by the ! 10: National Science Foundation under grant MIP-9311980. The original version ! 11: of this code was written as part of a project to build a fixed-point vector ! 12: processor in collaboration with the University of California at Berkeley, ! 13: overseen by Profs. Nelson Morgan and John Wawrzynek. More information ! 14: is available through the Web page `http://www.cs.berkeley.edu/~jhauser/ ! 15: arithmetic/SoftFloat.html'. ! 16: ! 17: THIS SOFTWARE IS DISTRIBUTED AS IS, FOR FREE. Although reasonable effort has ! 18: been made to avoid it, THIS SOFTWARE MAY CONTAIN FAULTS THAT WILL AT TIMES ! 19: RESULT IN INCORRECT BEHAVIOR. USE OF THIS SOFTWARE IS RESTRICTED TO PERSONS ! 20: AND ORGANIZATIONS WHO CAN AND WILL TAKE FULL RESPONSIBILITY FOR ALL LOSSES, ! 21: COSTS, OR OTHER PROBLEMS THEY INCUR DUE TO THE SOFTWARE, AND WHO FURTHERMORE ! 22: EFFECTIVELY INDEMNIFY JOHN HAUSER AND THE INTERNATIONAL COMPUTER SCIENCE ! 23: INSTITUTE (possibly via similar legal warning) AGAINST ALL LOSSES, COSTS, OR ! 24: OTHER PROBLEMS INCURRED BY THEIR CUSTOMERS AND CLIENTS DUE TO THE SOFTWARE. ! 25: ! 26: Derivative works are acceptable, even for commercial purposes, so long as ! 27: (1) the source code for the derivative work includes prominent notice that ! 28: the work is derivative, and (2) the source code includes prominent notice with ! 29: these four paragraphs for those parts of this code that are retained. ! 30: ! 31: =============================================================================*/ ! 32: ! 33: #include "milieu.h" ! 34: #include "softfloat.h" ! 35: ! 36: /*---------------------------------------------------------------------------- ! 37: | Floating-point rounding mode, extended double-precision rounding precision, ! 38: | and exception flags. ! 39: *----------------------------------------------------------------------------*/ ! 40: int8 float_exception_flags = 0; ! 41: #ifdef SOFTFLOAT_I860 ! 42: int8 float_exception_flags2 = 0; ! 43: #endif ! 44: #ifdef FLOATX80 ! 45: int8 floatx80_rounding_precision = 80; ! 46: #endif ! 47: ! 48: int8 float_rounding_mode = float_round_nearest_even; ! 49: #ifdef SOFTFLOAT_I860 ! 50: int8 float_rounding_mode2 = float_round_nearest_even; ! 51: #endif ! 52: ! 53: /*---------------------------------------------------------------------------- ! 54: | Variables for storing sign, exponent and significand of internal extended ! 55: | double-precision floating-point value for external use. ! 56: *----------------------------------------------------------------------------*/ ! 57: flag floatx80_internal_sign = 0; ! 58: int32 floatx80_internal_exp = 0; ! 59: bits64 floatx80_internal_sig0 = 0; ! 60: bits64 floatx80_internal_sig1 = 0; ! 61: int8 floatx80_internal_precision = 80; ! 62: int8 floatx80_internal_mode = float_round_nearest_even; ! 63: ! 64: /*---------------------------------------------------------------------------- ! 65: | Functions for storing sign, exponent and significand of extended ! 66: | double-precision floating-point intermediate result for external use. ! 67: *----------------------------------------------------------------------------*/ ! 68: void saveFloatx80Internal( int8 prec, flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ) ! 69: { ! 70: floatx80_internal_sign = zSign; ! 71: floatx80_internal_exp = zExp; ! 72: floatx80_internal_sig0 = zSig0; ! 73: floatx80_internal_sig1 = zSig1; ! 74: floatx80_internal_precision = prec; ! 75: floatx80_internal_mode = float_rounding_mode; ! 76: ! 77: } ! 78: ! 79: void saveFloat64Internal( flag zSign, int16 zExp, bits64 zSig ) ! 80: { ! 81: floatx80_internal_sign = zSign; ! 82: floatx80_internal_exp = zExp + 0x3C01; ! 83: floatx80_internal_sig0 = zSig<<1; ! 84: floatx80_internal_sig1 = 0; ! 85: floatx80_internal_precision = 64; ! 86: floatx80_internal_mode = float_rounding_mode; ! 87: ! 88: } ! 89: ! 90: void saveFloat32Internal( flag zSign, int16 zExp, bits32 zSig ) ! 91: { ! 92: floatx80_internal_sign = zSign; ! 93: floatx80_internal_exp = zExp + 0x3F81; ! 94: floatx80_internal_sig0 = ( (bits64) zSig )<<33; ! 95: floatx80_internal_sig1 = 0; ! 96: floatx80_internal_precision = 32; ! 97: floatx80_internal_mode = float_rounding_mode; ! 98: ! 99: } ! 100: ! 101: ! 102: /*---------------------------------------------------------------------------- ! 103: | Functions for returning sign, exponent and significand of extended ! 104: | double-precision floating-point intermediate result for external use. ! 105: *----------------------------------------------------------------------------*/ ! 106: ! 107: void getRoundedFloatInternal( int8 roundingPrecision, flag *pzSign, int32 *pzExp, bits64 *pzSig ) ! 108: { ! 109: int64 roundIncrement, roundMask, roundBits; ! 110: flag increment; ! 111: ! 112: flag zSign = floatx80_internal_sign; ! 113: int32 zExp = floatx80_internal_exp; ! 114: bits64 zSig0 = floatx80_internal_sig0; ! 115: bits64 zSig1 = floatx80_internal_sig1; ! 116: ! 117: if ( roundingPrecision == 80 ) { ! 118: goto precision80; ! 119: } else if ( roundingPrecision == 64 ) { ! 120: roundIncrement = LIT64( 0x0000000000000400 ); ! 121: roundMask = LIT64( 0x00000000000007FF ); ! 122: } else if ( roundingPrecision == 32 ) { ! 123: roundIncrement = LIT64( 0x0000008000000000 ); ! 124: roundMask = LIT64( 0x000000FFFFFFFFFF ); ! 125: } else { ! 126: goto precision80; ! 127: } ! 128: ! 129: zSig0 |= ( zSig1 != 0 ); ! 130: if ( floatx80_internal_mode != float_round_nearest_even ) { ! 131: if ( floatx80_internal_mode == float_round_to_zero ) { ! 132: roundIncrement = 0; ! 133: } else { ! 134: roundIncrement = roundMask; ! 135: if ( zSign ) { ! 136: if ( floatx80_internal_mode == float_round_up ) roundIncrement = 0; ! 137: } else { ! 138: if ( floatx80_internal_mode == float_round_down ) roundIncrement = 0; ! 139: } ! 140: } ! 141: } ! 142: ! 143: roundBits = zSig0 & roundMask; ! 144: ! 145: zSig0 += roundIncrement; ! 146: if ( zSig0 < roundIncrement ) { ! 147: ++zExp; ! 148: zSig0 = LIT64( 0x8000000000000000 ); ! 149: } ! 150: roundIncrement = roundMask + 1; ! 151: if ( floatx80_internal_mode == float_round_nearest_even && ( roundBits<<1 == roundIncrement ) ) { ! 152: roundMask |= roundIncrement; ! 153: } ! 154: zSig0 &= ~ roundMask; ! 155: if ( zSig0 == 0 ) zExp = 0; ! 156: ! 157: *pzSign = zSign; ! 158: *pzExp = zExp; ! 159: *pzSig = zSig0; ! 160: return; ! 161: ! 162: precision80: ! 163: increment = ( (sbits64) zSig1 < 0 ); ! 164: if ( floatx80_internal_mode != float_round_nearest_even ) { ! 165: if ( floatx80_internal_mode == float_round_to_zero ) { ! 166: increment = 0; ! 167: } else { ! 168: if ( zSign ) { ! 169: increment = ( floatx80_internal_mode == float_round_down ) && zSig1; ! 170: } else { ! 171: increment = ( floatx80_internal_mode == float_round_up ) && zSig1; ! 172: } ! 173: } ! 174: } ! 175: if ( increment ) { ! 176: ++zSig0; ! 177: if ( zSig0 == 0 ) { ! 178: ++zExp; ! 179: zSig0 = LIT64( 0x8000000000000000 ); ! 180: } else { ! 181: zSig0 &= ~ ( ( (bits64) ( zSig1<<1 ) == 0 ) & ( floatx80_internal_mode == float_round_nearest_even ) ); ! 182: } ! 183: } else { ! 184: if ( zSig0 == 0 ) zExp = 0; ! 185: } ! 186: ! 187: *pzSign = zSign; ! 188: *pzExp = zExp; ! 189: *pzSig = zSig0; ! 190: return; ! 191: ! 192: } ! 193: ! 194: floatx80 getFloatInternalOverflow( void ) ! 195: { ! 196: flag zSign; ! 197: int32 zExp; ! 198: bits64 zSig; ! 199: ! 200: getRoundedFloatInternal( floatx80_internal_precision, &zSign, &zExp, &zSig ); ! 201: ! 202: if (zExp > (0x7fff + 0x6000)) { // catastrophic ! 203: zExp = 0; ! 204: } else { ! 205: zExp -= 0x6000; ! 206: } ! 207: ! 208: return packFloatx80( zSign, zExp, zSig ); ! 209: ! 210: } ! 211: ! 212: floatx80 getFloatInternalUnderflow( void ) ! 213: { ! 214: flag zSign; ! 215: int32 zExp; ! 216: bits64 zSig; ! 217: ! 218: getRoundedFloatInternal( floatx80_internal_precision, &zSign, &zExp, &zSig ); ! 219: ! 220: if (zExp < (0x0000 - 0x6000)) { // catastrophic ! 221: zExp = 0; ! 222: } else { ! 223: zExp += 0x6000; ! 224: } ! 225: ! 226: return packFloatx80( zSign, zExp, zSig ); ! 227: ! 228: } ! 229: ! 230: floatx80 getFloatInternalRoundedAll( void ) ! 231: { ! 232: flag zSign; ! 233: int32 zExp; ! 234: bits64 zSig, zSig32, zSig64, zSig80; ! 235: ! 236: if (floatx80_internal_precision == 80) { ! 237: getRoundedFloatInternal( 80, &zSign, &zExp, &zSig80 ); ! 238: zSig = zSig80; ! 239: } else if (floatx80_internal_precision == 64) { ! 240: getRoundedFloatInternal( 80, &zSign, &zExp, &zSig80 ); ! 241: getRoundedFloatInternal( 64, &zSign, &zExp, &zSig64 ); ! 242: zSig = zSig64; ! 243: zSig |= zSig80 & LIT64( 0x00000000000007FF ); ! 244: } else { ! 245: getRoundedFloatInternal( 80, &zSign, &zExp, &zSig80 ); ! 246: getRoundedFloatInternal( 64, &zSign, &zExp, &zSig64 ); ! 247: getRoundedFloatInternal( 32, &zSign, &zExp, &zSig32 ); ! 248: zSig = zSig32; ! 249: zSig |= zSig64 & LIT64( 0x000000FFFFFFFFFF ); ! 250: zSig |= zSig80 & LIT64( 0x00000000000007FF ); ! 251: } ! 252: ! 253: return packFloatx80( zSign, zExp & 0x7FFF, zSig ); ! 254: ! 255: } ! 256: ! 257: floatx80 getFloatInternalRoundedSome( void ) ! 258: { ! 259: flag zSign; ! 260: int32 zExp; ! 261: bits64 zSig, zSig32, zSig64, zSig80; ! 262: ! 263: if (floatx80_internal_precision == 80) { ! 264: getRoundedFloatInternal( 80, &zSign, &zExp, &zSig80 ); ! 265: zSig = zSig80; ! 266: } else if (floatx80_internal_precision == 64) { ! 267: getRoundedFloatInternal( 64, &zSign, &zExp, &zSig64 ); ! 268: zSig80 = floatx80_internal_sig0; ! 269: if (zSig64 != (zSig80 & LIT64( 0xFFFFFFFFFFFFF800 ))) { ! 270: zSig80++; ! 271: } ! 272: zSig = zSig64; ! 273: zSig |= zSig80 & LIT64( 0x00000000000007FF ); ! 274: } else { ! 275: getRoundedFloatInternal( 32, &zSign, &zExp, &zSig32 ); ! 276: zSig80 = floatx80_internal_sig0; ! 277: if (zSig32 != (zSig80 & LIT64( 0xFFFFFF0000000000 ))) { ! 278: zSig80++; ! 279: } ! 280: zSig = zSig32; ! 281: zSig |= zSig80 & LIT64( 0x000000FFFFFFFFFF ); ! 282: } ! 283: ! 284: return packFloatx80( zSign, zExp & 0x7FFF, zSig ); ! 285: ! 286: } ! 287: ! 288: floatx80 getFloatInternalFloatx80( void ) ! 289: { ! 290: flag zSign; ! 291: int32 zExp; ! 292: bits64 zSig; ! 293: ! 294: getRoundedFloatInternal( 80, &zSign, &zExp, &zSig ); ! 295: ! 296: return packFloatx80( zSign, zExp & 0x7FFF, zSig ); ! 297: ! 298: } ! 299: ! 300: floatx80 getFloatInternalUnrounded( void ) ! 301: { ! 302: flag zSign = floatx80_internal_sign; ! 303: int32 zExp = floatx80_internal_exp; ! 304: bits64 zSig = floatx80_internal_sig0; ! 305: ! 306: return packFloatx80( zSign, zExp & 0x7FFF, zSig ); ! 307: ! 308: } ! 309: ! 310: bits64 getFloatInternalGRS( void ) ! 311: { ! 312: #if 1 ! 313: if (floatx80_internal_sig1) ! 314: return 5; ! 315: ! 316: if (floatx80_internal_precision == 64 && ! 317: floatx80_internal_sig0 & LIT64( 0x00000000000007FF )) { ! 318: return 1; ! 319: } ! 320: if (floatx80_internal_precision == 32 && ! 321: floatx80_internal_sig0 & LIT64( 0x000000FFFFFFFFFF )) { ! 322: return 1; ! 323: } ! 324: ! 325: return 0; ! 326: #else ! 327: bits64 roundbits; ! 328: shift64RightJamming(floatx80_internal_sig1, 61, &roundbits); ! 329: ! 330: return roundbits; ! 331: #endif ! 332: ! 333: } ! 334: ! 335: ! 336: /*---------------------------------------------------------------------------- ! 337: | Functions and definitions to determine: (1) whether tininess for underflow ! 338: | is detected before or after rounding by default, (2) what (if anything) ! 339: | happens when exceptions are raised, (3) how signaling NaNs are distinguished ! 340: | from quiet NaNs, (4) the default generated quiet NaNs, and (5) how NaNs ! 341: | are propagated from function inputs to output. These details are target- ! 342: | specific. ! 343: *----------------------------------------------------------------------------*/ ! 344: #include "softfloat-specialize.h" ! 345: ! 346: /*---------------------------------------------------------------------------- ! 347: | Takes a 64-bit fixed-point value `absZ' with binary point between bits 6 ! 348: | and 7, and returns the properly rounded 32-bit integer corresponding to the ! 349: | input. If `zSign' is 1, the input is negated before being converted to an ! 350: | integer. Bit 63 of `absZ' must be zero. Ordinarily, the fixed-point input ! 351: | is simply rounded to an integer, with the inexact exception raised if the ! 352: | input cannot be represented exactly as an integer. However, if the fixed- ! 353: | point input is too large, the invalid exception is raised and the largest ! 354: | positive or negative integer is returned. ! 355: *----------------------------------------------------------------------------*/ ! 356: ! 357: static int32 roundAndPackInt32( flag zSign, bits64 absZ ) ! 358: { ! 359: int8 roundingMode; ! 360: flag roundNearestEven; ! 361: int8 roundIncrement, roundBits; ! 362: int32 z; ! 363: ! 364: roundingMode = float_rounding_mode; ! 365: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 366: roundIncrement = 0x40; ! 367: if ( ! roundNearestEven ) { ! 368: if ( roundingMode == float_round_to_zero ) { ! 369: roundIncrement = 0; ! 370: } ! 371: else { ! 372: roundIncrement = 0x7F; ! 373: if ( zSign ) { ! 374: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 375: } ! 376: else { ! 377: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 378: } ! 379: } ! 380: } ! 381: roundBits = absZ & 0x7F; ! 382: absZ = ( absZ + roundIncrement )>>7; ! 383: absZ &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 384: z = absZ; ! 385: if ( zSign ) z = - z; ! 386: z = (sbits32) z; ! 387: if ( ( absZ>>32 ) || ( z && ( ( z < 0 ) ^ zSign ) ) ) { ! 388: float_raise( float_flag_invalid ); ! 389: return zSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 390: } ! 391: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 392: return z; ! 393: ! 394: } ! 395: ! 396: #ifdef SOFTFLOAT_68K // 30-01-2017: Added for Previous ! 397: static int16 roundAndPackInt16( flag zSign, bits64 absZ ) ! 398: { ! 399: int8 roundingMode; ! 400: flag roundNearestEven; ! 401: int8 roundIncrement, roundBits; ! 402: int16 z; ! 403: ! 404: roundingMode = float_rounding_mode; ! 405: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 406: roundIncrement = 0x40; ! 407: if ( ! roundNearestEven ) { ! 408: if ( roundingMode == float_round_to_zero ) { ! 409: roundIncrement = 0; ! 410: } ! 411: else { ! 412: roundIncrement = 0x7F; ! 413: if ( zSign ) { ! 414: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 415: } ! 416: else { ! 417: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 418: } ! 419: } ! 420: } ! 421: roundBits = absZ & 0x7F; ! 422: absZ = ( absZ + roundIncrement )>>7; ! 423: absZ &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 424: z = absZ; ! 425: if ( zSign ) z = - z; ! 426: z = (sbits16) z; ! 427: if ( ( absZ>>16 ) || ( z && ( ( z < 0 ) ^ zSign ) ) ) { ! 428: float_raise( float_flag_invalid ); ! 429: return zSign ? (sbits16) 0x8000 : 0x7FFF; ! 430: } ! 431: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 432: return z; ! 433: ! 434: } ! 435: ! 436: static int8 roundAndPackInt8( flag zSign, bits64 absZ ) ! 437: { ! 438: int8 roundingMode; ! 439: flag roundNearestEven; ! 440: int8 roundIncrement, roundBits; ! 441: int8 z; ! 442: ! 443: roundingMode = float_rounding_mode; ! 444: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 445: roundIncrement = 0x40; ! 446: if ( ! roundNearestEven ) { ! 447: if ( roundingMode == float_round_to_zero ) { ! 448: roundIncrement = 0; ! 449: } ! 450: else { ! 451: roundIncrement = 0x7F; ! 452: if ( zSign ) { ! 453: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 454: } ! 455: else { ! 456: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 457: } ! 458: } ! 459: } ! 460: roundBits = absZ & 0x7F; ! 461: absZ = ( absZ + roundIncrement )>>7; ! 462: absZ &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 463: z = absZ; ! 464: if ( zSign ) z = - z; ! 465: z = (sbits8) z; ! 466: if ( ( absZ>>8 ) || ( z && ( ( z < 0 ) ^ zSign ) ) ) { ! 467: float_raise( float_flag_invalid ); ! 468: return zSign ? (sbits8) 0x80 : 0x7F; ! 469: } ! 470: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 471: return z; ! 472: ! 473: } ! 474: #endif // End of addition for Previous ! 475: ! 476: #ifdef SOFTFLOAT_I860 // 29-04-2017: Added for Previous ! 477: static int32 roundAndPackInt32_2( flag zSign, bits64 absZ ) ! 478: { ! 479: int8 roundingMode; ! 480: flag roundNearestEven; ! 481: int8 roundIncrement, roundBits; ! 482: int32 z; ! 483: ! 484: roundingMode = float_rounding_mode2; ! 485: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 486: roundIncrement = 0x40; ! 487: if ( ! roundNearestEven ) { ! 488: if ( roundingMode == float_round_to_zero ) { ! 489: roundIncrement = 0; ! 490: } ! 491: else { ! 492: roundIncrement = 0x7F; ! 493: if ( zSign ) { ! 494: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 495: } ! 496: else { ! 497: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 498: } ! 499: } ! 500: } ! 501: roundBits = absZ & 0x7F; ! 502: absZ = ( absZ + roundIncrement )>>7; ! 503: absZ &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 504: z = absZ; ! 505: if ( zSign ) z = - z; ! 506: z = (sbits32) z; ! 507: if ( ( absZ>>32 ) || ( z && ( ( z < 0 ) ^ zSign ) ) ) { ! 508: float_raise2( float_flag_invalid ); ! 509: return zSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 510: } ! 511: if ( roundBits ) float_raise2( float_flag_inexact); ! 512: return z; ! 513: ! 514: } ! 515: #endif // End of addition for Previous ! 516: ! 517: /*---------------------------------------------------------------------------- ! 518: | Takes the 128-bit fixed-point value formed by concatenating `absZ0' and ! 519: | `absZ1', with binary point between bits 63 and 64 (between the input words), ! 520: | and returns the properly rounded 64-bit integer corresponding to the input. ! 521: | If `zSign' is 1, the input is negated before being converted to an integer. ! 522: | Ordinarily, the fixed-point input is simply rounded to an integer, with ! 523: | the inexact exception raised if the input cannot be represented exactly as ! 524: | an integer. However, if the fixed-point input is too large, the invalid ! 525: | exception is raised and the largest positive or negative integer is ! 526: | returned. ! 527: *----------------------------------------------------------------------------*/ ! 528: ! 529: static int64 roundAndPackInt64( flag zSign, bits64 absZ0, bits64 absZ1 ) ! 530: { ! 531: int8 roundingMode; ! 532: flag roundNearestEven, increment; ! 533: int64 z; ! 534: ! 535: roundingMode = float_rounding_mode; ! 536: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 537: increment = ( (sbits64) absZ1 < 0 ); ! 538: if ( ! roundNearestEven ) { ! 539: if ( roundingMode == float_round_to_zero ) { ! 540: increment = 0; ! 541: } ! 542: else { ! 543: if ( zSign ) { ! 544: increment = ( roundingMode == float_round_down ) && absZ1; ! 545: } ! 546: else { ! 547: increment = ( roundingMode == float_round_up ) && absZ1; ! 548: } ! 549: } ! 550: } ! 551: if ( increment ) { ! 552: ++absZ0; ! 553: if ( absZ0 == 0 ) goto overflow; ! 554: absZ0 &= ~ ( ( (bits64) ( absZ1<<1 ) == 0 ) & roundNearestEven ); ! 555: } ! 556: z = absZ0; ! 557: if ( zSign ) z = - z; ! 558: z = (sbits64) z; ! 559: if ( z && ( ( z < 0 ) ^ zSign ) ) { ! 560: overflow: ! 561: float_raise( float_flag_invalid ); ! 562: return ! 563: zSign ? (sbits64) LIT64( 0x8000000000000000 ) ! 564: : LIT64( 0x7FFFFFFFFFFFFFFF ); ! 565: } ! 566: if ( absZ1 ) float_exception_flags |= float_flag_inexact; ! 567: return z; ! 568: ! 569: } ! 570: ! 571: /*---------------------------------------------------------------------------- ! 572: | Returns the fraction bits of the single-precision floating-point value `a'. ! 573: *----------------------------------------------------------------------------*/ ! 574: ! 575: INLINE bits32 extractFloat32Frac( float32 a ) ! 576: { ! 577: return a & 0x007FFFFF; ! 578: ! 579: } ! 580: ! 581: /*---------------------------------------------------------------------------- ! 582: | Returns the exponent bits of the single-precision floating-point value `a'. ! 583: *----------------------------------------------------------------------------*/ ! 584: ! 585: INLINE int16 extractFloat32Exp( float32 a ) ! 586: { ! 587: return ( a>>23 ) & 0xFF; ! 588: ! 589: } ! 590: ! 591: /*---------------------------------------------------------------------------- ! 592: | Returns the sign bit of the single-precision floating-point value `a'. ! 593: *----------------------------------------------------------------------------*/ ! 594: ! 595: INLINE flag extractFloat32Sign( float32 a ) ! 596: { ! 597: return a>>31; ! 598: ! 599: } ! 600: ! 601: /*---------------------------------------------------------------------------- ! 602: | Normalizes the subnormal single-precision floating-point value represented ! 603: | by the denormalized significand `aSig'. The normalized exponent and ! 604: | significand are stored at the locations pointed to by `zExpPtr' and ! 605: | `zSigPtr', respectively. ! 606: *----------------------------------------------------------------------------*/ ! 607: ! 608: static void ! 609: normalizeFloat32Subnormal( bits32 aSig, int16 *zExpPtr, bits32 *zSigPtr ) ! 610: { ! 611: int8 shiftCount; ! 612: ! 613: shiftCount = countLeadingZeros32( aSig ) - 8; ! 614: *zSigPtr = aSig<<shiftCount; ! 615: *zExpPtr = 1 - shiftCount; ! 616: ! 617: } ! 618: ! 619: /*---------------------------------------------------------------------------- ! 620: | Packs the sign `zSign', exponent `zExp', and significand `zSig' into a ! 621: | single-precision floating-point value, returning the result. After being ! 622: | shifted into the proper positions, the three fields are simply added ! 623: | together to form the result. This means that any integer portion of `zSig' ! 624: | will be added into the exponent. Since a properly normalized significand ! 625: | will have an integer portion equal to 1, the `zExp' input should be 1 less ! 626: | than the desired result exponent whenever `zSig' is a complete, normalized ! 627: | significand. ! 628: *----------------------------------------------------------------------------*/ ! 629: ! 630: INLINE float32 packFloat32( flag zSign, int16 zExp, bits32 zSig ) ! 631: { ! 632: return ( ( (bits32) zSign )<<31 ) + ( ( (bits32) zExp )<<23 ) + zSig; ! 633: ! 634: } ! 635: ! 636: /*---------------------------------------------------------------------------- ! 637: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 638: | and significand `zSig', and returns the proper single-precision floating- ! 639: | point value corresponding to the abstract input. Ordinarily, the abstract ! 640: | value is simply rounded and packed into the single-precision format, with ! 641: | the inexact exception raised if the abstract input cannot be represented ! 642: | exactly. However, if the abstract value is too large, the overflow and ! 643: | inexact exceptions are raised and an infinity or maximal finite value is ! 644: | returned. If the abstract value is too small, the input value is rounded to ! 645: | a subnormal number, and the underflow and inexact exceptions are raised if ! 646: | the abstract input cannot be represented exactly as a subnormal single- ! 647: | precision floating-point number. ! 648: | The input significand `zSig' has its binary point between bits 30 ! 649: | and 29, which is 7 bits to the left of the usual location. This shifted ! 650: | significand must be normalized or smaller. If `zSig' is not normalized, ! 651: | `zExp' must be 0; in that case, the result returned is a subnormal number, ! 652: | and it must not require rounding. In the usual case that `zSig' is ! 653: | normalized, `zExp' must be 1 less than the ``true'' floating-point exponent. ! 654: | The handling of underflow and overflow follows the IEC/IEEE Standard for ! 655: | Binary Floating-Point Arithmetic. ! 656: *----------------------------------------------------------------------------*/ ! 657: ! 658: static float32 roundAndPackFloat32( flag zSign, int16 zExp, bits32 zSig ) ! 659: { ! 660: int8 roundingMode; ! 661: flag roundNearestEven; ! 662: int8 roundIncrement, roundBits; ! 663: flag isTiny; ! 664: ! 665: roundingMode = float_rounding_mode; ! 666: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 667: roundIncrement = 0x40; ! 668: if ( ! roundNearestEven ) { ! 669: if ( roundingMode == float_round_to_zero ) { ! 670: roundIncrement = 0; ! 671: } ! 672: else { ! 673: roundIncrement = 0x7F; ! 674: if ( zSign ) { ! 675: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 676: } ! 677: else { ! 678: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 679: } ! 680: } ! 681: } ! 682: roundBits = zSig & 0x7F; ! 683: if ( 0xFD <= (bits16) zExp ) { ! 684: if ( ( 0xFD < zExp ) ! 685: || ( ( zExp == 0xFD ) ! 686: && ( (sbits32) ( zSig + roundIncrement ) < 0 ) ) ! 687: ) { ! 688: #ifdef SOFTFLOAT_68K ! 689: float_raise( float_flag_overflow ); ! 690: saveFloat32Internal( zSign, zExp, zSig ); ! 691: if ( roundBits ) float_raise( float_flag_inexact ); ! 692: #else ! 693: float_raise( float_flag_overflow | float_flag_inexact ); ! 694: #endif ! 695: return packFloat32( zSign, 0xFF, 0 ) - ( roundIncrement == 0 ); ! 696: } ! 697: if ( zExp < 0 ) { ! 698: isTiny = ! 699: ( float_detect_tininess == float_tininess_before_rounding ) ! 700: || ( zExp < -1 ) ! 701: || ( zSig + roundIncrement < 0x80000000 ); ! 702: #ifdef SOFTFLOAT_68K ! 703: if ( isTiny ) { ! 704: float_raise( float_flag_underflow ); ! 705: saveFloat32Internal( zSign, zExp, zSig ); ! 706: } ! 707: #endif ! 708: shift32RightJamming( zSig, - zExp, &zSig ); ! 709: zExp = 0; ! 710: roundBits = zSig & 0x7F; ! 711: #ifndef SOFTFLOAT_68K ! 712: if ( isTiny && roundBits ) float_raise( float_flag_underflow ); ! 713: #endif ! 714: } ! 715: } ! 716: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 717: zSig = ( zSig + roundIncrement )>>7; ! 718: zSig &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 719: if ( zSig == 0 ) zExp = 0; ! 720: return packFloat32( zSign, zExp, zSig ); ! 721: ! 722: } ! 723: ! 724: #ifdef SOFTFLOAT_I860 // 29-04-2017: Added for Previous ! 725: static float32 roundAndPackFloat32_2( flag zSign, int16 zExp, bits32 zSig ) ! 726: { ! 727: int8 roundingMode; ! 728: flag roundNearestEven; ! 729: int8 roundIncrement, roundBits; ! 730: flag isTiny; ! 731: ! 732: roundingMode = float_rounding_mode2; ! 733: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 734: roundIncrement = 0x40; ! 735: if ( ! roundNearestEven ) { ! 736: if ( roundingMode == float_round_to_zero ) { ! 737: roundIncrement = 0; ! 738: } ! 739: else { ! 740: roundIncrement = 0x7F; ! 741: if ( zSign ) { ! 742: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 743: } ! 744: else { ! 745: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 746: } ! 747: } ! 748: } ! 749: roundBits = zSig & 0x7F; ! 750: if ( 0xFD <= (bits16) zExp ) { ! 751: if ( ( 0xFD < zExp ) ! 752: || ( ( zExp == 0xFD ) ! 753: && ( (sbits32) ( zSig + roundIncrement ) < 0 ) ) ! 754: ) { ! 755: float_raise2( float_flag_overflow | float_flag_inexact ); ! 756: return packFloat32( zSign, 0xFF, 0 ) - ( roundIncrement == 0 ); ! 757: } ! 758: if ( zExp < 0 ) { ! 759: isTiny = ! 760: #ifndef SOFTFLOAT_I860 ! 761: ( float_detect_tininess == float_tininess_before_rounding ) || ! 762: #endif ! 763: ( zExp < -1 ) || ! 764: ( zSig + roundIncrement < 0x80000000 ); ! 765: shift32RightJamming( zSig, - zExp, &zSig ); ! 766: zExp = 0; ! 767: roundBits = zSig & 0x7F; ! 768: if ( isTiny && roundBits ) float_raise2( float_flag_underflow ); ! 769: } ! 770: } ! 771: if ( roundBits ) float_raise2( float_flag_inexact ); ! 772: zSig = ( zSig + roundIncrement )>>7; ! 773: zSig &= ~ ( ( ( roundBits ^ 0x40 ) == 0 ) & roundNearestEven ); ! 774: if ( zSig == 0 ) zExp = 0; ! 775: return packFloat32( zSign, zExp, zSig ); ! 776: ! 777: } ! 778: #endif // end of addition for Previous ! 779: ! 780: /*---------------------------------------------------------------------------- ! 781: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 782: | and significand `zSig', and returns the proper single-precision floating- ! 783: | point value corresponding to the abstract input. This routine is just like ! 784: | `roundAndPackFloat32' except that `zSig' does not have to be normalized. ! 785: | Bit 31 of `zSig' must be zero, and `zExp' must be 1 less than the ``true'' ! 786: | floating-point exponent. ! 787: *----------------------------------------------------------------------------*/ ! 788: ! 789: static float32 ! 790: normalizeRoundAndPackFloat32( flag zSign, int16 zExp, bits32 zSig ) ! 791: { ! 792: int8 shiftCount; ! 793: ! 794: shiftCount = countLeadingZeros32( zSig ) - 1; ! 795: return roundAndPackFloat32( zSign, zExp - shiftCount, zSig<<shiftCount ); ! 796: ! 797: } ! 798: ! 799: /*---------------------------------------------------------------------------- ! 800: | Returns the fraction bits of the double-precision floating-point value `a'. ! 801: *----------------------------------------------------------------------------*/ ! 802: ! 803: INLINE bits64 extractFloat64Frac( float64 a ) ! 804: { ! 805: return a & LIT64( 0x000FFFFFFFFFFFFF ); ! 806: ! 807: } ! 808: ! 809: /*---------------------------------------------------------------------------- ! 810: | Returns the exponent bits of the double-precision floating-point value `a'. ! 811: *----------------------------------------------------------------------------*/ ! 812: ! 813: INLINE int16 extractFloat64Exp( float64 a ) ! 814: { ! 815: return ( a>>52 ) & 0x7FF; ! 816: ! 817: } ! 818: ! 819: /*---------------------------------------------------------------------------- ! 820: | Returns the sign bit of the double-precision floating-point value `a'. ! 821: *----------------------------------------------------------------------------*/ ! 822: ! 823: INLINE flag extractFloat64Sign( float64 a ) ! 824: { ! 825: return a>>63; ! 826: ! 827: } ! 828: ! 829: /*---------------------------------------------------------------------------- ! 830: | Normalizes the subnormal double-precision floating-point value represented ! 831: | by the denormalized significand `aSig'. The normalized exponent and ! 832: | significand are stored at the locations pointed to by `zExpPtr' and ! 833: | `zSigPtr', respectively. ! 834: *----------------------------------------------------------------------------*/ ! 835: ! 836: static void ! 837: normalizeFloat64Subnormal( bits64 aSig, int16 *zExpPtr, bits64 *zSigPtr ) ! 838: { ! 839: int8 shiftCount; ! 840: ! 841: shiftCount = countLeadingZeros64( aSig ) - 11; ! 842: *zSigPtr = aSig<<shiftCount; ! 843: *zExpPtr = 1 - shiftCount; ! 844: ! 845: } ! 846: ! 847: /*---------------------------------------------------------------------------- ! 848: | Packs the sign `zSign', exponent `zExp', and significand `zSig' into a ! 849: | double-precision floating-point value, returning the result. After being ! 850: | shifted into the proper positions, the three fields are simply added ! 851: | together to form the result. This means that any integer portion of `zSig' ! 852: | will be added into the exponent. Since a properly normalized significand ! 853: | will have an integer portion equal to 1, the `zExp' input should be 1 less ! 854: | than the desired result exponent whenever `zSig' is a complete, normalized ! 855: | significand. ! 856: *----------------------------------------------------------------------------*/ ! 857: ! 858: INLINE float64 packFloat64( flag zSign, int16 zExp, bits64 zSig ) ! 859: { ! 860: return ( ( (bits64) zSign )<<63 ) + ( ( (bits64) zExp )<<52 ) + zSig; ! 861: ! 862: } ! 863: ! 864: /*---------------------------------------------------------------------------- ! 865: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 866: | and significand `zSig', and returns the proper double-precision floating- ! 867: | point value corresponding to the abstract input. Ordinarily, the abstract ! 868: | value is simply rounded and packed into the double-precision format, with ! 869: | the inexact exception raised if the abstract input cannot be represented ! 870: | exactly. However, if the abstract value is too large, the overflow and ! 871: | inexact exceptions are raised and an infinity or maximal finite value is ! 872: | returned. If the abstract value is too small, the input value is rounded ! 873: | to a subnormal number, and the underflow and inexact exceptions are raised ! 874: | if the abstract input cannot be represented exactly as a subnormal double- ! 875: | precision floating-point number. ! 876: | The input significand `zSig' has its binary point between bits 62 ! 877: | and 61, which is 10 bits to the left of the usual location. This shifted ! 878: | significand must be normalized or smaller. If `zSig' is not normalized, ! 879: | `zExp' must be 0; in that case, the result returned is a subnormal number, ! 880: | and it must not require rounding. In the usual case that `zSig' is ! 881: | normalized, `zExp' must be 1 less than the ``true'' floating-point exponent. ! 882: | The handling of underflow and overflow follows the IEC/IEEE Standard for ! 883: | Binary Floating-Point Arithmetic. ! 884: *----------------------------------------------------------------------------*/ ! 885: ! 886: static float64 roundAndPackFloat64( flag zSign, int16 zExp, bits64 zSig ) ! 887: { ! 888: int8 roundingMode; ! 889: flag roundNearestEven; ! 890: int16 roundIncrement, roundBits; ! 891: flag isTiny; ! 892: ! 893: roundingMode = float_rounding_mode; ! 894: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 895: roundIncrement = 0x200; ! 896: if ( ! roundNearestEven ) { ! 897: if ( roundingMode == float_round_to_zero ) { ! 898: roundIncrement = 0; ! 899: } ! 900: else { ! 901: roundIncrement = 0x3FF; ! 902: if ( zSign ) { ! 903: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 904: } ! 905: else { ! 906: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 907: } ! 908: } ! 909: } ! 910: roundBits = zSig & 0x3FF; ! 911: if ( 0x7FD <= (bits16) zExp ) { ! 912: if ( ( 0x7FD < zExp ) ! 913: || ( ( zExp == 0x7FD ) ! 914: && ( (sbits64) ( zSig + roundIncrement ) < 0 ) ) ! 915: ) { ! 916: #ifdef SOFTFLOAT_68K ! 917: float_raise( float_flag_overflow ); ! 918: saveFloat64Internal( zSign, zExp, zSig ); ! 919: if ( roundBits ) float_raise( float_flag_inexact ); ! 920: #else ! 921: float_raise( float_flag_overflow | float_flag_inexact ); ! 922: #endif ! 923: return packFloat64( zSign, 0x7FF, 0 ) - ( roundIncrement == 0 ); ! 924: } ! 925: if ( zExp < 0 ) { ! 926: isTiny = ! 927: ( float_detect_tininess == float_tininess_before_rounding ) ! 928: || ( zExp < -1 ) ! 929: || ( zSig + roundIncrement < LIT64( 0x8000000000000000 ) ); ! 930: #ifdef SOFTFLOAT_68K ! 931: if ( isTiny ) { ! 932: float_raise( float_flag_underflow ); ! 933: saveFloat64Internal( zSign, zExp, zSig ); ! 934: } ! 935: #endif ! 936: shift64RightJamming( zSig, - zExp, &zSig ); ! 937: zExp = 0; ! 938: roundBits = zSig & 0x3FF; ! 939: #ifndef SOFTFLOAT_68K ! 940: if ( isTiny && roundBits ) float_raise( float_flag_underflow ); ! 941: #endif ! 942: } ! 943: } ! 944: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 945: zSig = ( zSig + roundIncrement )>>10; ! 946: zSig &= ~ ( ( ( roundBits ^ 0x200 ) == 0 ) & roundNearestEven ); ! 947: if ( zSig == 0 ) zExp = 0; ! 948: return packFloat64( zSign, zExp, zSig ); ! 949: ! 950: } ! 951: ! 952: #ifdef SOFTFLOAT_I860 // 29-04-2017: Added for Previous ! 953: static float64 roundAndPackFloat64_2( flag zSign, int16 zExp, bits64 zSig ) ! 954: { ! 955: int8 roundingMode; ! 956: flag roundNearestEven; ! 957: int16 roundIncrement, roundBits; ! 958: flag isTiny; ! 959: ! 960: roundingMode = float_rounding_mode2; ! 961: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 962: roundIncrement = 0x200; ! 963: if ( ! roundNearestEven ) { ! 964: if ( roundingMode == float_round_to_zero ) { ! 965: roundIncrement = 0; ! 966: } ! 967: else { ! 968: roundIncrement = 0x3FF; ! 969: if ( zSign ) { ! 970: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 971: } ! 972: else { ! 973: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 974: } ! 975: } ! 976: } ! 977: roundBits = zSig & 0x3FF; ! 978: if ( 0x7FD <= (bits16) zExp ) { ! 979: if ( ( 0x7FD < zExp ) ! 980: || ( ( zExp == 0x7FD ) ! 981: && ( (sbits64) ( zSig + roundIncrement ) < 0 ) ) ! 982: ) { ! 983: float_raise2( float_flag_overflow | float_flag_inexact ); ! 984: return packFloat64( zSign, 0x7FF, 0 ) - ( roundIncrement == 0 ); ! 985: } ! 986: if ( zExp < 0 ) { ! 987: isTiny = ! 988: #ifndef SOFTFLOAT_I860 ! 989: ( float_detect_tininess == float_tininess_before_rounding ) || ! 990: #endif ! 991: ( zExp < -1 ) || ! 992: ( zSig + roundIncrement < LIT64( 0x8000000000000000 ) ); ! 993: shift64RightJamming( zSig, - zExp, &zSig ); ! 994: zExp = 0; ! 995: roundBits = zSig & 0x3FF; ! 996: if ( isTiny && roundBits ) float_raise2( float_flag_underflow ); ! 997: } ! 998: } ! 999: if ( roundBits ) float_raise2( float_flag_inexact ); ! 1000: zSig = ( zSig + roundIncrement )>>10; ! 1001: zSig &= ~ ( ( ( roundBits ^ 0x200 ) == 0 ) & roundNearestEven ); ! 1002: if ( zSig == 0 ) zExp = 0; ! 1003: return packFloat64( zSign, zExp, zSig ); ! 1004: ! 1005: } ! 1006: #endif // End of addition for Previous ! 1007: ! 1008: /*---------------------------------------------------------------------------- ! 1009: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 1010: | and significand `zSig', and returns the proper double-precision floating- ! 1011: | point value corresponding to the abstract input. This routine is just like ! 1012: | `roundAndPackFloat64' except that `zSig' does not have to be normalized. ! 1013: | Bit 63 of `zSig' must be zero, and `zExp' must be 1 less than the ``true'' ! 1014: | floating-point exponent. ! 1015: *----------------------------------------------------------------------------*/ ! 1016: ! 1017: static float64 ! 1018: normalizeRoundAndPackFloat64( flag zSign, int16 zExp, bits64 zSig ) ! 1019: { ! 1020: int8 shiftCount; ! 1021: ! 1022: shiftCount = countLeadingZeros64( zSig ) - 1; ! 1023: return roundAndPackFloat64( zSign, zExp - shiftCount, zSig<<shiftCount ); ! 1024: ! 1025: } ! 1026: ! 1027: #ifdef FLOATX80 ! 1028: ! 1029: /*---------------------------------------------------------------------------- ! 1030: | Normalizes the subnormal extended double-precision floating-point value ! 1031: | represented by the denormalized significand `aSig'. The normalized exponent ! 1032: | and significand are stored at the locations pointed to by `zExpPtr' and ! 1033: | `zSigPtr', respectively. ! 1034: *----------------------------------------------------------------------------*/ ! 1035: ! 1036: /* static */ void ! 1037: normalizeFloatx80Subnormal( bits64 aSig, int32 *zExpPtr, bits64 *zSigPtr ) ! 1038: { ! 1039: int8 shiftCount; ! 1040: ! 1041: shiftCount = countLeadingZeros64( aSig ); ! 1042: *zSigPtr = aSig<<shiftCount; ! 1043: #ifdef SOFTFLOAT_68K ! 1044: *zExpPtr = -shiftCount; ! 1045: #else ! 1046: *zExpPtr = 1 - shiftCount; ! 1047: #endif ! 1048: ! 1049: } ! 1050: ! 1051: /*---------------------------------------------------------------------------- ! 1052: | Packs the sign `zSign', exponent `zExp', and significand `zSig' into an ! 1053: | extended double-precision floating-point value, returning the result. ! 1054: *----------------------------------------------------------------------------*/ ! 1055: ! 1056: floatx80 packFloatx80( flag zSign, int32 zExp, bits64 zSig ) ! 1057: { ! 1058: floatx80 z; ! 1059: ! 1060: z.low = zSig; ! 1061: z.high = ( ( (bits16) zSign )<<15 ) + zExp; ! 1062: return z; ! 1063: ! 1064: } ! 1065: ! 1066: /*---------------------------------------------------------------------------- ! 1067: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 1068: | and extended significand formed by the concatenation of `zSig0' and `zSig1', ! 1069: | and returns the proper extended double-precision floating-point value ! 1070: | corresponding to the abstract input. Ordinarily, the abstract value is ! 1071: | rounded and packed into the extended double-precision format, with the ! 1072: | inexact exception raised if the abstract input cannot be represented ! 1073: | exactly. However, if the abstract value is too large, the overflow and ! 1074: | inexact exceptions are raised and an infinity or maximal finite value is ! 1075: | returned. If the abstract value is too small, the input value is rounded to ! 1076: | a subnormal number, and the underflow and inexact exceptions are raised if ! 1077: | the abstract input cannot be represented exactly as a subnormal extended ! 1078: | double-precision floating-point number. ! 1079: | If `roundingPrecision' is 32 or 64, the result is rounded to the same ! 1080: | number of bits as single or double precision, respectively. Otherwise, the ! 1081: | result is rounded to the full precision of the extended double-precision ! 1082: | format. ! 1083: | The input significand must be normalized or smaller. If the input ! 1084: | significand is not normalized, `zExp' must be 0; in that case, the result ! 1085: | returned is a subnormal number, and it must not require rounding. The ! 1086: | handling of underflow and overflow follows the IEC/IEEE Standard for Binary ! 1087: | Floating-Point Arithmetic. ! 1088: *----------------------------------------------------------------------------*/ ! 1089: ! 1090: // roundAndPackFloatx80 is now also used in fyl2x.c ! 1091: #ifndef SOFTFLOAT_68K ! 1092: /* static */ floatx80 ! 1093: roundAndPackFloatx80( ! 1094: int8 roundingPrecision, flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ! 1095: ) ! 1096: { ! 1097: int8 roundingMode; ! 1098: flag roundNearestEven, increment, isTiny; ! 1099: int64 roundIncrement, roundMask, roundBits; ! 1100: ! 1101: roundingMode = float_rounding_mode; ! 1102: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 1103: if ( roundingPrecision == 80 ) goto precision80; ! 1104: if ( roundingPrecision == 64 ) { ! 1105: roundIncrement = LIT64( 0x0000000000000400 ); ! 1106: roundMask = LIT64( 0x00000000000007FF ); ! 1107: } ! 1108: else if ( roundingPrecision == 32 ) { ! 1109: roundIncrement = LIT64( 0x0000008000000000 ); ! 1110: roundMask = LIT64( 0x000000FFFFFFFFFF ); ! 1111: } ! 1112: else { ! 1113: goto precision80; ! 1114: } ! 1115: zSig0 |= ( zSig1 != 0 ); ! 1116: if ( ! roundNearestEven ) { ! 1117: if ( roundingMode == float_round_to_zero ) { ! 1118: roundIncrement = 0; ! 1119: } ! 1120: else { ! 1121: roundIncrement = roundMask; ! 1122: if ( zSign ) { ! 1123: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 1124: } ! 1125: else { ! 1126: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 1127: } ! 1128: } ! 1129: } ! 1130: roundBits = zSig0 & roundMask; ! 1131: #ifdef SOFTFLOAT_68K ! 1132: if ( 0x7FFE <= (bits32) zExp ) { ! 1133: #else ! 1134: if ( 0x7FFD <= (bits32) ( zExp - 1 ) ) { ! 1135: #endif ! 1136: if ( ( 0x7FFE < zExp ) ! 1137: || ( ( zExp == 0x7FFE ) && ( zSig0 + roundIncrement < zSig0 ) ) ! 1138: ) { ! 1139: goto overflow; ! 1140: } ! 1141: #ifdef SOFTFLOAT_68K ! 1142: if ( zExp < 0 ) { ! 1143: #else ! 1144: if ( zExp <= 0 ) { ! 1145: #endif ! 1146: isTiny = ! 1147: ( float_detect_tininess == float_tininess_before_rounding ) ! 1148: #ifdef SOFTFLOAT_68K ! 1149: || ( zExp < -1 ) ! 1150: #else ! 1151: || ( zExp < 0 ) ! 1152: #endif ! 1153: || ( zSig0 <= zSig0 + roundIncrement ); ! 1154: #ifdef SOFTFLOAT_68K ! 1155: if ( isTiny ) { ! 1156: float_raise( float_flag_underflow ); ! 1157: saveFloatx80Internal( zSign, zExp, zSig0, zSig1 ); ! 1158: } ! 1159: shift64RightJamming( zSig0, -zExp, &zSig0 ); ! 1160: #else ! 1161: shift64RightJamming( zSig0, 1 - zExp, &zSig0 ); ! 1162: #endif ! 1163: zExp = 0; ! 1164: roundBits = zSig0 & roundMask; ! 1165: #ifndef SOFTFLOAT_68K ! 1166: if ( isTiny && roundBits ) float_raise( float_flag_underflow ); ! 1167: #endif ! 1168: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 1169: zSig0 += roundIncrement; ! 1170: #ifndef SOFTFLOAT_68K ! 1171: if ( (sbits64) zSig0 < 0 ) zExp = 1; ! 1172: #endif ! 1173: roundIncrement = roundMask + 1; ! 1174: if ( roundNearestEven && ( roundBits<<1 == roundIncrement ) ) { ! 1175: roundMask |= roundIncrement; ! 1176: } ! 1177: zSig0 &= ~ roundMask; ! 1178: return packFloatx80( zSign, zExp, zSig0 ); ! 1179: } ! 1180: } ! 1181: if ( roundBits ) float_exception_flags |= float_flag_inexact; ! 1182: zSig0 += roundIncrement; ! 1183: if ( zSig0 < roundIncrement ) { ! 1184: ++zExp; ! 1185: zSig0 = LIT64( 0x8000000000000000 ); ! 1186: } ! 1187: roundIncrement = roundMask + 1; ! 1188: if ( roundNearestEven && ( roundBits<<1 == roundIncrement ) ) { ! 1189: roundMask |= roundIncrement; ! 1190: } ! 1191: zSig0 &= ~ roundMask; ! 1192: if ( zSig0 == 0 ) zExp = 0; ! 1193: return packFloatx80( zSign, zExp, zSig0 ); ! 1194: precision80: ! 1195: increment = ( (sbits64) zSig1 < 0 ); ! 1196: if ( ! roundNearestEven ) { ! 1197: if ( roundingMode == float_round_to_zero ) { ! 1198: increment = 0; ! 1199: } ! 1200: else { ! 1201: if ( zSign ) { ! 1202: increment = ( roundingMode == float_round_down ) && zSig1; ! 1203: } ! 1204: else { ! 1205: increment = ( roundingMode == float_round_up ) && zSig1; ! 1206: } ! 1207: } ! 1208: } ! 1209: #ifdef SOFTFLOAT_68K ! 1210: if ( 0x7FFE <= (bits32) zExp ) { ! 1211: #else ! 1212: if ( 0x7FFD <= (bits32) ( zExp - 1 ) ) { ! 1213: #endif ! 1214: if ( ( 0x7FFE < zExp ) ! 1215: || ( ( zExp == 0x7FFE ) ! 1216: && ( zSig0 == LIT64( 0xFFFFFFFFFFFFFFFF ) ) ! 1217: && increment ! 1218: ) ! 1219: ) { ! 1220: roundMask = 0; ! 1221: overflow: ! 1222: #ifndef SOFTFLOAT_68K ! 1223: float_raise( float_flag_overflow | float_flag_inexact ); ! 1224: #else ! 1225: float_raise( float_flag_overflow ); ! 1226: saveFloatx80Internal( zSign, zExp, zSig0, zSig1 ); ! 1227: if ( ( zSig0 & roundMask ) || zSig1 ) float_raise( float_flag_inexact ); ! 1228: #endif ! 1229: if ( ( roundingMode == float_round_to_zero ) ! 1230: || ( zSign && ( roundingMode == float_round_up ) ) ! 1231: || ( ! zSign && ( roundingMode == float_round_down ) ) ! 1232: ) { ! 1233: return packFloatx80( zSign, 0x7FFE, ~ roundMask ); ! 1234: } ! 1235: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 1236: } ! 1237: #ifdef SOFTFLOAT_68K ! 1238: if ( zExp < 0 ) { ! 1239: #else ! 1240: if ( zExp <= 0 ) { ! 1241: #endif ! 1242: isTiny = ! 1243: ( float_detect_tininess == float_tininess_before_rounding ) ! 1244: #ifdef SOFTFLOAT_68K ! 1245: || ( zExp < -1 ) ! 1246: #else ! 1247: || ( zExp < 0 ) ! 1248: #endif ! 1249: || ! increment ! 1250: || ( zSig0 < LIT64( 0xFFFFFFFFFFFFFFFF ) ); ! 1251: #ifdef SOFTFLOAT_68K ! 1252: if ( isTiny ) { ! 1253: float_raise( float_flag_underflow ); ! 1254: saveFloatx80Internal( zSign, zExp, zSig0, zSig1 ); ! 1255: } ! 1256: shift64ExtraRightJamming( zSig0, zSig1, -zExp, &zSig0, &zSig1 ); ! 1257: #else ! 1258: shift64ExtraRightJamming( zSig0, zSig1, 1 - zExp, &zSig0, &zSig1 ); ! 1259: #endif ! 1260: zExp = 0; ! 1261: #ifndef SOFTFLOAT_68K ! 1262: if ( isTiny && zSig1 ) float_raise( float_flag_underflow ); ! 1263: #endif ! 1264: if ( zSig1 ) float_exception_flags |= float_flag_inexact; ! 1265: if ( roundNearestEven ) { ! 1266: increment = ( (sbits64) zSig1 < 0 ); ! 1267: } ! 1268: else { ! 1269: if ( zSign ) { ! 1270: increment = ( roundingMode == float_round_down ) && zSig1; ! 1271: } ! 1272: else { ! 1273: increment = ( roundingMode == float_round_up ) && zSig1; ! 1274: } ! 1275: } ! 1276: if ( increment ) { ! 1277: ++zSig0; ! 1278: zSig0 &= ! 1279: ~ ( ( (bits64) ( zSig1<<1 ) == 0 ) & roundNearestEven ); ! 1280: #ifndef SOFTFLOAT_68K ! 1281: if ( (sbits64) zSig0 < 0 ) zExp = 1; ! 1282: #endif ! 1283: } ! 1284: return packFloatx80( zSign, zExp, zSig0 ); ! 1285: } ! 1286: } ! 1287: if ( zSig1 ) float_exception_flags |= float_flag_inexact; ! 1288: if ( increment ) { ! 1289: ++zSig0; ! 1290: if ( zSig0 == 0 ) { ! 1291: ++zExp; ! 1292: zSig0 = LIT64( 0x8000000000000000 ); ! 1293: } ! 1294: else { ! 1295: zSig0 &= ~ ( ( (bits64) ( zSig1<<1 ) == 0 ) & roundNearestEven ); ! 1296: } ! 1297: } ! 1298: else { ! 1299: if ( zSig0 == 0 ) zExp = 0; ! 1300: } ! 1301: return packFloatx80( zSign, zExp, zSig0 ); ! 1302: ! 1303: } ! 1304: #else // SOFTFLOAT_68K ! 1305: floatx80 roundAndPackFloatx80( int8 roundingPrecision, flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ) ! 1306: { ! 1307: int8 roundingMode; ! 1308: flag roundNearestEven, increment; ! 1309: int64 roundIncrement, roundMask, roundBits; ! 1310: int32 expOffset; ! 1311: ! 1312: roundingMode = float_rounding_mode; ! 1313: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 1314: if ( roundingPrecision == 80 ) goto precision80; ! 1315: if ( roundingPrecision == 64 ) { ! 1316: roundIncrement = LIT64( 0x0000000000000400 ); ! 1317: roundMask = LIT64( 0x00000000000007FF ); ! 1318: expOffset = 0x3C00; ! 1319: } else if ( roundingPrecision == 32 ) { ! 1320: roundIncrement = LIT64( 0x0000008000000000 ); ! 1321: roundMask = LIT64( 0x000000FFFFFFFFFF ); ! 1322: expOffset = 0x3F80; ! 1323: } else { ! 1324: goto precision80; ! 1325: } ! 1326: zSig0 |= ( zSig1 != 0 ); ! 1327: if ( ! roundNearestEven ) { ! 1328: if ( roundingMode == float_round_to_zero ) { ! 1329: roundIncrement = 0; ! 1330: } else { ! 1331: roundIncrement = roundMask; ! 1332: if ( zSign ) { ! 1333: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 1334: } else { ! 1335: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 1336: } ! 1337: } ! 1338: } ! 1339: roundBits = zSig0 & roundMask; ! 1340: if ( ( ( 0x7FFE - expOffset ) < zExp ) || ! 1341: ( ( zExp == ( 0x7FFE - expOffset ) ) && ( zSig0 + roundIncrement < zSig0 ) ) ) { ! 1342: float_raise( float_flag_overflow ); ! 1343: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1344: if ( zSig0 & roundMask ) float_raise( float_flag_inexact ); ! 1345: if ( ( roundingMode == float_round_to_zero ) ! 1346: || ( zSign && ( roundingMode == float_round_up ) ) ! 1347: || ( ! zSign && ( roundingMode == float_round_down ) ) ! 1348: ) { ! 1349: return packFloatx80( zSign, 0x7FFE - expOffset, ~ roundMask ); ! 1350: } ! 1351: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 1352: } ! 1353: if ( zExp < ( expOffset + 1 ) ) { ! 1354: float_raise( float_flag_underflow ); ! 1355: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1356: shift64RightJamming( zSig0, -( zExp - ( expOffset + 1 ) ), &zSig0 ); ! 1357: zExp = expOffset + 1; ! 1358: roundBits = zSig0 & roundMask; ! 1359: if ( roundBits ) float_raise( float_flag_inexact ); ! 1360: zSig0 += roundIncrement; ! 1361: roundIncrement = roundMask + 1; ! 1362: if ( roundNearestEven && ( roundBits<<1 == roundIncrement ) ) { ! 1363: roundMask |= roundIncrement; ! 1364: } ! 1365: zSig0 &= ~ roundMask; ! 1366: return packFloatx80( zSign, zExp, zSig0 ); ! 1367: } ! 1368: if ( roundBits ) { ! 1369: float_raise( float_flag_inexact ); ! 1370: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1371: } ! 1372: zSig0 += roundIncrement; ! 1373: if ( zSig0 < roundIncrement ) { ! 1374: ++zExp; ! 1375: zSig0 = LIT64( 0x8000000000000000 ); ! 1376: } ! 1377: roundIncrement = roundMask + 1; ! 1378: if ( roundNearestEven && ( roundBits<<1 == roundIncrement ) ) { ! 1379: roundMask |= roundIncrement; ! 1380: } ! 1381: zSig0 &= ~ roundMask; ! 1382: if ( zSig0 == 0 ) zExp = 0; ! 1383: return packFloatx80( zSign, zExp, zSig0 ); ! 1384: precision80: ! 1385: increment = ( (sbits64) zSig1 < 0 ); ! 1386: if ( ! roundNearestEven ) { ! 1387: if ( roundingMode == float_round_to_zero ) { ! 1388: increment = 0; ! 1389: } else { ! 1390: if ( zSign ) { ! 1391: increment = ( roundingMode == float_round_down ) && zSig1; ! 1392: } else { ! 1393: increment = ( roundingMode == float_round_up ) && zSig1; ! 1394: } ! 1395: } ! 1396: } ! 1397: if ( 0x7FFE <= (bits32) zExp ) { ! 1398: if ( ( 0x7FFE < zExp ) || ! 1399: ( ( zExp == 0x7FFE ) && ( zSig0 == LIT64( 0xFFFFFFFFFFFFFFFF ) ) && increment ) ! 1400: ) { ! 1401: roundMask = 0; ! 1402: float_raise( float_flag_overflow ); ! 1403: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1404: if ( ( zSig0 & roundMask ) || zSig1 ) float_raise( float_flag_inexact ); ! 1405: if ( ( roundingMode == float_round_to_zero ) ! 1406: || ( zSign && ( roundingMode == float_round_up ) ) ! 1407: || ( ! zSign && ( roundingMode == float_round_down ) ) ! 1408: ) { ! 1409: return packFloatx80( zSign, 0x7FFE, ~ roundMask ); ! 1410: } ! 1411: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 1412: } ! 1413: if ( zExp < 0 ) { ! 1414: float_raise( float_flag_underflow ); ! 1415: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1416: shift64ExtraRightJamming( zSig0, zSig1, -zExp, &zSig0, &zSig1 ); ! 1417: zExp = 0; ! 1418: if ( zSig1 ) float_raise( float_flag_inexact ); ! 1419: if ( roundNearestEven ) { ! 1420: increment = ( (sbits64) zSig1 < 0 ); ! 1421: } else { ! 1422: if ( zSign ) { ! 1423: increment = ( roundingMode == float_round_down ) && zSig1; ! 1424: } else { ! 1425: increment = ( roundingMode == float_round_up ) && zSig1; ! 1426: } ! 1427: } ! 1428: if ( increment ) { ! 1429: ++zSig0; ! 1430: zSig0 &= ! 1431: ~ ( ( (bits64) ( zSig1<<1 ) == 0 ) & roundNearestEven ); ! 1432: } ! 1433: return packFloatx80( zSign, zExp, zSig0 ); ! 1434: } ! 1435: } ! 1436: if ( zSig1 ) { ! 1437: float_raise( float_flag_inexact ); ! 1438: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1439: } ! 1440: if ( increment ) { ! 1441: ++zSig0; ! 1442: if ( zSig0 == 0 ) { ! 1443: ++zExp; ! 1444: zSig0 = LIT64( 0x8000000000000000 ); ! 1445: } else { ! 1446: zSig0 &= ~ ( ( (bits64) ( zSig1<<1 ) == 0 ) & roundNearestEven ); ! 1447: } ! 1448: } else { ! 1449: if ( zSig0 == 0 ) zExp = 0; ! 1450: } ! 1451: return packFloatx80( zSign, zExp, zSig0 ); ! 1452: ! 1453: } ! 1454: #endif ! 1455: ! 1456: #ifdef SOFTFLOAT_68K // 21-01-2017: Added for Previous ! 1457: floatx80 roundSigAndPackFloatx80( int8 roundingPrecision, flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ) ! 1458: { ! 1459: int8 roundingMode; ! 1460: flag roundNearestEven, isTiny; ! 1461: int64 roundIncrement, roundMask, roundBits; ! 1462: ! 1463: roundingMode = float_rounding_mode; ! 1464: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 1465: if ( roundingPrecision == 32 ) { ! 1466: roundIncrement = LIT64( 0x0000008000000000 ); ! 1467: roundMask = LIT64( 0x000000FFFFFFFFFF ); ! 1468: } else if ( roundingPrecision == 64 ) { ! 1469: roundIncrement = LIT64( 0x0000000000000400 ); ! 1470: roundMask = LIT64( 0x00000000000007FF ); ! 1471: } else { ! 1472: return roundAndPackFloatx80( 80, zSign, zExp, zSig0, zSig1 ); ! 1473: } ! 1474: zSig0 |= ( zSig1 != 0 ); ! 1475: if ( ! roundNearestEven ) { ! 1476: if ( roundingMode == float_round_to_zero ) { ! 1477: roundIncrement = 0; ! 1478: } ! 1479: else { ! 1480: roundIncrement = roundMask; ! 1481: if ( zSign ) { ! 1482: if ( roundingMode == float_round_up ) roundIncrement = 0; ! 1483: } ! 1484: else { ! 1485: if ( roundingMode == float_round_down ) roundIncrement = 0; ! 1486: } ! 1487: } ! 1488: } ! 1489: roundBits = zSig0 & roundMask; ! 1490: ! 1491: if ( 0x7FFE <= (bits32) zExp ) { ! 1492: if ( ( 0x7FFE < zExp ) ! 1493: || ( ( zExp == 0x7FFE ) && ( zSig0 + roundIncrement < zSig0 ) ) ! 1494: ) { ! 1495: float_raise( float_flag_overflow ); ! 1496: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1497: if ( zSig0 & roundMask ) float_raise( float_flag_inexact ); ! 1498: if ( ( roundingMode == float_round_to_zero ) ! 1499: || ( zSign && ( roundingMode == float_round_up ) ) ! 1500: || ( ! zSign && ( roundingMode == float_round_down ) ) ! 1501: ) { ! 1502: return packFloatx80( zSign, 0x7FFE, LIT64( 0xFFFFFFFFFFFFFFFF ) ); ! 1503: } ! 1504: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 1505: } ! 1506: ! 1507: if ( zExp < 0 ) { ! 1508: isTiny = ! 1509: ( float_detect_tininess == float_tininess_before_rounding ) ! 1510: || ( zExp < -1 ) ! 1511: || ( zSig0 <= zSig0 + roundIncrement ); ! 1512: if ( isTiny ) { ! 1513: float_raise( float_flag_underflow ); ! 1514: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1515: } ! 1516: shift64RightJamming( zSig0, -zExp, &zSig0 ); ! 1517: zExp = 0; ! 1518: roundBits = zSig0 & roundMask; ! 1519: if ( roundBits ) float_raise( float_flag_inexact ); ! 1520: zSig0 += roundIncrement; ! 1521: if ( roundNearestEven && ( roundBits == roundIncrement ) ) { ! 1522: roundMask |= roundIncrement<<1; ! 1523: } ! 1524: zSig0 &= ~ roundMask; ! 1525: return packFloatx80( zSign, zExp, zSig0 ); ! 1526: } ! 1527: } ! 1528: if ( roundBits ) { ! 1529: float_raise( float_flag_inexact ); ! 1530: saveFloatx80Internal( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1531: } ! 1532: zSig0 += roundIncrement; ! 1533: if ( zSig0 < roundIncrement ) { ! 1534: ++zExp; ! 1535: zSig0 = LIT64( 0x8000000000000000 ); ! 1536: } ! 1537: roundIncrement = roundMask + 1; ! 1538: if ( roundNearestEven && ( roundBits<<1 == roundIncrement ) ) { ! 1539: roundMask |= roundIncrement; ! 1540: } ! 1541: zSig0 &= ~ roundMask; ! 1542: if ( zSig0 == 0 ) zExp = 0; ! 1543: return packFloatx80( zSign, zExp, zSig0 ); ! 1544: ! 1545: } ! 1546: #endif // End of Addition for Previous ! 1547: /*---------------------------------------------------------------------------- ! 1548: | Takes an abstract floating-point value having sign `zSign', exponent ! 1549: | `zExp', and significand formed by the concatenation of `zSig0' and `zSig1', ! 1550: | and returns the proper extended double-precision floating-point value ! 1551: | corresponding to the abstract input. This routine is just like ! 1552: | `roundAndPackFloatx80' except that the input significand does not have to be ! 1553: | normalized. ! 1554: *----------------------------------------------------------------------------*/ ! 1555: ! 1556: static floatx80 ! 1557: normalizeRoundAndPackFloatx80( ! 1558: int8 roundingPrecision, flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ! 1559: ) ! 1560: { ! 1561: int8 shiftCount; ! 1562: ! 1563: if ( zSig0 == 0 ) { ! 1564: zSig0 = zSig1; ! 1565: zSig1 = 0; ! 1566: zExp -= 64; ! 1567: } ! 1568: shiftCount = countLeadingZeros64( zSig0 ); ! 1569: shortShift128Left( zSig0, zSig1, shiftCount, &zSig0, &zSig1 ); ! 1570: zExp -= shiftCount; ! 1571: return ! 1572: roundAndPackFloatx80( roundingPrecision, zSign, zExp, zSig0, zSig1 ); ! 1573: ! 1574: } ! 1575: ! 1576: #endif ! 1577: ! 1578: #ifdef FLOAT128 ! 1579: ! 1580: /*---------------------------------------------------------------------------- ! 1581: | Returns the least-significant 64 fraction bits of the quadruple-precision ! 1582: | floating-point value `a'. ! 1583: *----------------------------------------------------------------------------*/ ! 1584: ! 1585: INLINE bits64 extractFloat128Frac1( float128 a ) ! 1586: { ! 1587: return a.low; ! 1588: ! 1589: } ! 1590: ! 1591: /*---------------------------------------------------------------------------- ! 1592: | Returns the most-significant 48 fraction bits of the quadruple-precision ! 1593: | floating-point value `a'. ! 1594: *----------------------------------------------------------------------------*/ ! 1595: ! 1596: INLINE bits64 extractFloat128Frac0( float128 a ) ! 1597: { ! 1598: return a.high & LIT64( 0x0000FFFFFFFFFFFF ); ! 1599: ! 1600: } ! 1601: ! 1602: /*---------------------------------------------------------------------------- ! 1603: | Returns the exponent bits of the quadruple-precision floating-point value ! 1604: | `a'. ! 1605: *----------------------------------------------------------------------------*/ ! 1606: ! 1607: INLINE int32 extractFloat128Exp( float128 a ) ! 1608: { ! 1609: return ( a.high>>48 ) & 0x7FFF; ! 1610: ! 1611: } ! 1612: ! 1613: /*---------------------------------------------------------------------------- ! 1614: | Returns the sign bit of the quadruple-precision floating-point value `a'. ! 1615: *----------------------------------------------------------------------------*/ ! 1616: ! 1617: INLINE flag extractFloat128Sign( float128 a ) ! 1618: { ! 1619: return a.high>>63; ! 1620: ! 1621: } ! 1622: ! 1623: /*---------------------------------------------------------------------------- ! 1624: | Normalizes the subnormal quadruple-precision floating-point value ! 1625: | represented by the denormalized significand formed by the concatenation of ! 1626: | `aSig0' and `aSig1'. The normalized exponent is stored at the location ! 1627: | pointed to by `zExpPtr'. The most significant 49 bits of the normalized ! 1628: | significand are stored at the location pointed to by `zSig0Ptr', and the ! 1629: | least significant 64 bits of the normalized significand are stored at the ! 1630: | location pointed to by `zSig1Ptr'. ! 1631: *----------------------------------------------------------------------------*/ ! 1632: ! 1633: static void ! 1634: normalizeFloat128Subnormal( ! 1635: bits64 aSig0, ! 1636: bits64 aSig1, ! 1637: int32 *zExpPtr, ! 1638: bits64 *zSig0Ptr, ! 1639: bits64 *zSig1Ptr ! 1640: ) ! 1641: { ! 1642: int8 shiftCount; ! 1643: ! 1644: if ( aSig0 == 0 ) { ! 1645: shiftCount = countLeadingZeros64( aSig1 ) - 15; ! 1646: if ( shiftCount < 0 ) { ! 1647: *zSig0Ptr = aSig1>>( - shiftCount ); ! 1648: *zSig1Ptr = aSig1<<( shiftCount & 63 ); ! 1649: } ! 1650: else { ! 1651: *zSig0Ptr = aSig1<<shiftCount; ! 1652: *zSig1Ptr = 0; ! 1653: } ! 1654: *zExpPtr = - shiftCount - 63; ! 1655: } ! 1656: else { ! 1657: shiftCount = countLeadingZeros64( aSig0 ) - 15; ! 1658: shortShift128Left( aSig0, aSig1, shiftCount, zSig0Ptr, zSig1Ptr ); ! 1659: *zExpPtr = 1 - shiftCount; ! 1660: } ! 1661: ! 1662: } ! 1663: ! 1664: /*---------------------------------------------------------------------------- ! 1665: | Packs the sign `zSign', the exponent `zExp', and the significand formed ! 1666: | by the concatenation of `zSig0' and `zSig1' into a quadruple-precision ! 1667: | floating-point value, returning the result. After being shifted into the ! 1668: | proper positions, the three fields `zSign', `zExp', and `zSig0' are simply ! 1669: | added together to form the most significant 32 bits of the result. This ! 1670: | means that any integer portion of `zSig0' will be added into the exponent. ! 1671: | Since a properly normalized significand will have an integer portion equal ! 1672: | to 1, the `zExp' input should be 1 less than the desired result exponent ! 1673: | whenever `zSig0' and `zSig1' concatenated form a complete, normalized ! 1674: | significand. ! 1675: *----------------------------------------------------------------------------*/ ! 1676: ! 1677: float128 packFloat128( flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ) ! 1678: { ! 1679: float128 z; ! 1680: ! 1681: z.low = zSig1; ! 1682: z.high = ( ( (bits64) zSign )<<63 ) + ( ( (bits64) zExp )<<48 ) + zSig0; ! 1683: return z; ! 1684: ! 1685: } ! 1686: ! 1687: /*---------------------------------------------------------------------------- ! 1688: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 1689: | and extended significand formed by the concatenation of `zSig0', `zSig1', ! 1690: | and `zSig2', and returns the proper quadruple-precision floating-point value ! 1691: | corresponding to the abstract input. Ordinarily, the abstract value is ! 1692: | simply rounded and packed into the quadruple-precision format, with the ! 1693: | inexact exception raised if the abstract input cannot be represented ! 1694: | exactly. However, if the abstract value is too large, the overflow and ! 1695: | inexact exceptions are raised and an infinity or maximal finite value is ! 1696: | returned. If the abstract value is too small, the input value is rounded to ! 1697: | a subnormal number, and the underflow and inexact exceptions are raised if ! 1698: | the abstract input cannot be represented exactly as a subnormal quadruple- ! 1699: | precision floating-point number. ! 1700: | The input significand must be normalized or smaller. If the input ! 1701: | significand is not normalized, `zExp' must be 0; in that case, the result ! 1702: | returned is a subnormal number, and it must not require rounding. In the ! 1703: | usual case that the input significand is normalized, `zExp' must be 1 less ! 1704: | than the ``true'' floating-point exponent. The handling of underflow and ! 1705: | overflow follows the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1706: *----------------------------------------------------------------------------*/ ! 1707: ! 1708: float128 roundAndPackFloat128( ! 1709: flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1, bits64 zSig2 ) ! 1710: { ! 1711: int8 roundingMode; ! 1712: flag roundNearestEven, increment, isTiny; ! 1713: ! 1714: roundingMode = float_rounding_mode; ! 1715: roundNearestEven = ( roundingMode == float_round_nearest_even ); ! 1716: increment = ( (sbits64) zSig2 < 0 ); ! 1717: if ( ! roundNearestEven ) { ! 1718: if ( roundingMode == float_round_to_zero ) { ! 1719: increment = 0; ! 1720: } ! 1721: else { ! 1722: if ( zSign ) { ! 1723: increment = ( roundingMode == float_round_down ) && zSig2; ! 1724: } ! 1725: else { ! 1726: increment = ( roundingMode == float_round_up ) && zSig2; ! 1727: } ! 1728: } ! 1729: } ! 1730: if ( 0x7FFD <= (bits32) zExp ) { ! 1731: if ( ( 0x7FFD < zExp ) ! 1732: || ( ( zExp == 0x7FFD ) ! 1733: && eq128( ! 1734: LIT64( 0x0001FFFFFFFFFFFF ), ! 1735: LIT64( 0xFFFFFFFFFFFFFFFF ), ! 1736: zSig0, ! 1737: zSig1 ! 1738: ) ! 1739: && increment ! 1740: ) ! 1741: ) { ! 1742: #ifdef SOFTFLOAT_68K ! 1743: float_raise( float_flag_overflow ); ! 1744: if ( zSig2 ) float_raise( float_flag_inexact ); ! 1745: #else ! 1746: float_raise( float_flag_overflow | float_flag_inexact ); ! 1747: #endif ! 1748: if ( ( roundingMode == float_round_to_zero ) ! 1749: || ( zSign && ( roundingMode == float_round_up ) ) ! 1750: || ( ! zSign && ( roundingMode == float_round_down ) ) ! 1751: ) { ! 1752: return ! 1753: packFloat128( ! 1754: zSign, ! 1755: 0x7FFE, ! 1756: LIT64( 0x0000FFFFFFFFFFFF ), ! 1757: LIT64( 0xFFFFFFFFFFFFFFFF ) ! 1758: ); ! 1759: } ! 1760: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 1761: } ! 1762: if ( zExp < 0 ) { ! 1763: isTiny = ! 1764: ( float_detect_tininess == float_tininess_before_rounding ) ! 1765: || ( zExp < -1 ) ! 1766: || ! increment ! 1767: || lt128( ! 1768: zSig0, ! 1769: zSig1, ! 1770: LIT64( 0x0001FFFFFFFFFFFF ), ! 1771: LIT64( 0xFFFFFFFFFFFFFFFF ) ! 1772: ); ! 1773: shift128ExtraRightJamming( ! 1774: zSig0, zSig1, zSig2, - zExp, &zSig0, &zSig1, &zSig2 ); ! 1775: zExp = 0; ! 1776: #ifdef SOFTFLOAT_68K ! 1777: if ( isTiny ) float_raise( float_flag_underflow ); ! 1778: #else ! 1779: if ( isTiny && zSig2 ) float_raise( float_flag_underflow ); ! 1780: #endif ! 1781: if ( roundNearestEven ) { ! 1782: increment = ( (sbits64) zSig2 < 0 ); ! 1783: } ! 1784: else { ! 1785: if ( zSign ) { ! 1786: increment = ( roundingMode == float_round_down ) && zSig2; ! 1787: } ! 1788: else { ! 1789: increment = ( roundingMode == float_round_up ) && zSig2; ! 1790: } ! 1791: } ! 1792: } ! 1793: } ! 1794: if ( zSig2 ) float_exception_flags |= float_flag_inexact; ! 1795: if ( increment ) { ! 1796: add128( zSig0, zSig1, 0, 1, &zSig0, &zSig1 ); ! 1797: zSig1 &= ~ ( ( zSig2 + zSig2 == 0 ) & roundNearestEven ); ! 1798: } ! 1799: else { ! 1800: if ( ( zSig0 | zSig1 ) == 0 ) zExp = 0; ! 1801: } ! 1802: return packFloat128( zSign, zExp, zSig0, zSig1 ); ! 1803: ! 1804: } ! 1805: ! 1806: /*---------------------------------------------------------------------------- ! 1807: | Takes an abstract floating-point value having sign `zSign', exponent `zExp', ! 1808: | and significand formed by the concatenation of `zSig0' and `zSig1', and ! 1809: | returns the proper quadruple-precision floating-point value corresponding ! 1810: | to the abstract input. This routine is just like `roundAndPackFloat128' ! 1811: | except that the input significand has fewer bits and does not have to be ! 1812: | normalized. In all cases, `zExp' must be 1 less than the ``true'' floating- ! 1813: | point exponent. ! 1814: *----------------------------------------------------------------------------*/ ! 1815: ! 1816: float128 normalizeRoundAndPackFloat128( ! 1817: flag zSign, int32 zExp, bits64 zSig0, bits64 zSig1 ) ! 1818: { ! 1819: int8 shiftCount; ! 1820: bits64 zSig2; ! 1821: ! 1822: if ( zSig0 == 0 ) { ! 1823: zSig0 = zSig1; ! 1824: zSig1 = 0; ! 1825: zExp -= 64; ! 1826: } ! 1827: shiftCount = countLeadingZeros64( zSig0 ) - 15; ! 1828: if ( 0 <= shiftCount ) { ! 1829: zSig2 = 0; ! 1830: shortShift128Left( zSig0, zSig1, shiftCount, &zSig0, &zSig1 ); ! 1831: } ! 1832: else { ! 1833: shift128ExtraRightJamming( ! 1834: zSig0, zSig1, 0, - shiftCount, &zSig0, &zSig1, &zSig2 ); ! 1835: } ! 1836: zExp -= shiftCount; ! 1837: return roundAndPackFloat128( zSign, zExp, zSig0, zSig1, zSig2 ); ! 1838: ! 1839: } ! 1840: ! 1841: #endif ! 1842: ! 1843: /*---------------------------------------------------------------------------- ! 1844: | Returns the result of converting the 32-bit two's complement integer `a' ! 1845: | to the single-precision floating-point format. The conversion is performed ! 1846: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1847: *----------------------------------------------------------------------------*/ ! 1848: ! 1849: float32 int32_to_float32( int32 a ) ! 1850: { ! 1851: flag zSign; ! 1852: ! 1853: if ( a == 0 ) return 0; ! 1854: if ( a == (sbits32) 0x80000000 ) return packFloat32( 1, 0x9E, 0 ); ! 1855: zSign = ( a < 0 ); ! 1856: return normalizeRoundAndPackFloat32( zSign, 0x9C, zSign ? - a : a ); ! 1857: ! 1858: } ! 1859: ! 1860: /*---------------------------------------------------------------------------- ! 1861: | Returns the result of converting the 32-bit two's complement integer `a' ! 1862: | to the double-precision floating-point format. The conversion is performed ! 1863: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1864: *----------------------------------------------------------------------------*/ ! 1865: ! 1866: float64 int32_to_float64( int32 a ) ! 1867: { ! 1868: flag zSign; ! 1869: uint32 absA; ! 1870: int8 shiftCount; ! 1871: bits64 zSig; ! 1872: ! 1873: if ( a == 0 ) return 0; ! 1874: zSign = ( a < 0 ); ! 1875: absA = zSign ? - a : a; ! 1876: shiftCount = countLeadingZeros32( absA ) + 21; ! 1877: zSig = absA; ! 1878: return packFloat64( zSign, 0x432 - shiftCount, zSig<<shiftCount ); ! 1879: ! 1880: } ! 1881: ! 1882: #ifdef FLOATX80 ! 1883: ! 1884: /*---------------------------------------------------------------------------- ! 1885: | Returns the result of converting the 32-bit two's complement integer `a' ! 1886: | to the extended double-precision floating-point format. The conversion ! 1887: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 1888: | Arithmetic. ! 1889: *----------------------------------------------------------------------------*/ ! 1890: ! 1891: floatx80 int32_to_floatx80( int32 a ) ! 1892: { ! 1893: flag zSign; ! 1894: uint32 absA; ! 1895: int8 shiftCount; ! 1896: bits64 zSig; ! 1897: ! 1898: if ( a == 0 ) return packFloatx80( 0, 0, 0 ); ! 1899: zSign = ( a < 0 ); ! 1900: absA = zSign ? - a : a; ! 1901: shiftCount = countLeadingZeros32( absA ) + 32; ! 1902: zSig = absA; ! 1903: return packFloatx80( zSign, 0x403E - shiftCount, zSig<<shiftCount ); ! 1904: ! 1905: } ! 1906: ! 1907: #endif ! 1908: ! 1909: #ifdef FLOAT128 ! 1910: ! 1911: /*---------------------------------------------------------------------------- ! 1912: | Returns the result of converting the 32-bit two's complement integer `a' to ! 1913: | the quadruple-precision floating-point format. The conversion is performed ! 1914: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1915: *----------------------------------------------------------------------------*/ ! 1916: ! 1917: float128 int32_to_float128( int32 a ) ! 1918: { ! 1919: flag zSign; ! 1920: uint32 absA; ! 1921: int8 shiftCount; ! 1922: bits64 zSig0; ! 1923: ! 1924: if ( a == 0 ) return packFloat128( 0, 0, 0, 0 ); ! 1925: zSign = ( a < 0 ); ! 1926: absA = zSign ? - a : a; ! 1927: shiftCount = countLeadingZeros32( absA ) + 17; ! 1928: zSig0 = absA; ! 1929: return packFloat128( zSign, 0x402E - shiftCount, zSig0<<shiftCount, 0 ); ! 1930: ! 1931: } ! 1932: ! 1933: #endif ! 1934: ! 1935: /*---------------------------------------------------------------------------- ! 1936: | Returns the result of converting the 64-bit two's complement integer `a' ! 1937: | to the single-precision floating-point format. The conversion is performed ! 1938: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1939: *----------------------------------------------------------------------------*/ ! 1940: ! 1941: float32 int64_to_float32( int64 a ) ! 1942: { ! 1943: flag zSign; ! 1944: uint64 absA; ! 1945: int8 shiftCount; ! 1946: // bits32 zSig; ! 1947: ! 1948: if ( a == 0 ) return 0; ! 1949: zSign = ( a < 0 ); ! 1950: absA = zSign ? - a : a; ! 1951: shiftCount = countLeadingZeros64( absA ) - 40; ! 1952: if ( 0 <= shiftCount ) { ! 1953: return packFloat32( zSign, 0x95 - shiftCount, absA<<shiftCount ); ! 1954: } ! 1955: else { ! 1956: shiftCount += 7; ! 1957: if ( shiftCount < 0 ) { ! 1958: shift64RightJamming( absA, - shiftCount, &absA ); ! 1959: } ! 1960: else { ! 1961: absA <<= shiftCount; ! 1962: } ! 1963: return roundAndPackFloat32( zSign, 0x9C - shiftCount, absA ); ! 1964: } ! 1965: ! 1966: } ! 1967: ! 1968: /*---------------------------------------------------------------------------- ! 1969: | Returns the result of converting the 64-bit two's complement integer `a' ! 1970: | to the double-precision floating-point format. The conversion is performed ! 1971: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 1972: *----------------------------------------------------------------------------*/ ! 1973: ! 1974: float64 int64_to_float64( int64 a ) ! 1975: { ! 1976: flag zSign; ! 1977: ! 1978: if ( a == 0 ) return 0; ! 1979: if ( a == (sbits64) LIT64( 0x8000000000000000 ) ) { ! 1980: return packFloat64( 1, 0x43E, 0 ); ! 1981: } ! 1982: zSign = ( a < 0 ); ! 1983: return normalizeRoundAndPackFloat64( zSign, 0x43C, zSign ? - a : a ); ! 1984: ! 1985: } ! 1986: ! 1987: #ifdef FLOATX80 ! 1988: ! 1989: /*---------------------------------------------------------------------------- ! 1990: | Returns the result of converting the 64-bit two's complement integer `a' ! 1991: | to the extended double-precision floating-point format. The conversion ! 1992: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 1993: | Arithmetic. ! 1994: *----------------------------------------------------------------------------*/ ! 1995: ! 1996: floatx80 int64_to_floatx80( int64 a ) ! 1997: { ! 1998: flag zSign; ! 1999: uint64 absA; ! 2000: int8 shiftCount; ! 2001: ! 2002: if ( a == 0 ) return packFloatx80( 0, 0, 0 ); ! 2003: zSign = ( a < 0 ); ! 2004: absA = zSign ? - a : a; ! 2005: shiftCount = countLeadingZeros64( absA ); ! 2006: return packFloatx80( zSign, 0x403E - shiftCount, absA<<shiftCount ); ! 2007: ! 2008: } ! 2009: ! 2010: #endif ! 2011: ! 2012: #ifdef FLOAT128 ! 2013: ! 2014: /*---------------------------------------------------------------------------- ! 2015: | Returns the result of converting the 64-bit two's complement integer `a' to ! 2016: | the quadruple-precision floating-point format. The conversion is performed ! 2017: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 2018: *----------------------------------------------------------------------------*/ ! 2019: ! 2020: float128 int64_to_float128( int64 a ) ! 2021: { ! 2022: flag zSign; ! 2023: uint64 absA; ! 2024: int8 shiftCount; ! 2025: int32 zExp; ! 2026: bits64 zSig0, zSig1; ! 2027: ! 2028: if ( a == 0 ) return packFloat128( 0, 0, 0, 0 ); ! 2029: zSign = ( a < 0 ); ! 2030: absA = zSign ? - a : a; ! 2031: shiftCount = countLeadingZeros64( absA ) + 49; ! 2032: zExp = 0x406E - shiftCount; ! 2033: if ( 64 <= shiftCount ) { ! 2034: zSig1 = 0; ! 2035: zSig0 = absA; ! 2036: shiftCount -= 64; ! 2037: } ! 2038: else { ! 2039: zSig1 = absA; ! 2040: zSig0 = 0; ! 2041: } ! 2042: shortShift128Left( zSig0, zSig1, shiftCount, &zSig0, &zSig1 ); ! 2043: return packFloat128( zSign, zExp, zSig0, zSig1 ); ! 2044: ! 2045: } ! 2046: ! 2047: #endif ! 2048: ! 2049: /*---------------------------------------------------------------------------- ! 2050: | Returns the result of converting the single-precision floating-point value ! 2051: | `a' to the 32-bit two's complement integer format. The conversion is ! 2052: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2053: | Arithmetic---which means in particular that the conversion is rounded ! 2054: | according to the current rounding mode. If `a' is a NaN, the largest ! 2055: | positive integer is returned. Otherwise, if the conversion overflows, the ! 2056: | largest integer with the same sign as `a' is returned. ! 2057: *----------------------------------------------------------------------------*/ ! 2058: ! 2059: int32 float32_to_int32( float32 a ) ! 2060: { ! 2061: flag aSign; ! 2062: int16 aExp, shiftCount; ! 2063: bits32 aSig; ! 2064: bits64 aSig64; ! 2065: ! 2066: aSig = extractFloat32Frac( a ); ! 2067: aExp = extractFloat32Exp( a ); ! 2068: aSign = extractFloat32Sign( a ); ! 2069: if ( ( aExp == 0xFF ) && aSig ) aSign = 0; ! 2070: if ( aExp ) aSig |= 0x00800000; ! 2071: shiftCount = 0xAF - aExp; ! 2072: aSig64 = aSig; ! 2073: aSig64 <<= 32; ! 2074: if ( 0 < shiftCount ) shift64RightJamming( aSig64, shiftCount, &aSig64 ); ! 2075: #ifdef SOFTFLOAT_I860 ! 2076: return roundAndPackInt32_2( aSign, aSig64 ); ! 2077: #else ! 2078: return roundAndPackInt32( aSign, aSig64 ); ! 2079: #endif ! 2080: ! 2081: } ! 2082: ! 2083: /*---------------------------------------------------------------------------- ! 2084: | Returns the result of converting the single-precision floating-point value ! 2085: | `a' to the 32-bit two's complement integer format. The conversion is ! 2086: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2087: | Arithmetic, except that the conversion is always rounded toward zero. ! 2088: | If `a' is a NaN, the largest positive integer is returned. Otherwise, if ! 2089: | the conversion overflows, the largest integer with the same sign as `a' is ! 2090: | returned. ! 2091: *----------------------------------------------------------------------------*/ ! 2092: ! 2093: int32 float32_to_int32_round_to_zero( float32 a ) ! 2094: { ! 2095: flag aSign; ! 2096: int16 aExp, shiftCount; ! 2097: bits32 aSig; ! 2098: int32 z; ! 2099: ! 2100: aSig = extractFloat32Frac( a ); ! 2101: aExp = extractFloat32Exp( a ); ! 2102: aSign = extractFloat32Sign( a ); ! 2103: shiftCount = aExp - 0x9E; ! 2104: if ( 0 <= shiftCount ) { ! 2105: if ( a != 0xCF000000 ) { ! 2106: #ifdef SOFTFLOAT_I860 ! 2107: float_raise2( float_flag_invalid ); ! 2108: #else ! 2109: float_raise( float_flag_invalid ); ! 2110: #endif ! 2111: if ( ! aSign || ( ( aExp == 0xFF ) && aSig ) ) return 0x7FFFFFFF; ! 2112: } ! 2113: return (sbits32) 0x80000000; ! 2114: } ! 2115: else if ( aExp <= 0x7E ) { ! 2116: #ifdef SOFTFLOAT_I860 ! 2117: if ( aExp | aSig ) float_raise2( float_flag_inexact ); ! 2118: #else ! 2119: if ( aExp | aSig ) float_exception_flags |= float_flag_inexact; ! 2120: #endif ! 2121: return 0; ! 2122: } ! 2123: aSig = ( aSig | 0x00800000 )<<8; ! 2124: z = aSig>>( - shiftCount ); ! 2125: if ( (bits32) ( aSig<<( shiftCount & 31 ) ) ) { ! 2126: #ifdef SOFTFLOAT_I860 ! 2127: float_raise2( float_flag_inexact ); ! 2128: #else ! 2129: float_exception_flags |= float_flag_inexact; ! 2130: #endif ! 2131: } ! 2132: if ( aSign ) z = - z; ! 2133: return z; ! 2134: ! 2135: } ! 2136: ! 2137: /*---------------------------------------------------------------------------- ! 2138: | Returns the result of converting the single-precision floating-point value ! 2139: | `a' to the 64-bit two's complement integer format. The conversion is ! 2140: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2141: | Arithmetic---which means in particular that the conversion is rounded ! 2142: | according to the current rounding mode. If `a' is a NaN, the largest ! 2143: | positive integer is returned. Otherwise, if the conversion overflows, the ! 2144: | largest integer with the same sign as `a' is returned. ! 2145: *----------------------------------------------------------------------------*/ ! 2146: ! 2147: int64 float32_to_int64( float32 a ) ! 2148: { ! 2149: flag aSign; ! 2150: int16 aExp, shiftCount; ! 2151: bits32 aSig; ! 2152: bits64 aSig64, aSigExtra; ! 2153: ! 2154: aSig = extractFloat32Frac( a ); ! 2155: aExp = extractFloat32Exp( a ); ! 2156: aSign = extractFloat32Sign( a ); ! 2157: shiftCount = 0xBE - aExp; ! 2158: if ( shiftCount < 0 ) { ! 2159: float_raise( float_flag_invalid ); ! 2160: if ( ! aSign || ( ( aExp == 0xFF ) && aSig ) ) { ! 2161: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 2162: } ! 2163: return (sbits64) LIT64( 0x8000000000000000 ); ! 2164: } ! 2165: if ( aExp ) aSig |= 0x00800000; ! 2166: aSig64 = aSig; ! 2167: aSig64 <<= 40; ! 2168: shift64ExtraRightJamming( aSig64, 0, shiftCount, &aSig64, &aSigExtra ); ! 2169: return roundAndPackInt64( aSign, aSig64, aSigExtra ); ! 2170: ! 2171: } ! 2172: ! 2173: /*---------------------------------------------------------------------------- ! 2174: | Returns the result of converting the single-precision floating-point value ! 2175: | `a' to the 64-bit two's complement integer format. The conversion is ! 2176: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2177: | Arithmetic, except that the conversion is always rounded toward zero. If ! 2178: | `a' is a NaN, the largest positive integer is returned. Otherwise, if the ! 2179: | conversion overflows, the largest integer with the same sign as `a' is ! 2180: | returned. ! 2181: *----------------------------------------------------------------------------*/ ! 2182: ! 2183: int64 float32_to_int64_round_to_zero( float32 a ) ! 2184: { ! 2185: flag aSign; ! 2186: int16 aExp, shiftCount; ! 2187: bits32 aSig; ! 2188: bits64 aSig64; ! 2189: int64 z; ! 2190: ! 2191: aSig = extractFloat32Frac( a ); ! 2192: aExp = extractFloat32Exp( a ); ! 2193: aSign = extractFloat32Sign( a ); ! 2194: shiftCount = aExp - 0xBE; ! 2195: if ( 0 <= shiftCount ) { ! 2196: if ( a != 0xDF000000 ) { ! 2197: float_raise( float_flag_invalid ); ! 2198: if ( ! aSign || ( ( aExp == 0xFF ) && aSig ) ) { ! 2199: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 2200: } ! 2201: } ! 2202: return (sbits64) LIT64( 0x8000000000000000 ); ! 2203: } ! 2204: else if ( aExp <= 0x7E ) { ! 2205: if ( aExp | aSig ) float_exception_flags |= float_flag_inexact; ! 2206: return 0; ! 2207: } ! 2208: aSig64 = aSig | 0x00800000; ! 2209: aSig64 <<= 40; ! 2210: z = aSig64>>( - shiftCount ); ! 2211: if ( (bits64) ( aSig64<<( shiftCount & 63 ) ) ) { ! 2212: float_exception_flags |= float_flag_inexact; ! 2213: } ! 2214: if ( aSign ) z = - z; ! 2215: return z; ! 2216: ! 2217: } ! 2218: ! 2219: /*---------------------------------------------------------------------------- ! 2220: | Returns the result of converting the single-precision floating-point value ! 2221: | `a' to the double-precision floating-point format. The conversion is ! 2222: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2223: | Arithmetic. ! 2224: *----------------------------------------------------------------------------*/ ! 2225: ! 2226: float64 float32_to_float64( float32 a ) ! 2227: { ! 2228: flag aSign; ! 2229: int16 aExp; ! 2230: bits32 aSig; ! 2231: ! 2232: aSig = extractFloat32Frac( a ); ! 2233: aExp = extractFloat32Exp( a ); ! 2234: aSign = extractFloat32Sign( a ); ! 2235: if ( aExp == 0xFF ) { ! 2236: if ( aSig ) return commonNaNToFloat64( float32ToCommonNaN( a ) ); ! 2237: return packFloat64( aSign, 0x7FF, 0 ); ! 2238: } ! 2239: if ( aExp == 0 ) { ! 2240: if ( aSig == 0 ) return packFloat64( aSign, 0, 0 ); ! 2241: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2242: --aExp; ! 2243: } ! 2244: return packFloat64( aSign, aExp + 0x380, ( (bits64) aSig )<<29 ); ! 2245: ! 2246: } ! 2247: ! 2248: #ifdef FLOATX80 ! 2249: ! 2250: /*---------------------------------------------------------------------------- ! 2251: | Returns the result of converting the single-precision floating-point value ! 2252: | `a' to the extended double-precision floating-point format. The conversion ! 2253: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2254: | Arithmetic. ! 2255: *----------------------------------------------------------------------------*/ ! 2256: ! 2257: floatx80 float32_to_floatx80( float32 a ) ! 2258: { ! 2259: flag aSign; ! 2260: int16 aExp; ! 2261: bits32 aSig; ! 2262: ! 2263: aSig = extractFloat32Frac( a ); ! 2264: aExp = extractFloat32Exp( a ); ! 2265: aSign = extractFloat32Sign( a ); ! 2266: if ( aExp == 0xFF ) { ! 2267: if ( aSig ) return commonNaNToFloatx80( float32ToCommonNaN( a ) ); ! 2268: return packFloatx80( aSign, 0x7FFF, floatx80_default_infinity_low ); ! 2269: } ! 2270: if ( aExp == 0 ) { ! 2271: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 2272: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2273: } ! 2274: aSig |= 0x00800000; ! 2275: return packFloatx80( aSign, aExp + 0x3F80, ( (bits64) aSig )<<40 ); ! 2276: ! 2277: } ! 2278: ! 2279: #ifdef SOFTFLOAT_68K // 31-12-2016: Added for Previous ! 2280: floatx80 float32_to_floatx80_allowunnormal( float32 a ) ! 2281: { ! 2282: flag aSign; ! 2283: int16 aExp; ! 2284: bits32 aSig; ! 2285: ! 2286: aSig = extractFloat32Frac( a ); ! 2287: aExp = extractFloat32Exp( a ); ! 2288: aSign = extractFloat32Sign( a ); ! 2289: if ( aExp == 0xFF ) { ! 2290: return packFloatx80( aSign, 0x7FFF, ( (bits64) aSig )<<40 ); ! 2291: } ! 2292: if ( aExp == 0 ) { ! 2293: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 2294: return packFloatx80( aSign, 0x3F81, ( (bits64) aSig )<<40 ); ! 2295: } ! 2296: aSig |= 0x00800000; ! 2297: return packFloatx80( aSign, aExp + 0x3F80, ( (bits64) aSig )<<40 ); ! 2298: ! 2299: } ! 2300: #endif // end of addition for Previous ! 2301: ! 2302: #endif ! 2303: ! 2304: #ifdef FLOAT128 ! 2305: ! 2306: /*---------------------------------------------------------------------------- ! 2307: | Returns the result of converting the single-precision floating-point value ! 2308: | `a' to the double-precision floating-point format. The conversion is ! 2309: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2310: | Arithmetic. ! 2311: *----------------------------------------------------------------------------*/ ! 2312: ! 2313: float128 float32_to_float128( float32 a ) ! 2314: { ! 2315: flag aSign; ! 2316: int16 aExp; ! 2317: bits32 aSig; ! 2318: ! 2319: aSig = extractFloat32Frac( a ); ! 2320: aExp = extractFloat32Exp( a ); ! 2321: aSign = extractFloat32Sign( a ); ! 2322: if ( aExp == 0xFF ) { ! 2323: if ( aSig ) return commonNaNToFloat128( float32ToCommonNaN( a ) ); ! 2324: return packFloat128( aSign, 0x7FFF, 0, 0 ); ! 2325: } ! 2326: if ( aExp == 0 ) { ! 2327: if ( aSig == 0 ) return packFloat128( aSign, 0, 0, 0 ); ! 2328: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2329: --aExp; ! 2330: } ! 2331: return packFloat128( aSign, aExp + 0x3F80, ( (bits64) aSig )<<25, 0 ); ! 2332: ! 2333: } ! 2334: ! 2335: #endif ! 2336: ! 2337: /*---------------------------------------------------------------------------- ! 2338: | Rounds the single-precision floating-point value `a' to an integer, and ! 2339: | returns the result as a single-precision floating-point value. The ! 2340: | operation is performed according to the IEC/IEEE Standard for Binary ! 2341: | Floating-Point Arithmetic. ! 2342: *----------------------------------------------------------------------------*/ ! 2343: ! 2344: float32 float32_round_to_int( float32 a ) ! 2345: { ! 2346: flag aSign; ! 2347: int16 aExp; ! 2348: bits32 lastBitMask, roundBitsMask; ! 2349: int8 roundingMode; ! 2350: float32 z; ! 2351: ! 2352: aExp = extractFloat32Exp( a ); ! 2353: if ( 0x96 <= aExp ) { ! 2354: if ( ( aExp == 0xFF ) && extractFloat32Frac( a ) ) { ! 2355: return propagateFloat32NaN( a, a ); ! 2356: } ! 2357: return a; ! 2358: } ! 2359: if ( aExp <= 0x7E ) { ! 2360: if ( (bits32) ( a<<1 ) == 0 ) return a; ! 2361: float_exception_flags |= float_flag_inexact; ! 2362: aSign = extractFloat32Sign( a ); ! 2363: switch ( float_rounding_mode ) { ! 2364: case float_round_nearest_even: ! 2365: if ( ( aExp == 0x7E ) && extractFloat32Frac( a ) ) { ! 2366: return packFloat32( aSign, 0x7F, 0 ); ! 2367: } ! 2368: break; ! 2369: case float_round_down: ! 2370: return aSign ? 0xBF800000 : 0; ! 2371: case float_round_up: ! 2372: return aSign ? 0x80000000 : 0x3F800000; ! 2373: } ! 2374: return packFloat32( aSign, 0, 0 ); ! 2375: } ! 2376: lastBitMask = 1; ! 2377: lastBitMask <<= 0x96 - aExp; ! 2378: roundBitsMask = lastBitMask - 1; ! 2379: z = a; ! 2380: roundingMode = float_rounding_mode; ! 2381: if ( roundingMode == float_round_nearest_even ) { ! 2382: z += lastBitMask>>1; ! 2383: if ( ( z & roundBitsMask ) == 0 ) z &= ~ lastBitMask; ! 2384: } ! 2385: else if ( roundingMode != float_round_to_zero ) { ! 2386: if ( extractFloat32Sign( z ) ^ ( roundingMode == float_round_up ) ) { ! 2387: z += roundBitsMask; ! 2388: } ! 2389: } ! 2390: z &= ~ roundBitsMask; ! 2391: if ( z != a ) float_exception_flags |= float_flag_inexact; ! 2392: return z; ! 2393: ! 2394: } ! 2395: ! 2396: /*---------------------------------------------------------------------------- ! 2397: | Returns the result of adding the absolute values of the single-precision ! 2398: | floating-point values `a' and `b'. If `zSign' is 1, the sum is negated ! 2399: | before being returned. `zSign' is ignored if the result is a NaN. ! 2400: | The addition is performed according to the IEC/IEEE Standard for Binary ! 2401: | Floating-Point Arithmetic. ! 2402: *----------------------------------------------------------------------------*/ ! 2403: ! 2404: static float32 addFloat32Sigs( float32 a, float32 b, flag zSign ) ! 2405: { ! 2406: int16 aExp, bExp, zExp; ! 2407: bits32 aSig, bSig, zSig; ! 2408: int16 expDiff; ! 2409: ! 2410: aSig = extractFloat32Frac( a ); ! 2411: aExp = extractFloat32Exp( a ); ! 2412: bSig = extractFloat32Frac( b ); ! 2413: bExp = extractFloat32Exp( b ); ! 2414: expDiff = aExp - bExp; ! 2415: aSig <<= 6; ! 2416: bSig <<= 6; ! 2417: if ( 0 < expDiff ) { ! 2418: if ( aExp == 0xFF ) { ! 2419: if ( aSig ) return propagateFloat32NaN( a, b ); ! 2420: return a; ! 2421: } ! 2422: if ( bExp == 0 ) { ! 2423: --expDiff; ! 2424: } ! 2425: else { ! 2426: bSig |= 0x20000000; ! 2427: } ! 2428: shift32RightJamming( bSig, expDiff, &bSig ); ! 2429: zExp = aExp; ! 2430: } ! 2431: else if ( expDiff < 0 ) { ! 2432: if ( bExp == 0xFF ) { ! 2433: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2434: return packFloat32( zSign, 0xFF, 0 ); ! 2435: } ! 2436: if ( aExp == 0 ) { ! 2437: ++expDiff; ! 2438: } ! 2439: else { ! 2440: aSig |= 0x20000000; ! 2441: } ! 2442: shift32RightJamming( aSig, - expDiff, &aSig ); ! 2443: zExp = bExp; ! 2444: } ! 2445: else { ! 2446: if ( aExp == 0xFF ) { ! 2447: if ( aSig | bSig ) return propagateFloat32NaN( a, b ); ! 2448: return a; ! 2449: } ! 2450: if ( aExp == 0 ) return packFloat32( zSign, 0, ( aSig + bSig )>>6 ); ! 2451: zSig = 0x40000000 + aSig + bSig; ! 2452: zExp = aExp; ! 2453: goto roundAndPack; ! 2454: } ! 2455: aSig |= 0x20000000; ! 2456: zSig = ( aSig + bSig )<<1; ! 2457: --zExp; ! 2458: if ( (sbits32) zSig < 0 ) { ! 2459: zSig = aSig + bSig; ! 2460: ++zExp; ! 2461: } ! 2462: roundAndPack: ! 2463: #ifdef SOFTFLOAT_I860 ! 2464: return roundAndPackFloat32_2( zSign, zExp, zSig ); ! 2465: #else ! 2466: return roundAndPackFloat32( zSign, zExp, zSig ); ! 2467: #endif ! 2468: ! 2469: } ! 2470: ! 2471: /*---------------------------------------------------------------------------- ! 2472: | Returns the result of subtracting the absolute values of the single- ! 2473: | precision floating-point values `a' and `b'. If `zSign' is 1, the ! 2474: | difference is negated before being returned. `zSign' is ignored if the ! 2475: | result is a NaN. The subtraction is performed according to the IEC/IEEE ! 2476: | Standard for Binary Floating-Point Arithmetic. ! 2477: *----------------------------------------------------------------------------*/ ! 2478: ! 2479: static float32 subFloat32Sigs( float32 a, float32 b, flag zSign ) ! 2480: { ! 2481: int16 aExp, bExp, zExp; ! 2482: bits32 aSig, bSig, zSig; ! 2483: int16 expDiff; ! 2484: #ifdef SOFTFLOAT_I860 ! 2485: int8 shiftCount; ! 2486: #endif ! 2487: ! 2488: aSig = extractFloat32Frac( a ); ! 2489: aExp = extractFloat32Exp( a ); ! 2490: bSig = extractFloat32Frac( b ); ! 2491: bExp = extractFloat32Exp( b ); ! 2492: expDiff = aExp - bExp; ! 2493: aSig <<= 7; ! 2494: bSig <<= 7; ! 2495: if ( 0 < expDiff ) goto aExpBigger; ! 2496: if ( expDiff < 0 ) goto bExpBigger; ! 2497: if ( aExp == 0xFF ) { ! 2498: if ( aSig | bSig ) return propagateFloat32NaN( a, b ); ! 2499: #ifdef SOFTFLOAT_I860 ! 2500: float_raise2( float_flag_invalid ); ! 2501: #else ! 2502: float_raise( float_flag_invalid ); ! 2503: #endif ! 2504: return float32_default_nan; ! 2505: } ! 2506: if ( aExp == 0 ) { ! 2507: aExp = 1; ! 2508: bExp = 1; ! 2509: } ! 2510: if ( bSig < aSig ) goto aBigger; ! 2511: if ( aSig < bSig ) goto bBigger; ! 2512: #ifdef SOFTFLOAT_I860 ! 2513: return packFloat32( float_rounding_mode2 == float_round_down, 0, 0 ); ! 2514: #else ! 2515: return packFloat32( float_rounding_mode == float_round_down, 0, 0 ); ! 2516: #endif ! 2517: bExpBigger: ! 2518: if ( bExp == 0xFF ) { ! 2519: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2520: return packFloat32( zSign ^ 1, 0xFF, 0 ); ! 2521: } ! 2522: if ( aExp == 0 ) { ! 2523: ++expDiff; ! 2524: } ! 2525: else { ! 2526: aSig |= 0x40000000; ! 2527: } ! 2528: shift32RightJamming( aSig, - expDiff, &aSig ); ! 2529: bSig |= 0x40000000; ! 2530: bBigger: ! 2531: zSig = bSig - aSig; ! 2532: zExp = bExp; ! 2533: zSign ^= 1; ! 2534: goto normalizeRoundAndPack; ! 2535: aExpBigger: ! 2536: if ( aExp == 0xFF ) { ! 2537: if ( aSig ) return propagateFloat32NaN( a, b ); ! 2538: return a; ! 2539: } ! 2540: if ( bExp == 0 ) { ! 2541: --expDiff; ! 2542: } ! 2543: else { ! 2544: bSig |= 0x40000000; ! 2545: } ! 2546: shift32RightJamming( bSig, expDiff, &bSig ); ! 2547: aSig |= 0x40000000; ! 2548: aBigger: ! 2549: zSig = aSig - bSig; ! 2550: zExp = aExp; ! 2551: normalizeRoundAndPack: ! 2552: --zExp; ! 2553: #ifdef SOFTFLOAT_I860 ! 2554: shiftCount = countLeadingZeros32( zSig ) - 1; ! 2555: return roundAndPackFloat32_2( zSign, zExp - shiftCount, zSig<<shiftCount ); ! 2556: #else ! 2557: return normalizeRoundAndPackFloat32( zSign, zExp, zSig ); ! 2558: #endif ! 2559: ! 2560: } ! 2561: ! 2562: /*---------------------------------------------------------------------------- ! 2563: | Returns the result of adding the single-precision floating-point values `a' ! 2564: | and `b'. The operation is performed according to the IEC/IEEE Standard for ! 2565: | Binary Floating-Point Arithmetic. ! 2566: *----------------------------------------------------------------------------*/ ! 2567: ! 2568: float32 float32_add( float32 a, float32 b ) ! 2569: { ! 2570: flag aSign, bSign; ! 2571: ! 2572: aSign = extractFloat32Sign( a ); ! 2573: bSign = extractFloat32Sign( b ); ! 2574: if ( aSign == bSign ) { ! 2575: return addFloat32Sigs( a, b, aSign ); ! 2576: } ! 2577: else { ! 2578: return subFloat32Sigs( a, b, aSign ); ! 2579: } ! 2580: ! 2581: } ! 2582: ! 2583: /*---------------------------------------------------------------------------- ! 2584: | Returns the result of subtracting the single-precision floating-point values ! 2585: | `a' and `b'. The operation is performed according to the IEC/IEEE Standard ! 2586: | for Binary Floating-Point Arithmetic. ! 2587: *----------------------------------------------------------------------------*/ ! 2588: ! 2589: float32 float32_sub( float32 a, float32 b ) ! 2590: { ! 2591: flag aSign, bSign; ! 2592: ! 2593: aSign = extractFloat32Sign( a ); ! 2594: bSign = extractFloat32Sign( b ); ! 2595: if ( aSign == bSign ) { ! 2596: return subFloat32Sigs( a, b, aSign ); ! 2597: } ! 2598: else { ! 2599: return addFloat32Sigs( a, b, aSign ); ! 2600: } ! 2601: ! 2602: } ! 2603: ! 2604: /*---------------------------------------------------------------------------- ! 2605: | Returns the result of multiplying the single-precision floating-point values ! 2606: | `a' and `b'. The operation is performed according to the IEC/IEEE Standard ! 2607: | for Binary Floating-Point Arithmetic. ! 2608: *----------------------------------------------------------------------------*/ ! 2609: ! 2610: float32 float32_mul( float32 a, float32 b ) ! 2611: { ! 2612: flag aSign, bSign, zSign; ! 2613: int16 aExp, bExp, zExp; ! 2614: bits32 aSig, bSig; ! 2615: bits64 zSig64; ! 2616: bits32 zSig; ! 2617: ! 2618: aSig = extractFloat32Frac( a ); ! 2619: aExp = extractFloat32Exp( a ); ! 2620: aSign = extractFloat32Sign( a ); ! 2621: bSig = extractFloat32Frac( b ); ! 2622: bExp = extractFloat32Exp( b ); ! 2623: bSign = extractFloat32Sign( b ); ! 2624: zSign = aSign ^ bSign; ! 2625: if ( aExp == 0xFF ) { ! 2626: if ( aSig || ( ( bExp == 0xFF ) && bSig ) ) { ! 2627: return propagateFloat32NaN( a, b ); ! 2628: } ! 2629: if ( ( bExp | bSig ) == 0 ) { ! 2630: #ifdef SOFTFLOAT_I860 ! 2631: float_raise2( float_flag_invalid ); ! 2632: #else ! 2633: float_raise( float_flag_invalid ); ! 2634: #endif ! 2635: return float32_default_nan; ! 2636: } ! 2637: return packFloat32( zSign, 0xFF, 0 ); ! 2638: } ! 2639: if ( bExp == 0xFF ) { ! 2640: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2641: if ( ( aExp | aSig ) == 0 ) { ! 2642: #ifdef SOFTFLOAT_I860 ! 2643: float_raise2( float_flag_invalid ); ! 2644: #else ! 2645: float_raise( float_flag_invalid ); ! 2646: #endif ! 2647: return float32_default_nan; ! 2648: } ! 2649: return packFloat32( zSign, 0xFF, 0 ); ! 2650: } ! 2651: if ( aExp == 0 ) { ! 2652: if ( aSig == 0 ) return packFloat32( zSign, 0, 0 ); ! 2653: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2654: } ! 2655: if ( bExp == 0 ) { ! 2656: if ( bSig == 0 ) return packFloat32( zSign, 0, 0 ); ! 2657: normalizeFloat32Subnormal( bSig, &bExp, &bSig ); ! 2658: } ! 2659: zExp = aExp + bExp - 0x7F; ! 2660: aSig = ( aSig | 0x00800000 )<<7; ! 2661: bSig = ( bSig | 0x00800000 )<<8; ! 2662: shift64RightJamming( ( (bits64) aSig ) * bSig, 32, &zSig64 ); ! 2663: zSig = zSig64; ! 2664: if ( 0 <= (sbits32) ( zSig<<1 ) ) { ! 2665: zSig <<= 1; ! 2666: --zExp; ! 2667: } ! 2668: #ifdef SOFTFLOAT_I860 ! 2669: return roundAndPackFloat32_2( zSign, zExp, zSig ); ! 2670: #else ! 2671: return roundAndPackFloat32( zSign, zExp, zSig ); ! 2672: #endif ! 2673: ! 2674: } ! 2675: ! 2676: /*---------------------------------------------------------------------------- ! 2677: | Returns the result of dividing the single-precision floating-point value `a' ! 2678: | by the corresponding value `b'. The operation is performed according to the ! 2679: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 2680: *----------------------------------------------------------------------------*/ ! 2681: ! 2682: float32 float32_div( float32 a, float32 b ) ! 2683: { ! 2684: flag aSign, bSign, zSign; ! 2685: int16 aExp, bExp, zExp; ! 2686: bits32 aSig, bSig, zSig; ! 2687: ! 2688: aSig = extractFloat32Frac( a ); ! 2689: aExp = extractFloat32Exp( a ); ! 2690: aSign = extractFloat32Sign( a ); ! 2691: bSig = extractFloat32Frac( b ); ! 2692: bExp = extractFloat32Exp( b ); ! 2693: bSign = extractFloat32Sign( b ); ! 2694: zSign = aSign ^ bSign; ! 2695: if ( aExp == 0xFF ) { ! 2696: if ( aSig ) return propagateFloat32NaN( a, b ); ! 2697: if ( bExp == 0xFF ) { ! 2698: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2699: #ifdef SOFTFLOAT_I860 ! 2700: float_raise2( float_flag_invalid ); ! 2701: #else ! 2702: float_raise( float_flag_invalid ); ! 2703: #endif ! 2704: return float32_default_nan; ! 2705: } ! 2706: return packFloat32( zSign, 0xFF, 0 ); ! 2707: } ! 2708: if ( bExp == 0xFF ) { ! 2709: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2710: return packFloat32( zSign, 0, 0 ); ! 2711: } ! 2712: if ( bExp == 0 ) { ! 2713: if ( bSig == 0 ) { ! 2714: if ( ( aExp | aSig ) == 0 ) { ! 2715: #ifdef SOFTFLOAT_I860 ! 2716: float_raise2( float_flag_invalid ); ! 2717: #else ! 2718: float_raise( float_flag_invalid ); ! 2719: #endif ! 2720: return float32_default_nan; ! 2721: } ! 2722: #ifdef SOFTFLOAT_I860 ! 2723: float_raise2( float_flag_divbyzero ); ! 2724: #else ! 2725: float_raise( float_flag_divbyzero ); ! 2726: #endif ! 2727: return packFloat32( zSign, 0xFF, 0 ); ! 2728: } ! 2729: normalizeFloat32Subnormal( bSig, &bExp, &bSig ); ! 2730: } ! 2731: if ( aExp == 0 ) { ! 2732: if ( aSig == 0 ) return packFloat32( zSign, 0, 0 ); ! 2733: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2734: } ! 2735: zExp = aExp - bExp + 0x7D; ! 2736: aSig = ( aSig | 0x00800000 )<<7; ! 2737: bSig = ( bSig | 0x00800000 )<<8; ! 2738: if ( bSig <= ( aSig + aSig ) ) { ! 2739: aSig >>= 1; ! 2740: ++zExp; ! 2741: } ! 2742: zSig = ( ( (bits64) aSig )<<32 ) / bSig; ! 2743: if ( ( zSig & 0x3F ) == 0 ) { ! 2744: zSig |= ( (bits64) bSig * zSig != ( (bits64) aSig )<<32 ); ! 2745: } ! 2746: #ifdef SOFTFLOAT_I860 ! 2747: return roundAndPackFloat32_2( zSign, zExp, zSig ); ! 2748: #else ! 2749: return roundAndPackFloat32( zSign, zExp, zSig ); ! 2750: #endif ! 2751: ! 2752: } ! 2753: ! 2754: /*---------------------------------------------------------------------------- ! 2755: | Returns the remainder of the single-precision floating-point value `a' ! 2756: | with respect to the corresponding value `b'. The operation is performed ! 2757: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 2758: *----------------------------------------------------------------------------*/ ! 2759: ! 2760: float32 float32_rem( float32 a, float32 b ) ! 2761: { ! 2762: flag aSign, zSign; ! 2763: int16 aExp, bExp, expDiff; ! 2764: bits32 aSig, bSig; ! 2765: bits32 q; ! 2766: bits64 aSig64, bSig64, q64; ! 2767: bits32 alternateASig; ! 2768: sbits32 sigMean; ! 2769: ! 2770: aSig = extractFloat32Frac( a ); ! 2771: aExp = extractFloat32Exp( a ); ! 2772: aSign = extractFloat32Sign( a ); ! 2773: bSig = extractFloat32Frac( b ); ! 2774: bExp = extractFloat32Exp( b ); ! 2775: // bSign = extractFloat32Sign( b ); ! 2776: if ( aExp == 0xFF ) { ! 2777: if ( aSig || ( ( bExp == 0xFF ) && bSig ) ) { ! 2778: return propagateFloat32NaN( a, b ); ! 2779: } ! 2780: float_raise( float_flag_invalid ); ! 2781: return float32_default_nan; ! 2782: } ! 2783: if ( bExp == 0xFF ) { ! 2784: if ( bSig ) return propagateFloat32NaN( a, b ); ! 2785: return a; ! 2786: } ! 2787: if ( bExp == 0 ) { ! 2788: if ( bSig == 0 ) { ! 2789: float_raise( float_flag_invalid ); ! 2790: return float32_default_nan; ! 2791: } ! 2792: normalizeFloat32Subnormal( bSig, &bExp, &bSig ); ! 2793: } ! 2794: if ( aExp == 0 ) { ! 2795: if ( aSig == 0 ) return a; ! 2796: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2797: } ! 2798: expDiff = aExp - bExp; ! 2799: aSig |= 0x00800000; ! 2800: bSig |= 0x00800000; ! 2801: if ( expDiff < 32 ) { ! 2802: aSig <<= 8; ! 2803: bSig <<= 8; ! 2804: if ( expDiff < 0 ) { ! 2805: if ( expDiff < -1 ) return a; ! 2806: aSig >>= 1; ! 2807: } ! 2808: q = ( bSig <= aSig ); ! 2809: if ( q ) aSig -= bSig; ! 2810: if ( 0 < expDiff ) { ! 2811: q = ( ( (bits64) aSig )<<32 ) / bSig; ! 2812: q >>= 32 - expDiff; ! 2813: bSig >>= 2; ! 2814: aSig = ( ( aSig>>1 )<<( expDiff - 1 ) ) - bSig * q; ! 2815: } ! 2816: else { ! 2817: aSig >>= 2; ! 2818: bSig >>= 2; ! 2819: } ! 2820: } ! 2821: else { ! 2822: if ( bSig <= aSig ) aSig -= bSig; ! 2823: aSig64 = ( (bits64) aSig )<<40; ! 2824: bSig64 = ( (bits64) bSig )<<40; ! 2825: expDiff -= 64; ! 2826: while ( 0 < expDiff ) { ! 2827: q64 = estimateDiv128To64( aSig64, 0, bSig64 ); ! 2828: q64 = ( 2 < q64 ) ? q64 - 2 : 0; ! 2829: aSig64 = - ( ( bSig * q64 )<<38 ); ! 2830: expDiff -= 62; ! 2831: } ! 2832: expDiff += 64; ! 2833: q64 = estimateDiv128To64( aSig64, 0, bSig64 ); ! 2834: q64 = ( 2 < q64 ) ? q64 - 2 : 0; ! 2835: q = q64>>( 64 - expDiff ); ! 2836: bSig <<= 6; ! 2837: aSig = ( ( aSig64>>33 )<<( expDiff - 1 ) ) - bSig * q; ! 2838: } ! 2839: do { ! 2840: alternateASig = aSig; ! 2841: ++q; ! 2842: aSig -= bSig; ! 2843: } while ( 0 <= (sbits32) aSig ); ! 2844: sigMean = aSig + alternateASig; ! 2845: if ( ( sigMean < 0 ) || ( ( sigMean == 0 ) && ( q & 1 ) ) ) { ! 2846: aSig = alternateASig; ! 2847: } ! 2848: zSign = ( (sbits32) aSig < 0 ); ! 2849: if ( zSign ) aSig = - aSig; ! 2850: return normalizeRoundAndPackFloat32( aSign ^ zSign, bExp, aSig ); ! 2851: ! 2852: } ! 2853: ! 2854: /*---------------------------------------------------------------------------- ! 2855: | Returns the square root of the single-precision floating-point value `a'. ! 2856: | The operation is performed according to the IEC/IEEE Standard for Binary ! 2857: | Floating-Point Arithmetic. ! 2858: *----------------------------------------------------------------------------*/ ! 2859: ! 2860: float32 float32_sqrt( float32 a ) ! 2861: { ! 2862: flag aSign; ! 2863: int16 aExp, zExp; ! 2864: bits32 aSig, zSig; ! 2865: bits64 rem, term; ! 2866: ! 2867: aSig = extractFloat32Frac( a ); ! 2868: aExp = extractFloat32Exp( a ); ! 2869: aSign = extractFloat32Sign( a ); ! 2870: if ( aExp == 0xFF ) { ! 2871: if ( aSig ) return propagateFloat32NaN( a, 0 ); ! 2872: if ( ! aSign ) return a; ! 2873: #ifdef SOFTFLOAT_I860 ! 2874: float_raise2( float_flag_invalid ); ! 2875: #else ! 2876: float_raise( float_flag_invalid ); ! 2877: #endif ! 2878: return float32_default_nan; ! 2879: } ! 2880: if ( aSign ) { ! 2881: if ( ( aExp | aSig ) == 0 ) return a; ! 2882: #ifdef SOFTFLOAT_I860 ! 2883: float_raise2( float_flag_invalid ); ! 2884: #else ! 2885: float_raise( float_flag_invalid ); ! 2886: #endif ! 2887: return float32_default_nan; ! 2888: } ! 2889: if ( aExp == 0 ) { ! 2890: if ( aSig == 0 ) return 0; ! 2891: normalizeFloat32Subnormal( aSig, &aExp, &aSig ); ! 2892: } ! 2893: zExp = ( ( aExp - 0x7F )>>1 ) + 0x7E; ! 2894: aSig = ( aSig | 0x00800000 )<<8; ! 2895: zSig = estimateSqrt32( aExp, aSig ) + 2; ! 2896: if ( ( zSig & 0x7F ) <= 5 ) { ! 2897: if ( zSig < 2 ) { ! 2898: zSig = 0x7FFFFFFF; ! 2899: goto roundAndPack; ! 2900: } ! 2901: aSig >>= aExp & 1; ! 2902: term = ( (bits64) zSig ) * zSig; ! 2903: rem = ( ( (bits64) aSig )<<32 ) - term; ! 2904: while ( (sbits64) rem < 0 ) { ! 2905: --zSig; ! 2906: rem += ( ( (bits64) zSig )<<1 ) | 1; ! 2907: } ! 2908: zSig |= ( rem != 0 ); ! 2909: } ! 2910: shift32RightJamming( zSig, 1, &zSig ); ! 2911: roundAndPack: ! 2912: #ifdef SOFTFLOAT_I860 ! 2913: return roundAndPackFloat32_2( 0, zExp, zSig ); ! 2914: #else ! 2915: return roundAndPackFloat32( 0, zExp, zSig ); ! 2916: #endif ! 2917: ! 2918: } ! 2919: ! 2920: /*---------------------------------------------------------------------------- ! 2921: | Returns 1 if the single-precision floating-point value `a' is equal to ! 2922: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 2923: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 2924: *----------------------------------------------------------------------------*/ ! 2925: ! 2926: flag float32_eq( float32 a, float32 b ) ! 2927: { ! 2928: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 2929: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 2930: ) { ! 2931: if ( float32_is_signaling_nan( a ) || float32_is_signaling_nan( b ) ) { ! 2932: #ifdef SOFTFLOAT_I860 ! 2933: float_raise2( float_flag_invalid ); ! 2934: #else ! 2935: float_raise( float_flag_invalid ); ! 2936: #endif ! 2937: } ! 2938: return 0; ! 2939: } ! 2940: return ( a == b ) || ( (bits32) ( ( a | b )<<1 ) == 0 ); ! 2941: ! 2942: } ! 2943: ! 2944: /*---------------------------------------------------------------------------- ! 2945: | Returns 1 if the single-precision floating-point value `a' is less than ! 2946: | or equal to the corresponding value `b', and 0 otherwise. The comparison ! 2947: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 2948: | Arithmetic. ! 2949: *----------------------------------------------------------------------------*/ ! 2950: ! 2951: flag float32_le( float32 a, float32 b ) ! 2952: { ! 2953: flag aSign, bSign; ! 2954: ! 2955: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 2956: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 2957: ) { ! 2958: #ifdef SOFTFLOAT_I860 ! 2959: float_raise2( float_flag_invalid ); ! 2960: #else ! 2961: float_raise( float_flag_invalid ); ! 2962: #endif ! 2963: return 0; ! 2964: } ! 2965: aSign = extractFloat32Sign( a ); ! 2966: bSign = extractFloat32Sign( b ); ! 2967: if ( aSign != bSign ) return aSign || ( (bits32) ( ( a | b )<<1 ) == 0 ); ! 2968: return ( a == b ) || ( aSign ^ ( a < b ) ); ! 2969: ! 2970: } ! 2971: ! 2972: /*---------------------------------------------------------------------------- ! 2973: | Returns 1 if the single-precision floating-point value `a' is less than ! 2974: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 2975: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 2976: *----------------------------------------------------------------------------*/ ! 2977: ! 2978: flag float32_lt( float32 a, float32 b ) ! 2979: { ! 2980: flag aSign, bSign; ! 2981: ! 2982: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 2983: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 2984: ) { ! 2985: #ifdef SOFTFLOAT_I860 ! 2986: float_raise2( float_flag_invalid ); ! 2987: #else ! 2988: float_raise( float_flag_invalid ); ! 2989: #endif ! 2990: return 0; ! 2991: } ! 2992: aSign = extractFloat32Sign( a ); ! 2993: bSign = extractFloat32Sign( b ); ! 2994: if ( aSign != bSign ) return aSign && ( (bits32) ( ( a | b )<<1 ) != 0 ); ! 2995: return ( a != b ) && ( aSign ^ ( a < b ) ); ! 2996: ! 2997: } ! 2998: ! 2999: #ifdef SOFTFLOAT_I860 // 29-04-2017: Added for Previous ! 3000: /*---------------------------------------------------------------------------- ! 3001: | Returns 1 if the single-precision floating-point value `a' is greater than ! 3002: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 3003: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3004: *----------------------------------------------------------------------------*/ ! 3005: ! 3006: flag float32_gt( float32 a, float32 b ) ! 3007: { ! 3008: flag aSign, bSign; ! 3009: ! 3010: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 3011: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 3012: ) { ! 3013: #ifdef SOFTFLOAT_I860 ! 3014: float_raise2( float_flag_invalid ); ! 3015: #else ! 3016: float_raise( float_flag_invalid ); ! 3017: #endif ! 3018: return 0; ! 3019: } ! 3020: aSign = extractFloat32Sign( a ); ! 3021: bSign = extractFloat32Sign( b ); ! 3022: if ( aSign != bSign ) return bSign && ( (bits32) ( ( a | b )<<1 ) != 0 ); ! 3023: return ( a != b ) && ( bSign ^ ( a > b ) ); ! 3024: ! 3025: } ! 3026: #endif // End of addition for Previous ! 3027: ! 3028: /*---------------------------------------------------------------------------- ! 3029: | Returns 1 if the single-precision floating-point value `a' is equal to ! 3030: | the corresponding value `b', and 0 otherwise. The invalid exception is ! 3031: | raised if either operand is a NaN. Otherwise, the comparison is performed ! 3032: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3033: *----------------------------------------------------------------------------*/ ! 3034: ! 3035: flag float32_eq_signaling( float32 a, float32 b ) ! 3036: { ! 3037: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 3038: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 3039: ) { ! 3040: float_raise( float_flag_invalid ); ! 3041: return 0; ! 3042: } ! 3043: return ( a == b ) || ( (bits32) ( ( a | b )<<1 ) == 0 ); ! 3044: ! 3045: } ! 3046: ! 3047: /*---------------------------------------------------------------------------- ! 3048: | Returns 1 if the single-precision floating-point value `a' is less than or ! 3049: | equal to the corresponding value `b', and 0 otherwise. Quiet NaNs do not ! 3050: | cause an exception. Otherwise, the comparison is performed according to the ! 3051: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3052: *----------------------------------------------------------------------------*/ ! 3053: ! 3054: flag float32_le_quiet( float32 a, float32 b ) ! 3055: { ! 3056: flag aSign, bSign; ! 3057: // int16 aExp, bExp; ! 3058: ! 3059: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 3060: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 3061: ) { ! 3062: if ( float32_is_signaling_nan( a ) || float32_is_signaling_nan( b ) ) { ! 3063: float_raise( float_flag_invalid ); ! 3064: } ! 3065: return 0; ! 3066: } ! 3067: aSign = extractFloat32Sign( a ); ! 3068: bSign = extractFloat32Sign( b ); ! 3069: if ( aSign != bSign ) return aSign || ( (bits32) ( ( a | b )<<1 ) == 0 ); ! 3070: return ( a == b ) || ( aSign ^ ( a < b ) ); ! 3071: ! 3072: } ! 3073: ! 3074: /*---------------------------------------------------------------------------- ! 3075: | Returns 1 if the single-precision floating-point value `a' is less than ! 3076: | the corresponding value `b', and 0 otherwise. Quiet NaNs do not cause an ! 3077: | exception. Otherwise, the comparison is performed according to the IEC/IEEE ! 3078: | Standard for Binary Floating-Point Arithmetic. ! 3079: *----------------------------------------------------------------------------*/ ! 3080: ! 3081: flag float32_lt_quiet( float32 a, float32 b ) ! 3082: { ! 3083: flag aSign, bSign; ! 3084: ! 3085: if ( ( ( extractFloat32Exp( a ) == 0xFF ) && extractFloat32Frac( a ) ) ! 3086: || ( ( extractFloat32Exp( b ) == 0xFF ) && extractFloat32Frac( b ) ) ! 3087: ) { ! 3088: if ( float32_is_signaling_nan( a ) || float32_is_signaling_nan( b ) ) { ! 3089: float_raise( float_flag_invalid ); ! 3090: } ! 3091: return 0; ! 3092: } ! 3093: aSign = extractFloat32Sign( a ); ! 3094: bSign = extractFloat32Sign( b ); ! 3095: if ( aSign != bSign ) return aSign && ( (bits32) ( ( a | b )<<1 ) != 0 ); ! 3096: return ( a != b ) && ( aSign ^ ( a < b ) ); ! 3097: ! 3098: } ! 3099: ! 3100: /*---------------------------------------------------------------------------- ! 3101: | Returns the result of converting the double-precision floating-point value ! 3102: | `a' to the 32-bit two's complement integer format. The conversion is ! 3103: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3104: | Arithmetic---which means in particular that the conversion is rounded ! 3105: | according to the current rounding mode. If `a' is a NaN, the largest ! 3106: | positive integer is returned. Otherwise, if the conversion overflows, the ! 3107: | largest integer with the same sign as `a' is returned. ! 3108: *----------------------------------------------------------------------------*/ ! 3109: ! 3110: int32 float64_to_int32( float64 a ) ! 3111: { ! 3112: flag aSign; ! 3113: int16 aExp, shiftCount; ! 3114: bits64 aSig; ! 3115: ! 3116: aSig = extractFloat64Frac( a ); ! 3117: aExp = extractFloat64Exp( a ); ! 3118: aSign = extractFloat64Sign( a ); ! 3119: if ( ( aExp == 0x7FF ) && aSig ) aSign = 0; ! 3120: if ( aExp ) aSig |= LIT64( 0x0010000000000000 ); ! 3121: shiftCount = 0x42C - aExp; ! 3122: if ( 0 < shiftCount ) shift64RightJamming( aSig, shiftCount, &aSig ); ! 3123: #ifdef SOFTFLOAT_I860 ! 3124: return roundAndPackInt32_2( aSign, aSig ); ! 3125: #else ! 3126: return roundAndPackInt32( aSign, aSig ); ! 3127: #endif ! 3128: ! 3129: } ! 3130: ! 3131: /*---------------------------------------------------------------------------- ! 3132: | Returns the result of converting the double-precision floating-point value ! 3133: | `a' to the 32-bit two's complement integer format. The conversion is ! 3134: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3135: | Arithmetic, except that the conversion is always rounded toward zero. ! 3136: | If `a' is a NaN, the largest positive integer is returned. Otherwise, if ! 3137: | the conversion overflows, the largest integer with the same sign as `a' is ! 3138: | returned. ! 3139: *----------------------------------------------------------------------------*/ ! 3140: ! 3141: int32 float64_to_int32_round_to_zero( float64 a ) ! 3142: { ! 3143: flag aSign; ! 3144: int16 aExp, shiftCount; ! 3145: bits64 aSig, savedASig; ! 3146: int32 z; ! 3147: ! 3148: aSig = extractFloat64Frac( a ); ! 3149: aExp = extractFloat64Exp( a ); ! 3150: aSign = extractFloat64Sign( a ); ! 3151: if ( 0x41E < aExp ) { ! 3152: if ( ( aExp == 0x7FF ) && aSig ) aSign = 0; ! 3153: goto invalid; ! 3154: } ! 3155: else if ( aExp < 0x3FF ) { ! 3156: #ifdef SOFTFLOAT_I860 ! 3157: if ( aExp || aSig ) float_raise2( float_flag_inexact ); ! 3158: #else ! 3159: if ( aExp || aSig ) float_exception_flags |= float_flag_inexact; ! 3160: #endif ! 3161: return 0; ! 3162: } ! 3163: aSig |= LIT64( 0x0010000000000000 ); ! 3164: shiftCount = 0x433 - aExp; ! 3165: savedASig = aSig; ! 3166: aSig >>= shiftCount; ! 3167: z = aSig; ! 3168: if ( aSign ) z = - z; ! 3169: z = (sbits32) z; ! 3170: if ( ( z < 0 ) ^ aSign ) { ! 3171: invalid: ! 3172: #ifdef SOFTFLOAT_I860 ! 3173: float_raise2( float_flag_invalid ); ! 3174: #else ! 3175: float_raise( float_flag_invalid ); ! 3176: #endif ! 3177: return aSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 3178: } ! 3179: if ( ( aSig<<shiftCount ) != savedASig ) { ! 3180: #ifdef SOFTFLOAT_I860 ! 3181: float_raise2( float_flag_inexact ); ! 3182: #else ! 3183: float_exception_flags |= float_flag_inexact; ! 3184: #endif ! 3185: } ! 3186: return z; ! 3187: ! 3188: } ! 3189: ! 3190: /*---------------------------------------------------------------------------- ! 3191: | Returns the result of converting the double-precision floating-point value ! 3192: | `a' to the 64-bit two's complement integer format. The conversion is ! 3193: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3194: | Arithmetic---which means in particular that the conversion is rounded ! 3195: | according to the current rounding mode. If `a' is a NaN, the largest ! 3196: | positive integer is returned. Otherwise, if the conversion overflows, the ! 3197: | largest integer with the same sign as `a' is returned. ! 3198: *----------------------------------------------------------------------------*/ ! 3199: ! 3200: int64 float64_to_int64( float64 a ) ! 3201: { ! 3202: flag aSign; ! 3203: int16 aExp, shiftCount; ! 3204: bits64 aSig, aSigExtra; ! 3205: ! 3206: aSig = extractFloat64Frac( a ); ! 3207: aExp = extractFloat64Exp( a ); ! 3208: aSign = extractFloat64Sign( a ); ! 3209: if ( aExp ) aSig |= LIT64( 0x0010000000000000 ); ! 3210: shiftCount = 0x433 - aExp; ! 3211: if ( shiftCount <= 0 ) { ! 3212: if ( 0x43E < aExp ) { ! 3213: float_raise( float_flag_invalid ); ! 3214: if ( ! aSign ! 3215: || ( ( aExp == 0x7FF ) ! 3216: && ( aSig != LIT64( 0x0010000000000000 ) ) ) ! 3217: ) { ! 3218: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 3219: } ! 3220: return (sbits64) LIT64( 0x8000000000000000 ); ! 3221: } ! 3222: aSigExtra = 0; ! 3223: aSig <<= - shiftCount; ! 3224: } ! 3225: else { ! 3226: shift64ExtraRightJamming( aSig, 0, shiftCount, &aSig, &aSigExtra ); ! 3227: } ! 3228: return roundAndPackInt64( aSign, aSig, aSigExtra ); ! 3229: ! 3230: } ! 3231: ! 3232: /*---------------------------------------------------------------------------- ! 3233: | Returns the result of converting the double-precision floating-point value ! 3234: | `a' to the 64-bit two's complement integer format. The conversion is ! 3235: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3236: | Arithmetic, except that the conversion is always rounded toward zero. ! 3237: | If `a' is a NaN, the largest positive integer is returned. Otherwise, if ! 3238: | the conversion overflows, the largest integer with the same sign as `a' is ! 3239: | returned. ! 3240: *----------------------------------------------------------------------------*/ ! 3241: ! 3242: int64 float64_to_int64_round_to_zero( float64 a ) ! 3243: { ! 3244: flag aSign; ! 3245: int16 aExp, shiftCount; ! 3246: bits64 aSig; ! 3247: int64 z; ! 3248: ! 3249: aSig = extractFloat64Frac( a ); ! 3250: aExp = extractFloat64Exp( a ); ! 3251: aSign = extractFloat64Sign( a ); ! 3252: if ( aExp ) aSig |= LIT64( 0x0010000000000000 ); ! 3253: shiftCount = aExp - 0x433; ! 3254: if ( 0 <= shiftCount ) { ! 3255: if ( 0x43E <= aExp ) { ! 3256: if ( a != LIT64( 0xC3E0000000000000 ) ) { ! 3257: float_raise( float_flag_invalid ); ! 3258: if ( ! aSign ! 3259: || ( ( aExp == 0x7FF ) ! 3260: && ( aSig != LIT64( 0x0010000000000000 ) ) ) ! 3261: ) { ! 3262: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 3263: } ! 3264: } ! 3265: return (sbits64) LIT64( 0x8000000000000000 ); ! 3266: } ! 3267: z = aSig<<shiftCount; ! 3268: } ! 3269: else { ! 3270: if ( aExp < 0x3FE ) { ! 3271: if ( aExp | aSig ) float_exception_flags |= float_flag_inexact; ! 3272: return 0; ! 3273: } ! 3274: z = aSig>>( - shiftCount ); ! 3275: if ( (bits64) ( aSig<<( shiftCount & 63 ) ) ) { ! 3276: float_exception_flags |= float_flag_inexact; ! 3277: } ! 3278: } ! 3279: if ( aSign ) z = - z; ! 3280: return z; ! 3281: ! 3282: } ! 3283: ! 3284: /*---------------------------------------------------------------------------- ! 3285: | Returns the result of converting the double-precision floating-point value ! 3286: | `a' to the single-precision floating-point format. The conversion is ! 3287: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3288: | Arithmetic. ! 3289: *----------------------------------------------------------------------------*/ ! 3290: ! 3291: float32 float64_to_float32( float64 a ) ! 3292: { ! 3293: flag aSign; ! 3294: int16 aExp; ! 3295: bits64 aSig; ! 3296: bits32 zSig; ! 3297: ! 3298: aSig = extractFloat64Frac( a ); ! 3299: aExp = extractFloat64Exp( a ); ! 3300: aSign = extractFloat64Sign( a ); ! 3301: if ( aExp == 0x7FF ) { ! 3302: if ( aSig ) return commonNaNToFloat32( float64ToCommonNaN( a ) ); ! 3303: return packFloat32( aSign, 0xFF, 0 ); ! 3304: } ! 3305: shift64RightJamming( aSig, 22, &aSig ); ! 3306: zSig = aSig; ! 3307: if ( aExp || zSig ) { ! 3308: zSig |= 0x40000000; ! 3309: aExp -= 0x381; ! 3310: } ! 3311: #ifdef SOFTFLOAT_I860 ! 3312: return roundAndPackFloat32_2( aSign, aExp, zSig ); ! 3313: #else ! 3314: return roundAndPackFloat32( aSign, aExp, zSig ); ! 3315: #endif ! 3316: ! 3317: } ! 3318: ! 3319: #ifdef FLOATX80 ! 3320: ! 3321: /*---------------------------------------------------------------------------- ! 3322: | Returns the result of converting the double-precision floating-point value ! 3323: | `a' to the extended double-precision floating-point format. The conversion ! 3324: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3325: | Arithmetic. ! 3326: *----------------------------------------------------------------------------*/ ! 3327: ! 3328: floatx80 float64_to_floatx80( float64 a ) ! 3329: { ! 3330: flag aSign; ! 3331: int16 aExp; ! 3332: bits64 aSig; ! 3333: ! 3334: aSig = extractFloat64Frac( a ); ! 3335: aExp = extractFloat64Exp( a ); ! 3336: aSign = extractFloat64Sign( a ); ! 3337: if ( aExp == 0x7FF ) { ! 3338: if ( aSig ) return commonNaNToFloatx80( float64ToCommonNaN( a ) ); ! 3339: return packFloatx80( aSign, 0x7FFF, floatx80_default_infinity_low ); ! 3340: } ! 3341: if ( aExp == 0 ) { ! 3342: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 3343: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3344: } ! 3345: return ! 3346: packFloatx80( ! 3347: aSign, aExp + 0x3C00, ( aSig | LIT64( 0x0010000000000000 ) )<<11 ); ! 3348: ! 3349: } ! 3350: ! 3351: #ifdef SOFTFLOAT_68K // 31-12-2016: Added for Previous ! 3352: floatx80 float64_to_floatx80_allowunnormal( float64 a ) ! 3353: { ! 3354: flag aSign; ! 3355: int16 aExp; ! 3356: bits64 aSig; ! 3357: ! 3358: aSig = extractFloat64Frac( a ); ! 3359: aExp = extractFloat64Exp( a ); ! 3360: aSign = extractFloat64Sign( a ); ! 3361: if ( aExp == 0x7FF ) { ! 3362: return packFloatx80( aSign, 0x7FFF, aSig<<11 ); ! 3363: } ! 3364: if ( aExp == 0 ) { ! 3365: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 3366: return packFloatx80( aSign, 0x3C01, aSig<<11 ); ! 3367: } ! 3368: return ! 3369: packFloatx80( ! 3370: aSign, aExp + 0x3C00, ( aSig | LIT64( 0x0010000000000000 ) )<<11 ); ! 3371: ! 3372: } ! 3373: #endif // end of addition for Previous ! 3374: ! 3375: #endif ! 3376: ! 3377: #ifdef FLOAT128 ! 3378: ! 3379: /*---------------------------------------------------------------------------- ! 3380: | Returns the result of converting the double-precision floating-point value ! 3381: | `a' to the quadruple-precision floating-point format. The conversion is ! 3382: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 3383: | Arithmetic. ! 3384: *----------------------------------------------------------------------------*/ ! 3385: ! 3386: float128 float64_to_float128( float64 a ) ! 3387: { ! 3388: flag aSign; ! 3389: int16 aExp; ! 3390: bits64 aSig, zSig0, zSig1; ! 3391: ! 3392: aSig = extractFloat64Frac( a ); ! 3393: aExp = extractFloat64Exp( a ); ! 3394: aSign = extractFloat64Sign( a ); ! 3395: if ( aExp == 0x7FF ) { ! 3396: if ( aSig ) return commonNaNToFloat128( float64ToCommonNaN( a ) ); ! 3397: return packFloat128( aSign, 0x7FFF, 0, 0 ); ! 3398: } ! 3399: if ( aExp == 0 ) { ! 3400: if ( aSig == 0 ) return packFloat128( aSign, 0, 0, 0 ); ! 3401: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3402: --aExp; ! 3403: } ! 3404: shift128Right( aSig, 0, 4, &zSig0, &zSig1 ); ! 3405: return packFloat128( aSign, aExp + 0x3C00, zSig0, zSig1 ); ! 3406: ! 3407: } ! 3408: ! 3409: #endif ! 3410: ! 3411: /*---------------------------------------------------------------------------- ! 3412: | Rounds the double-precision floating-point value `a' to an integer, and ! 3413: | returns the result as a double-precision floating-point value. The ! 3414: | operation is performed according to the IEC/IEEE Standard for Binary ! 3415: | Floating-Point Arithmetic. ! 3416: *----------------------------------------------------------------------------*/ ! 3417: ! 3418: float64 float64_round_to_int( float64 a ) ! 3419: { ! 3420: flag aSign; ! 3421: int16 aExp; ! 3422: bits64 lastBitMask, roundBitsMask; ! 3423: int8 roundingMode; ! 3424: float64 z; ! 3425: ! 3426: aExp = extractFloat64Exp( a ); ! 3427: if ( 0x433 <= aExp ) { ! 3428: if ( ( aExp == 0x7FF ) && extractFloat64Frac( a ) ) { ! 3429: return propagateFloat64NaN( a, a ); ! 3430: } ! 3431: return a; ! 3432: } ! 3433: if ( aExp < 0x3FF ) { ! 3434: if ( (bits64) ( a<<1 ) == 0 ) return a; ! 3435: float_exception_flags |= float_flag_inexact; ! 3436: aSign = extractFloat64Sign( a ); ! 3437: switch ( float_rounding_mode ) { ! 3438: case float_round_nearest_even: ! 3439: if ( ( aExp == 0x3FE ) && extractFloat64Frac( a ) ) { ! 3440: return packFloat64( aSign, 0x3FF, 0 ); ! 3441: } ! 3442: break; ! 3443: case float_round_down: ! 3444: return aSign ? LIT64( 0xBFF0000000000000 ) : 0; ! 3445: case float_round_up: ! 3446: return ! 3447: aSign ? LIT64( 0x8000000000000000 ) : LIT64( 0x3FF0000000000000 ); ! 3448: } ! 3449: return packFloat64( aSign, 0, 0 ); ! 3450: } ! 3451: lastBitMask = 1; ! 3452: lastBitMask <<= 0x433 - aExp; ! 3453: roundBitsMask = lastBitMask - 1; ! 3454: z = a; ! 3455: roundingMode = float_rounding_mode; ! 3456: if ( roundingMode == float_round_nearest_even ) { ! 3457: z += lastBitMask>>1; ! 3458: if ( ( z & roundBitsMask ) == 0 ) z &= ~ lastBitMask; ! 3459: } ! 3460: else if ( roundingMode != float_round_to_zero ) { ! 3461: if ( extractFloat64Sign( z ) ^ ( roundingMode == float_round_up ) ) { ! 3462: z += roundBitsMask; ! 3463: } ! 3464: } ! 3465: z &= ~ roundBitsMask; ! 3466: if ( z != a ) float_exception_flags |= float_flag_inexact; ! 3467: return z; ! 3468: ! 3469: } ! 3470: ! 3471: /*---------------------------------------------------------------------------- ! 3472: | Returns the result of adding the absolute values of the double-precision ! 3473: | floating-point values `a' and `b'. If `zSign' is 1, the sum is negated ! 3474: | before being returned. `zSign' is ignored if the result is a NaN. ! 3475: | The addition is performed according to the IEC/IEEE Standard for Binary ! 3476: | Floating-Point Arithmetic. ! 3477: *----------------------------------------------------------------------------*/ ! 3478: ! 3479: static float64 addFloat64Sigs( float64 a, float64 b, flag zSign ) ! 3480: { ! 3481: int16 aExp, bExp, zExp; ! 3482: bits64 aSig, bSig, zSig; ! 3483: int16 expDiff; ! 3484: ! 3485: aSig = extractFloat64Frac( a ); ! 3486: aExp = extractFloat64Exp( a ); ! 3487: bSig = extractFloat64Frac( b ); ! 3488: bExp = extractFloat64Exp( b ); ! 3489: expDiff = aExp - bExp; ! 3490: aSig <<= 9; ! 3491: bSig <<= 9; ! 3492: if ( 0 < expDiff ) { ! 3493: if ( aExp == 0x7FF ) { ! 3494: if ( aSig ) return propagateFloat64NaN( a, b ); ! 3495: return a; ! 3496: } ! 3497: if ( bExp == 0 ) { ! 3498: --expDiff; ! 3499: } ! 3500: else { ! 3501: bSig |= LIT64( 0x2000000000000000 ); ! 3502: } ! 3503: shift64RightJamming( bSig, expDiff, &bSig ); ! 3504: zExp = aExp; ! 3505: } ! 3506: else if ( expDiff < 0 ) { ! 3507: if ( bExp == 0x7FF ) { ! 3508: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3509: return packFloat64( zSign, 0x7FF, 0 ); ! 3510: } ! 3511: if ( aExp == 0 ) { ! 3512: ++expDiff; ! 3513: } ! 3514: else { ! 3515: aSig |= LIT64( 0x2000000000000000 ); ! 3516: } ! 3517: shift64RightJamming( aSig, - expDiff, &aSig ); ! 3518: zExp = bExp; ! 3519: } ! 3520: else { ! 3521: if ( aExp == 0x7FF ) { ! 3522: if ( aSig | bSig ) return propagateFloat64NaN( a, b ); ! 3523: return a; ! 3524: } ! 3525: if ( aExp == 0 ) return packFloat64( zSign, 0, ( aSig + bSig )>>9 ); ! 3526: zSig = LIT64( 0x4000000000000000 ) + aSig + bSig; ! 3527: zExp = aExp; ! 3528: goto roundAndPack; ! 3529: } ! 3530: aSig |= LIT64( 0x2000000000000000 ); ! 3531: zSig = ( aSig + bSig )<<1; ! 3532: --zExp; ! 3533: if ( (sbits64) zSig < 0 ) { ! 3534: zSig = aSig + bSig; ! 3535: ++zExp; ! 3536: } ! 3537: roundAndPack: ! 3538: #ifdef SOFTFLOAT_I860 ! 3539: return roundAndPackFloat64_2( zSign, zExp, zSig ); ! 3540: #else ! 3541: return roundAndPackFloat64( zSign, zExp, zSig ); ! 3542: #endif ! 3543: ! 3544: } ! 3545: ! 3546: /*---------------------------------------------------------------------------- ! 3547: | Returns the result of subtracting the absolute values of the double- ! 3548: | precision floating-point values `a' and `b'. If `zSign' is 1, the ! 3549: | difference is negated before being returned. `zSign' is ignored if the ! 3550: | result is a NaN. The subtraction is performed according to the IEC/IEEE ! 3551: | Standard for Binary Floating-Point Arithmetic. ! 3552: *----------------------------------------------------------------------------*/ ! 3553: ! 3554: static float64 subFloat64Sigs( float64 a, float64 b, flag zSign ) ! 3555: { ! 3556: int16 aExp, bExp, zExp; ! 3557: bits64 aSig, bSig, zSig; ! 3558: int16 expDiff; ! 3559: #ifdef SOFTFLOAT_I860 ! 3560: int8 shiftCount; ! 3561: #endif ! 3562: ! 3563: aSig = extractFloat64Frac( a ); ! 3564: aExp = extractFloat64Exp( a ); ! 3565: bSig = extractFloat64Frac( b ); ! 3566: bExp = extractFloat64Exp( b ); ! 3567: expDiff = aExp - bExp; ! 3568: aSig <<= 10; ! 3569: bSig <<= 10; ! 3570: if ( 0 < expDiff ) goto aExpBigger; ! 3571: if ( expDiff < 0 ) goto bExpBigger; ! 3572: if ( aExp == 0x7FF ) { ! 3573: if ( aSig | bSig ) return propagateFloat64NaN( a, b ); ! 3574: #ifdef SOFTFLOAT_I860 ! 3575: float_raise2( float_flag_invalid ); ! 3576: #else ! 3577: float_raise( float_flag_invalid ); ! 3578: #endif ! 3579: return float64_default_nan; ! 3580: } ! 3581: if ( aExp == 0 ) { ! 3582: aExp = 1; ! 3583: bExp = 1; ! 3584: } ! 3585: if ( bSig < aSig ) goto aBigger; ! 3586: if ( aSig < bSig ) goto bBigger; ! 3587: #ifdef SOFTFLOAT_I860 ! 3588: return packFloat64( float_rounding_mode2 == float_round_down, 0, 0 ); ! 3589: #else ! 3590: return packFloat64( float_rounding_mode == float_round_down, 0, 0 ); ! 3591: #endif ! 3592: bExpBigger: ! 3593: if ( bExp == 0x7FF ) { ! 3594: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3595: return packFloat64( zSign ^ 1, 0x7FF, 0 ); ! 3596: } ! 3597: if ( aExp == 0 ) { ! 3598: ++expDiff; ! 3599: } ! 3600: else { ! 3601: aSig |= LIT64( 0x4000000000000000 ); ! 3602: } ! 3603: shift64RightJamming( aSig, - expDiff, &aSig ); ! 3604: bSig |= LIT64( 0x4000000000000000 ); ! 3605: bBigger: ! 3606: zSig = bSig - aSig; ! 3607: zExp = bExp; ! 3608: zSign ^= 1; ! 3609: goto normalizeRoundAndPack; ! 3610: aExpBigger: ! 3611: if ( aExp == 0x7FF ) { ! 3612: if ( aSig ) return propagateFloat64NaN( a, b ); ! 3613: return a; ! 3614: } ! 3615: if ( bExp == 0 ) { ! 3616: --expDiff; ! 3617: } ! 3618: else { ! 3619: bSig |= LIT64( 0x4000000000000000 ); ! 3620: } ! 3621: shift64RightJamming( bSig, expDiff, &bSig ); ! 3622: aSig |= LIT64( 0x4000000000000000 ); ! 3623: aBigger: ! 3624: zSig = aSig - bSig; ! 3625: zExp = aExp; ! 3626: normalizeRoundAndPack: ! 3627: --zExp; ! 3628: #ifdef SOFTFLOAT_I860 ! 3629: shiftCount = countLeadingZeros64( zSig ) - 1; ! 3630: return roundAndPackFloat64_2( zSign, zExp - shiftCount, zSig<<shiftCount ); ! 3631: #else ! 3632: return normalizeRoundAndPackFloat64( zSign, zExp, zSig ); ! 3633: #endif ! 3634: ! 3635: } ! 3636: ! 3637: /*---------------------------------------------------------------------------- ! 3638: | Returns the result of adding the double-precision floating-point values `a' ! 3639: | and `b'. The operation is performed according to the IEC/IEEE Standard for ! 3640: | Binary Floating-Point Arithmetic. ! 3641: *----------------------------------------------------------------------------*/ ! 3642: ! 3643: float64 float64_add( float64 a, float64 b ) ! 3644: { ! 3645: flag aSign, bSign; ! 3646: ! 3647: aSign = extractFloat64Sign( a ); ! 3648: bSign = extractFloat64Sign( b ); ! 3649: if ( aSign == bSign ) { ! 3650: return addFloat64Sigs( a, b, aSign ); ! 3651: } ! 3652: else { ! 3653: return subFloat64Sigs( a, b, aSign ); ! 3654: } ! 3655: ! 3656: } ! 3657: ! 3658: /*---------------------------------------------------------------------------- ! 3659: | Returns the result of subtracting the double-precision floating-point values ! 3660: | `a' and `b'. The operation is performed according to the IEC/IEEE Standard ! 3661: | for Binary Floating-Point Arithmetic. ! 3662: *----------------------------------------------------------------------------*/ ! 3663: ! 3664: float64 float64_sub( float64 a, float64 b ) ! 3665: { ! 3666: flag aSign, bSign; ! 3667: ! 3668: aSign = extractFloat64Sign( a ); ! 3669: bSign = extractFloat64Sign( b ); ! 3670: if ( aSign == bSign ) { ! 3671: return subFloat64Sigs( a, b, aSign ); ! 3672: } ! 3673: else { ! 3674: return addFloat64Sigs( a, b, aSign ); ! 3675: } ! 3676: ! 3677: } ! 3678: ! 3679: /*---------------------------------------------------------------------------- ! 3680: | Returns the result of multiplying the double-precision floating-point values ! 3681: | `a' and `b'. The operation is performed according to the IEC/IEEE Standard ! 3682: | for Binary Floating-Point Arithmetic. ! 3683: *----------------------------------------------------------------------------*/ ! 3684: ! 3685: float64 float64_mul( float64 a, float64 b ) ! 3686: { ! 3687: flag aSign, bSign, zSign; ! 3688: int16 aExp, bExp, zExp; ! 3689: bits64 aSig, bSig, zSig0, zSig1; ! 3690: ! 3691: aSig = extractFloat64Frac( a ); ! 3692: aExp = extractFloat64Exp( a ); ! 3693: aSign = extractFloat64Sign( a ); ! 3694: bSig = extractFloat64Frac( b ); ! 3695: bExp = extractFloat64Exp( b ); ! 3696: bSign = extractFloat64Sign( b ); ! 3697: zSign = aSign ^ bSign; ! 3698: if ( aExp == 0x7FF ) { ! 3699: if ( aSig || ( ( bExp == 0x7FF ) && bSig ) ) { ! 3700: return propagateFloat64NaN( a, b ); ! 3701: } ! 3702: if ( ( bExp | bSig ) == 0 ) { ! 3703: #ifdef SOFTFLOAT_I860 ! 3704: float_raise2( float_flag_invalid ); ! 3705: #else ! 3706: float_raise( float_flag_invalid ); ! 3707: #endif ! 3708: return float64_default_nan; ! 3709: } ! 3710: return packFloat64( zSign, 0x7FF, 0 ); ! 3711: } ! 3712: if ( bExp == 0x7FF ) { ! 3713: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3714: if ( ( aExp | aSig ) == 0 ) { ! 3715: #ifdef SOFTFLOAT_I860 ! 3716: float_raise2( float_flag_invalid ); ! 3717: #else ! 3718: float_raise( float_flag_invalid ); ! 3719: #endif ! 3720: return float64_default_nan; ! 3721: } ! 3722: return packFloat64( zSign, 0x7FF, 0 ); ! 3723: } ! 3724: if ( aExp == 0 ) { ! 3725: if ( aSig == 0 ) return packFloat64( zSign, 0, 0 ); ! 3726: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3727: } ! 3728: if ( bExp == 0 ) { ! 3729: if ( bSig == 0 ) return packFloat64( zSign, 0, 0 ); ! 3730: normalizeFloat64Subnormal( bSig, &bExp, &bSig ); ! 3731: } ! 3732: zExp = aExp + bExp - 0x3FF; ! 3733: aSig = ( aSig | LIT64( 0x0010000000000000 ) )<<10; ! 3734: bSig = ( bSig | LIT64( 0x0010000000000000 ) )<<11; ! 3735: mul64To128( aSig, bSig, &zSig0, &zSig1 ); ! 3736: zSig0 |= ( zSig1 != 0 ); ! 3737: if ( 0 <= (sbits64) ( zSig0<<1 ) ) { ! 3738: zSig0 <<= 1; ! 3739: --zExp; ! 3740: } ! 3741: #ifdef SOFTFLOAT_I860 ! 3742: return roundAndPackFloat64_2( zSign, zExp, zSig0 ); ! 3743: #else ! 3744: return roundAndPackFloat64( zSign, zExp, zSig0 ); ! 3745: #endif ! 3746: ! 3747: } ! 3748: ! 3749: /*---------------------------------------------------------------------------- ! 3750: | Returns the result of dividing the double-precision floating-point value `a' ! 3751: | by the corresponding value `b'. The operation is performed according to ! 3752: | the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3753: *----------------------------------------------------------------------------*/ ! 3754: ! 3755: float64 float64_div( float64 a, float64 b ) ! 3756: { ! 3757: flag aSign, bSign, zSign; ! 3758: int16 aExp, bExp, zExp; ! 3759: bits64 aSig, bSig, zSig; ! 3760: bits64 rem0, rem1; ! 3761: bits64 term0, term1; ! 3762: ! 3763: aSig = extractFloat64Frac( a ); ! 3764: aExp = extractFloat64Exp( a ); ! 3765: aSign = extractFloat64Sign( a ); ! 3766: bSig = extractFloat64Frac( b ); ! 3767: bExp = extractFloat64Exp( b ); ! 3768: bSign = extractFloat64Sign( b ); ! 3769: zSign = aSign ^ bSign; ! 3770: if ( aExp == 0x7FF ) { ! 3771: if ( aSig ) return propagateFloat64NaN( a, b ); ! 3772: if ( bExp == 0x7FF ) { ! 3773: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3774: #ifdef SOFTFLOAT_I860 ! 3775: float_raise2( float_flag_invalid ); ! 3776: #else ! 3777: float_raise( float_flag_invalid ); ! 3778: #endif ! 3779: return float64_default_nan; ! 3780: } ! 3781: return packFloat64( zSign, 0x7FF, 0 ); ! 3782: } ! 3783: if ( bExp == 0x7FF ) { ! 3784: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3785: return packFloat64( zSign, 0, 0 ); ! 3786: } ! 3787: if ( bExp == 0 ) { ! 3788: if ( bSig == 0 ) { ! 3789: if ( ( aExp | aSig ) == 0 ) { ! 3790: #ifdef SOFTFLOAT_I860 ! 3791: float_raise2( float_flag_invalid ); ! 3792: #else ! 3793: float_raise( float_flag_invalid ); ! 3794: #endif ! 3795: return float64_default_nan; ! 3796: } ! 3797: #ifdef SOFTFLOAT_I860 ! 3798: float_raise2( float_flag_divbyzero ); ! 3799: #else ! 3800: float_raise( float_flag_divbyzero ); ! 3801: #endif ! 3802: return packFloat64( zSign, 0x7FF, 0 ); ! 3803: } ! 3804: normalizeFloat64Subnormal( bSig, &bExp, &bSig ); ! 3805: } ! 3806: if ( aExp == 0 ) { ! 3807: if ( aSig == 0 ) return packFloat64( zSign, 0, 0 ); ! 3808: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3809: } ! 3810: zExp = aExp - bExp + 0x3FD; ! 3811: aSig = ( aSig | LIT64( 0x0010000000000000 ) )<<10; ! 3812: bSig = ( bSig | LIT64( 0x0010000000000000 ) )<<11; ! 3813: if ( bSig <= ( aSig + aSig ) ) { ! 3814: aSig >>= 1; ! 3815: ++zExp; ! 3816: } ! 3817: zSig = estimateDiv128To64( aSig, 0, bSig ); ! 3818: if ( ( zSig & 0x1FF ) <= 2 ) { ! 3819: mul64To128( bSig, zSig, &term0, &term1 ); ! 3820: sub128( aSig, 0, term0, term1, &rem0, &rem1 ); ! 3821: while ( (sbits64) rem0 < 0 ) { ! 3822: --zSig; ! 3823: add128( rem0, rem1, 0, bSig, &rem0, &rem1 ); ! 3824: } ! 3825: zSig |= ( rem1 != 0 ); ! 3826: } ! 3827: #ifdef SOFTFLOAT_I860 ! 3828: return roundAndPackFloat64_2( zSign, zExp, zSig ); ! 3829: #else ! 3830: return roundAndPackFloat64( zSign, zExp, zSig ); ! 3831: #endif ! 3832: ! 3833: } ! 3834: ! 3835: /*---------------------------------------------------------------------------- ! 3836: | Returns the remainder of the double-precision floating-point value `a' ! 3837: | with respect to the corresponding value `b'. The operation is performed ! 3838: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3839: *----------------------------------------------------------------------------*/ ! 3840: ! 3841: float64 float64_rem( float64 a, float64 b ) ! 3842: { ! 3843: flag aSign, zSign; ! 3844: int16 aExp, bExp, expDiff; ! 3845: bits64 aSig, bSig; ! 3846: bits64 q, alternateASig; ! 3847: sbits64 sigMean; ! 3848: ! 3849: aSig = extractFloat64Frac( a ); ! 3850: aExp = extractFloat64Exp( a ); ! 3851: aSign = extractFloat64Sign( a ); ! 3852: bSig = extractFloat64Frac( b ); ! 3853: bExp = extractFloat64Exp( b ); ! 3854: // bSign = extractFloat64Sign( b ); ! 3855: if ( aExp == 0x7FF ) { ! 3856: if ( aSig || ( ( bExp == 0x7FF ) && bSig ) ) { ! 3857: return propagateFloat64NaN( a, b ); ! 3858: } ! 3859: float_raise( float_flag_invalid ); ! 3860: return float64_default_nan; ! 3861: } ! 3862: if ( bExp == 0x7FF ) { ! 3863: if ( bSig ) return propagateFloat64NaN( a, b ); ! 3864: return a; ! 3865: } ! 3866: if ( bExp == 0 ) { ! 3867: if ( bSig == 0 ) { ! 3868: float_raise( float_flag_invalid ); ! 3869: return float64_default_nan; ! 3870: } ! 3871: normalizeFloat64Subnormal( bSig, &bExp, &bSig ); ! 3872: } ! 3873: if ( aExp == 0 ) { ! 3874: if ( aSig == 0 ) return a; ! 3875: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3876: } ! 3877: expDiff = aExp - bExp; ! 3878: aSig = ( aSig | LIT64( 0x0010000000000000 ) )<<11; ! 3879: bSig = ( bSig | LIT64( 0x0010000000000000 ) )<<11; ! 3880: if ( expDiff < 0 ) { ! 3881: if ( expDiff < -1 ) return a; ! 3882: aSig >>= 1; ! 3883: } ! 3884: q = ( bSig <= aSig ); ! 3885: if ( q ) aSig -= bSig; ! 3886: expDiff -= 64; ! 3887: while ( 0 < expDiff ) { ! 3888: q = estimateDiv128To64( aSig, 0, bSig ); ! 3889: q = ( 2 < q ) ? q - 2 : 0; ! 3890: aSig = - ( ( bSig>>2 ) * q ); ! 3891: expDiff -= 62; ! 3892: } ! 3893: expDiff += 64; ! 3894: if ( 0 < expDiff ) { ! 3895: q = estimateDiv128To64( aSig, 0, bSig ); ! 3896: q = ( 2 < q ) ? q - 2 : 0; ! 3897: q >>= 64 - expDiff; ! 3898: bSig >>= 2; ! 3899: aSig = ( ( aSig>>1 )<<( expDiff - 1 ) ) - bSig * q; ! 3900: } ! 3901: else { ! 3902: aSig >>= 2; ! 3903: bSig >>= 2; ! 3904: } ! 3905: do { ! 3906: alternateASig = aSig; ! 3907: ++q; ! 3908: aSig -= bSig; ! 3909: } while ( 0 <= (sbits64) aSig ); ! 3910: sigMean = aSig + alternateASig; ! 3911: if ( ( sigMean < 0 ) || ( ( sigMean == 0 ) && ( q & 1 ) ) ) { ! 3912: aSig = alternateASig; ! 3913: } ! 3914: zSign = ( (sbits64) aSig < 0 ); ! 3915: if ( zSign ) aSig = - aSig; ! 3916: return normalizeRoundAndPackFloat64( aSign ^ zSign, bExp, aSig ); ! 3917: ! 3918: } ! 3919: ! 3920: /*---------------------------------------------------------------------------- ! 3921: | Returns the square root of the double-precision floating-point value `a'. ! 3922: | The operation is performed according to the IEC/IEEE Standard for Binary ! 3923: | Floating-Point Arithmetic. ! 3924: *----------------------------------------------------------------------------*/ ! 3925: ! 3926: float64 float64_sqrt( float64 a ) ! 3927: { ! 3928: flag aSign; ! 3929: int16 aExp, zExp; ! 3930: bits64 aSig, zSig, doubleZSig; ! 3931: bits64 rem0, rem1, term0, term1; ! 3932: ! 3933: aSig = extractFloat64Frac( a ); ! 3934: aExp = extractFloat64Exp( a ); ! 3935: aSign = extractFloat64Sign( a ); ! 3936: if ( aExp == 0x7FF ) { ! 3937: if ( aSig ) return propagateFloat64NaN( a, a ); ! 3938: if ( ! aSign ) return a; ! 3939: #ifdef SOFTFLOAT_I860 ! 3940: float_raise2( float_flag_invalid ); ! 3941: #else ! 3942: float_raise( float_flag_invalid ); ! 3943: #endif ! 3944: return float64_default_nan; ! 3945: } ! 3946: if ( aSign ) { ! 3947: if ( ( aExp | aSig ) == 0 ) return a; ! 3948: #ifdef SOFTFLOAT_I860 ! 3949: float_raise2( float_flag_invalid ); ! 3950: #else ! 3951: float_raise( float_flag_invalid ); ! 3952: #endif ! 3953: return float64_default_nan; ! 3954: } ! 3955: if ( aExp == 0 ) { ! 3956: if ( aSig == 0 ) return 0; ! 3957: normalizeFloat64Subnormal( aSig, &aExp, &aSig ); ! 3958: } ! 3959: zExp = ( ( aExp - 0x3FF )>>1 ) + 0x3FE; ! 3960: aSig |= LIT64( 0x0010000000000000 ); ! 3961: zSig = estimateSqrt32( aExp, aSig>>21 ); ! 3962: aSig <<= 9 - ( aExp & 1 ); ! 3963: zSig = estimateDiv128To64( aSig, 0, zSig<<32 ) + ( zSig<<30 ); ! 3964: if ( ( zSig & 0x1FF ) <= 5 ) { ! 3965: doubleZSig = zSig<<1; ! 3966: mul64To128( zSig, zSig, &term0, &term1 ); ! 3967: sub128( aSig, 0, term0, term1, &rem0, &rem1 ); ! 3968: while ( (sbits64) rem0 < 0 ) { ! 3969: --zSig; ! 3970: doubleZSig -= 2; ! 3971: add128( rem0, rem1, zSig>>63, doubleZSig | 1, &rem0, &rem1 ); ! 3972: } ! 3973: zSig |= ( ( rem0 | rem1 ) != 0 ); ! 3974: } ! 3975: #ifdef SOFTFLOAT_I860 ! 3976: return roundAndPackFloat64_2( 0, zExp, zSig ); ! 3977: #else ! 3978: return roundAndPackFloat64( 0, zExp, zSig ); ! 3979: #endif ! 3980: ! 3981: } ! 3982: ! 3983: /*---------------------------------------------------------------------------- ! 3984: | Returns 1 if the double-precision floating-point value `a' is equal to the ! 3985: | corresponding value `b', and 0 otherwise. The comparison is performed ! 3986: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 3987: *----------------------------------------------------------------------------*/ ! 3988: ! 3989: flag float64_eq( float64 a, float64 b ) ! 3990: { ! 3991: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 3992: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 3993: ) { ! 3994: if ( float64_is_signaling_nan( a ) || float64_is_signaling_nan( b ) ) { ! 3995: #ifdef SOFTFLOAT_I860 ! 3996: float_raise2( float_flag_invalid ); ! 3997: #else ! 3998: float_raise( float_flag_invalid ); ! 3999: #endif ! 4000: } ! 4001: return 0; ! 4002: } ! 4003: return ( a == b ) || ( (bits64) ( ( a | b )<<1 ) == 0 ); ! 4004: ! 4005: } ! 4006: ! 4007: /*---------------------------------------------------------------------------- ! 4008: | Returns 1 if the double-precision floating-point value `a' is less than or ! 4009: | equal to the corresponding value `b', and 0 otherwise. The comparison is ! 4010: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 4011: | Arithmetic. ! 4012: *----------------------------------------------------------------------------*/ ! 4013: ! 4014: flag float64_le( float64 a, float64 b ) ! 4015: { ! 4016: flag aSign, bSign; ! 4017: ! 4018: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4019: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4020: ) { ! 4021: #ifdef SOFTFLOAT_I860 ! 4022: float_raise2( float_flag_invalid ); ! 4023: #else ! 4024: float_raise( float_flag_invalid ); ! 4025: #endif ! 4026: return 0; ! 4027: } ! 4028: aSign = extractFloat64Sign( a ); ! 4029: bSign = extractFloat64Sign( b ); ! 4030: if ( aSign != bSign ) return aSign || ( (bits64) ( ( a | b )<<1 ) == 0 ); ! 4031: return ( a == b ) || ( aSign ^ ( a < b ) ); ! 4032: ! 4033: } ! 4034: ! 4035: /*---------------------------------------------------------------------------- ! 4036: | Returns 1 if the double-precision floating-point value `a' is less than ! 4037: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 4038: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4039: *----------------------------------------------------------------------------*/ ! 4040: ! 4041: flag float64_lt( float64 a, float64 b ) ! 4042: { ! 4043: flag aSign, bSign; ! 4044: ! 4045: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4046: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4047: ) { ! 4048: #ifdef SOFTFLOAT_I860 ! 4049: float_raise2( float_flag_invalid ); ! 4050: #else ! 4051: float_raise( float_flag_invalid ); ! 4052: #endif ! 4053: return 0; ! 4054: } ! 4055: aSign = extractFloat64Sign( a ); ! 4056: bSign = extractFloat64Sign( b ); ! 4057: if ( aSign != bSign ) return aSign && ( (bits64) ( ( a | b )<<1 ) != 0 ); ! 4058: return ( a != b ) && ( aSign ^ ( a < b ) ); ! 4059: ! 4060: } ! 4061: ! 4062: #ifdef SOFTFLOAT_I860 // 29-04-2017: Added for Previous ! 4063: /*---------------------------------------------------------------------------- ! 4064: | Returns 1 if the double-precision floating-point value `a' is greater than ! 4065: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 4066: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4067: *----------------------------------------------------------------------------*/ ! 4068: ! 4069: flag float64_gt( float64 a, float64 b ) ! 4070: { ! 4071: flag aSign, bSign; ! 4072: ! 4073: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4074: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4075: ) { ! 4076: #ifdef SOFTFLOAT_I860 ! 4077: float_raise2( float_flag_invalid ); ! 4078: #else ! 4079: float_raise( float_flag_invalid ); ! 4080: #endif ! 4081: return 0; ! 4082: } ! 4083: aSign = extractFloat64Sign( a ); ! 4084: bSign = extractFloat64Sign( b ); ! 4085: if ( aSign != bSign ) return bSign && ( (bits64) ( ( a | b )<<1 ) != 0 ); ! 4086: return ( a != b ) && ( bSign ^ ( a > b ) ); ! 4087: ! 4088: } ! 4089: #endif // End of addition for Previous ! 4090: ! 4091: /*---------------------------------------------------------------------------- ! 4092: | Returns 1 if the double-precision floating-point value `a' is equal to the ! 4093: | corresponding value `b', and 0 otherwise. The invalid exception is raised ! 4094: | if either operand is a NaN. Otherwise, the comparison is performed ! 4095: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4096: *----------------------------------------------------------------------------*/ ! 4097: ! 4098: flag float64_eq_signaling( float64 a, float64 b ) ! 4099: { ! 4100: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4101: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4102: ) { ! 4103: float_raise( float_flag_invalid ); ! 4104: return 0; ! 4105: } ! 4106: return ( a == b ) || ( (bits64) ( ( a | b )<<1 ) == 0 ); ! 4107: ! 4108: } ! 4109: ! 4110: /*---------------------------------------------------------------------------- ! 4111: | Returns 1 if the double-precision floating-point value `a' is less than or ! 4112: | equal to the corresponding value `b', and 0 otherwise. Quiet NaNs do not ! 4113: | cause an exception. Otherwise, the comparison is performed according to the ! 4114: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4115: *----------------------------------------------------------------------------*/ ! 4116: ! 4117: flag float64_le_quiet( float64 a, float64 b ) ! 4118: { ! 4119: flag aSign, bSign; ! 4120: // int16 aExp, bExp; ! 4121: ! 4122: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4123: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4124: ) { ! 4125: if ( float64_is_signaling_nan( a ) || float64_is_signaling_nan( b ) ) { ! 4126: float_raise( float_flag_invalid ); ! 4127: } ! 4128: return 0; ! 4129: } ! 4130: aSign = extractFloat64Sign( a ); ! 4131: bSign = extractFloat64Sign( b ); ! 4132: if ( aSign != bSign ) return aSign || ( (bits64) ( ( a | b )<<1 ) == 0 ); ! 4133: return ( a == b ) || ( aSign ^ ( a < b ) ); ! 4134: ! 4135: } ! 4136: ! 4137: /*---------------------------------------------------------------------------- ! 4138: | Returns 1 if the double-precision floating-point value `a' is less than ! 4139: | the corresponding value `b', and 0 otherwise. Quiet NaNs do not cause an ! 4140: | exception. Otherwise, the comparison is performed according to the IEC/IEEE ! 4141: | Standard for Binary Floating-Point Arithmetic. ! 4142: *----------------------------------------------------------------------------*/ ! 4143: ! 4144: flag float64_lt_quiet( float64 a, float64 b ) ! 4145: { ! 4146: flag aSign, bSign; ! 4147: ! 4148: if ( ( ( extractFloat64Exp( a ) == 0x7FF ) && extractFloat64Frac( a ) ) ! 4149: || ( ( extractFloat64Exp( b ) == 0x7FF ) && extractFloat64Frac( b ) ) ! 4150: ) { ! 4151: if ( float64_is_signaling_nan( a ) || float64_is_signaling_nan( b ) ) { ! 4152: float_raise( float_flag_invalid ); ! 4153: } ! 4154: return 0; ! 4155: } ! 4156: aSign = extractFloat64Sign( a ); ! 4157: bSign = extractFloat64Sign( b ); ! 4158: if ( aSign != bSign ) return aSign && ( (bits64) ( ( a | b )<<1 ) != 0 ); ! 4159: return ( a != b ) && ( aSign ^ ( a < b ) ); ! 4160: ! 4161: } ! 4162: ! 4163: #ifdef FLOATX80 ! 4164: ! 4165: /*---------------------------------------------------------------------------- ! 4166: | Returns the result of converting the extended double-precision floating- ! 4167: | point value `a' to the 32-bit two's complement integer format. The ! 4168: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4169: | Floating-Point Arithmetic---which means in particular that the conversion ! 4170: | is rounded according to the current rounding mode. If `a' is a NaN, the ! 4171: | largest positive integer is returned. Otherwise, if the conversion ! 4172: | overflows, the largest integer with the same sign as `a' is returned. ! 4173: *----------------------------------------------------------------------------*/ ! 4174: ! 4175: int32 floatx80_to_int32( floatx80 a ) ! 4176: { ! 4177: flag aSign; ! 4178: int32 aExp, shiftCount; ! 4179: bits64 aSig; ! 4180: ! 4181: aSig = extractFloatx80Frac( a ); ! 4182: aExp = extractFloatx80Exp( a ); ! 4183: aSign = extractFloatx80Sign( a ); ! 4184: #ifdef SOFTFLOAT_68K ! 4185: if ( aExp == 0x7FFF ) { ! 4186: if ( (bits64) ( aSig<<1 ) ) { ! 4187: a = propagateFloatx80NaNOneArg( a ); ! 4188: if ( a.low == aSig ) float_raise( float_flag_invalid ); ! 4189: return (sbits32)(a.low>>32); ! 4190: } ! 4191: float_raise( float_flag_invalid ); ! 4192: return aSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 4193: } ! 4194: #else ! 4195: if ( ( aExp == 0x7FFF ) && (bits64) ( aSig<<1 ) ) aSign = 0; ! 4196: #endif ! 4197: shiftCount = 0x4037 - aExp; ! 4198: if ( shiftCount <= 0 ) shiftCount = 1; ! 4199: shift64RightJamming( aSig, shiftCount, &aSig ); ! 4200: return roundAndPackInt32( aSign, aSig ); ! 4201: ! 4202: } ! 4203: #ifdef SOFTFLOAT_68K // 30-01-2017: Addition for Previous ! 4204: int16 floatx80_to_int16( floatx80 a ) ! 4205: { ! 4206: flag aSign; ! 4207: int32 aExp, shiftCount; ! 4208: bits64 aSig; ! 4209: ! 4210: aSig = extractFloatx80Frac( a ); ! 4211: aExp = extractFloatx80Exp( a ); ! 4212: aSign = extractFloatx80Sign( a ); ! 4213: if ( aExp == 0x7FFF ) { ! 4214: if ( (bits64) ( aSig<<1 ) ) { ! 4215: a = propagateFloatx80NaNOneArg( a ); ! 4216: if ( a.low == aSig ) float_raise( float_flag_invalid ); ! 4217: return (sbits16)(a.low>>48); ! 4218: } ! 4219: float_raise( float_flag_invalid ); ! 4220: return aSign ? (sbits16) 0x8000 : 0x7FFF; ! 4221: } ! 4222: shiftCount = 0x4037 - aExp; ! 4223: if ( shiftCount <= 0 ) shiftCount = 1; ! 4224: shift64RightJamming( aSig, shiftCount, &aSig ); ! 4225: return roundAndPackInt16( aSign, aSig ); ! 4226: ! 4227: } ! 4228: int8 floatx80_to_int8( floatx80 a ) ! 4229: { ! 4230: flag aSign; ! 4231: int32 aExp, shiftCount; ! 4232: bits64 aSig; ! 4233: ! 4234: aSig = extractFloatx80Frac( a ); ! 4235: aExp = extractFloatx80Exp( a ); ! 4236: aSign = extractFloatx80Sign( a ); ! 4237: if ( aExp == 0x7FFF ) { ! 4238: if ( (bits64) ( aSig<<1 ) ) { ! 4239: a = propagateFloatx80NaNOneArg( a ); ! 4240: if ( a.low == aSig ) float_raise( float_flag_invalid ); ! 4241: return (sbits8)(a.low>>56); ! 4242: } ! 4243: float_raise( float_flag_invalid ); ! 4244: return aSign ? (sbits8) 0x80 : 0x7F; ! 4245: } ! 4246: shiftCount = 0x4037 - aExp; ! 4247: if ( shiftCount <= 0 ) shiftCount = 1; ! 4248: shift64RightJamming( aSig, shiftCount, &aSig ); ! 4249: return roundAndPackInt8( aSign, aSig ); ! 4250: ! 4251: } ! 4252: #endif // End of addition for Previous ! 4253: ! 4254: /*---------------------------------------------------------------------------- ! 4255: | Returns the result of converting the extended double-precision floating- ! 4256: | point value `a' to the 32-bit two's complement integer format. The ! 4257: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4258: | Floating-Point Arithmetic, except that the conversion is always rounded ! 4259: | toward zero. If `a' is a NaN, the largest positive integer is returned. ! 4260: | Otherwise, if the conversion overflows, the largest integer with the same ! 4261: | sign as `a' is returned. ! 4262: *----------------------------------------------------------------------------*/ ! 4263: ! 4264: int32 floatx80_to_int32_round_to_zero( floatx80 a ) ! 4265: { ! 4266: flag aSign; ! 4267: int32 aExp, shiftCount; ! 4268: bits64 aSig, savedASig; ! 4269: int32 z; ! 4270: ! 4271: aSig = extractFloatx80Frac( a ); ! 4272: aExp = extractFloatx80Exp( a ); ! 4273: aSign = extractFloatx80Sign( a ); ! 4274: if ( 0x401E < aExp ) { ! 4275: if ( ( aExp == 0x7FFF ) && (bits64) ( aSig<<1 ) ) aSign = 0; ! 4276: goto invalid; ! 4277: } ! 4278: else if ( aExp < 0x3FFF ) { ! 4279: if ( aExp || aSig ) float_exception_flags |= float_flag_inexact; ! 4280: return 0; ! 4281: } ! 4282: shiftCount = 0x403E - aExp; ! 4283: savedASig = aSig; ! 4284: aSig >>= shiftCount; ! 4285: z = aSig; ! 4286: if ( aSign ) z = - z; ! 4287: z = (sbits32) z; ! 4288: if ( ( z < 0 ) ^ aSign ) { ! 4289: invalid: ! 4290: float_raise( float_flag_invalid ); ! 4291: return aSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 4292: } ! 4293: if ( ( aSig<<shiftCount ) != savedASig ) { ! 4294: float_exception_flags |= float_flag_inexact; ! 4295: } ! 4296: return z; ! 4297: ! 4298: } ! 4299: ! 4300: /*---------------------------------------------------------------------------- ! 4301: | Returns the result of converting the extended double-precision floating- ! 4302: | point value `a' to the 64-bit two's complement integer format. The ! 4303: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4304: | Floating-Point Arithmetic---which means in particular that the conversion ! 4305: | is rounded according to the current rounding mode. If `a' is a NaN, ! 4306: | the largest positive integer is returned. Otherwise, if the conversion ! 4307: | overflows, the largest integer with the same sign as `a' is returned. ! 4308: *----------------------------------------------------------------------------*/ ! 4309: ! 4310: int64 floatx80_to_int64( floatx80 a ) ! 4311: { ! 4312: flag aSign; ! 4313: int32 aExp, shiftCount; ! 4314: bits64 aSig, aSigExtra; ! 4315: ! 4316: aSig = extractFloatx80Frac( a ); ! 4317: aExp = extractFloatx80Exp( a ); ! 4318: aSign = extractFloatx80Sign( a ); ! 4319: shiftCount = 0x403E - aExp; ! 4320: if ( shiftCount <= 0 ) { ! 4321: if ( shiftCount ) { ! 4322: float_raise( float_flag_invalid ); ! 4323: if ( ! aSign ! 4324: || ( ( aExp == 0x7FFF ) ! 4325: && ( (bits64) ( aSig<<1 ) ) ) ! 4326: ) { ! 4327: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 4328: } ! 4329: return (sbits64) LIT64( 0x8000000000000000 ); ! 4330: } ! 4331: aSigExtra = 0; ! 4332: } ! 4333: else { ! 4334: shift64ExtraRightJamming( aSig, 0, shiftCount, &aSig, &aSigExtra ); ! 4335: } ! 4336: return roundAndPackInt64( aSign, aSig, aSigExtra ); ! 4337: ! 4338: } ! 4339: ! 4340: /*---------------------------------------------------------------------------- ! 4341: | Returns the result of converting the extended double-precision floating- ! 4342: | point value `a' to the 64-bit two's complement integer format. The ! 4343: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4344: | Floating-Point Arithmetic, except that the conversion is always rounded ! 4345: | toward zero. If `a' is a NaN, the largest positive integer is returned. ! 4346: | Otherwise, if the conversion overflows, the largest integer with the same ! 4347: | sign as `a' is returned. ! 4348: *----------------------------------------------------------------------------*/ ! 4349: ! 4350: int64 floatx80_to_int64_round_to_zero( floatx80 a ) ! 4351: { ! 4352: flag aSign; ! 4353: int32 aExp, shiftCount; ! 4354: bits64 aSig; ! 4355: int64 z; ! 4356: ! 4357: aSig = extractFloatx80Frac( a ); ! 4358: aExp = extractFloatx80Exp( a ); ! 4359: aSign = extractFloatx80Sign( a ); ! 4360: shiftCount = aExp - 0x403E; ! 4361: if ( 0 <= shiftCount ) { ! 4362: aSig &= LIT64( 0x7FFFFFFFFFFFFFFF ); ! 4363: if ( ( a.high != 0xC03E ) || aSig ) { ! 4364: float_raise( float_flag_invalid ); ! 4365: if ( ! aSign || ( ( aExp == 0x7FFF ) && aSig ) ) { ! 4366: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 4367: } ! 4368: } ! 4369: return (sbits64) LIT64( 0x8000000000000000 ); ! 4370: } ! 4371: else if ( aExp < 0x3FFF ) { ! 4372: if ( aExp | aSig ) float_exception_flags |= float_flag_inexact; ! 4373: return 0; ! 4374: } ! 4375: z = aSig>>( - shiftCount ); ! 4376: if ( (bits64) ( aSig<<( shiftCount & 63 ) ) ) { ! 4377: float_exception_flags |= float_flag_inexact; ! 4378: } ! 4379: if ( aSign ) z = - z; ! 4380: return z; ! 4381: ! 4382: } ! 4383: ! 4384: /*---------------------------------------------------------------------------- ! 4385: | Returns the result of converting the extended double-precision floating- ! 4386: | point value `a' to the single-precision floating-point format. The ! 4387: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4388: | Floating-Point Arithmetic. ! 4389: *----------------------------------------------------------------------------*/ ! 4390: ! 4391: float32 floatx80_to_float32( floatx80 a ) ! 4392: { ! 4393: flag aSign; ! 4394: int32 aExp; ! 4395: bits64 aSig; ! 4396: ! 4397: aSig = extractFloatx80Frac( a ); ! 4398: aExp = extractFloatx80Exp( a ); ! 4399: aSign = extractFloatx80Sign( a ); ! 4400: if ( aExp == 0x7FFF ) { ! 4401: if ( (bits64) ( aSig<<1 ) ) { ! 4402: return commonNaNToFloat32( floatx80ToCommonNaN( a ) ); ! 4403: } ! 4404: return packFloat32( aSign, 0xFF, 0 ); ! 4405: } ! 4406: #ifdef SOFTFLOAT_68K ! 4407: if ( aExp == 0 ) { ! 4408: if ( aSig == 0) return packFloat32( aSign, 0, 0 ); ! 4409: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4410: } ! 4411: shift64RightJamming( aSig, 33, &aSig ); ! 4412: aExp -= 0x3F81; ! 4413: #else ! 4414: shift64RightJamming( aSig, 33, &aSig ); ! 4415: if ( aExp || aSig ) aExp -= 0x3F81; ! 4416: #endif ! 4417: return roundAndPackFloat32( aSign, aExp, aSig ); ! 4418: ! 4419: } ! 4420: ! 4421: /*---------------------------------------------------------------------------- ! 4422: | Returns the result of converting the extended double-precision floating- ! 4423: | point value `a' to the double-precision floating-point format. The ! 4424: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4425: | Floating-Point Arithmetic. ! 4426: *----------------------------------------------------------------------------*/ ! 4427: ! 4428: float64 floatx80_to_float64( floatx80 a ) ! 4429: { ! 4430: flag aSign; ! 4431: int32 aExp; ! 4432: bits64 aSig, zSig; ! 4433: ! 4434: aSig = extractFloatx80Frac( a ); ! 4435: aExp = extractFloatx80Exp( a ); ! 4436: aSign = extractFloatx80Sign( a ); ! 4437: if ( aExp == 0x7FFF ) { ! 4438: if ( (bits64) ( aSig<<1 ) ) { ! 4439: return commonNaNToFloat64( floatx80ToCommonNaN( a ) ); ! 4440: } ! 4441: return packFloat64( aSign, 0x7FF, 0 ); ! 4442: } ! 4443: #ifdef SOFTFLOAT_68K ! 4444: if ( aExp == 0 ) { ! 4445: if ( aSig == 0) return packFloat64( aSign, 0, 0 ); ! 4446: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4447: } ! 4448: shift64RightJamming( aSig, 1, &zSig ); ! 4449: aExp -= 0x3C01; ! 4450: #else ! 4451: shift64RightJamming( aSig, 1, &zSig ); ! 4452: if ( aExp || aSig ) aExp -= 0x3C01; ! 4453: #endif ! 4454: return roundAndPackFloat64( aSign, aExp, zSig ); ! 4455: ! 4456: } ! 4457: ! 4458: #ifdef SOFTFLOAT_68K // 31-01-2017 ! 4459: /*---------------------------------------------------------------------------- ! 4460: | Returns the result of converting the extended double-precision floating- ! 4461: | point value `a' to the extended double-precision floating-point format. ! 4462: | The conversion is performed according to the IEC/IEEE Standard for Binary ! 4463: | Floating-Point Arithmetic. ! 4464: *----------------------------------------------------------------------------*/ ! 4465: ! 4466: floatx80 floatx80_to_floatx80( floatx80 a ) ! 4467: { ! 4468: flag aSign; ! 4469: int32 aExp; ! 4470: bits64 aSig; ! 4471: ! 4472: aSig = extractFloatx80Frac( a ); ! 4473: aExp = extractFloatx80Exp( a ); ! 4474: aSign = extractFloatx80Sign( a ); ! 4475: ! 4476: if ( aExp == 0x7FFF && (bits64) ( aSig<<1 ) ) { ! 4477: return propagateFloatx80NaNOneArg( a ); ! 4478: } ! 4479: if ( aExp == 0 && aSig != 0 ) { ! 4480: return normalizeRoundAndPackFloatx80( floatx80_rounding_precision, aSign, aExp, aSig, 0 ); ! 4481: } ! 4482: return a; ! 4483: ! 4484: } ! 4485: #endif ! 4486: ! 4487: #ifdef FLOAT128 ! 4488: ! 4489: /*---------------------------------------------------------------------------- ! 4490: | Returns the result of converting the extended double-precision floating- ! 4491: | point value `a' to the quadruple-precision floating-point format. The ! 4492: | conversion is performed according to the IEC/IEEE Standard for Binary ! 4493: | Floating-Point Arithmetic. ! 4494: *----------------------------------------------------------------------------*/ ! 4495: ! 4496: float128 floatx80_to_float128( floatx80 a ) ! 4497: { ! 4498: flag aSign; ! 4499: int32 aExp; ! 4500: bits64 aSig, zSig0, zSig1; ! 4501: ! 4502: aSig = extractFloatx80Frac( a ); ! 4503: aExp = extractFloatx80Exp( a ); ! 4504: aSign = extractFloatx80Sign( a ); ! 4505: if ( ( aExp == 0x7FFF ) && (bits64) ( aSig<<1 ) ) { ! 4506: return commonNaNToFloat128( floatx80ToCommonNaN( a ) ); ! 4507: } ! 4508: #ifdef SOFTFLOAT_68K ! 4509: if ( aExp == 0 ) { ! 4510: if ( aSig == 0 ) return packFloat128( aSign, 0, 0, 0 ); ! 4511: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4512: } ! 4513: #endif ! 4514: shift128Right( aSig<<1, 0, 16, &zSig0, &zSig1 ); ! 4515: return packFloat128( aSign, aExp, zSig0, zSig1 ); ! 4516: ! 4517: } ! 4518: ! 4519: #endif ! 4520: ! 4521: #ifdef SOFTFLOAT_68K // 30-01-2016: Added for Previous ! 4522: floatx80 floatx80_round32( floatx80 a ) ! 4523: { ! 4524: flag aSign; ! 4525: int32 aExp; ! 4526: bits64 aSig; ! 4527: ! 4528: aSig = extractFloatx80Frac( a ); ! 4529: aExp = extractFloatx80Exp( a ); ! 4530: aSign = extractFloatx80Sign( a ); ! 4531: ! 4532: if ( aExp == 0x7FFF || aSig == 0 ) { ! 4533: return a; ! 4534: } ! 4535: if ( aExp == 0 ) { ! 4536: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4537: } ! 4538: ! 4539: return roundSigAndPackFloatx80( 32, aSign, aExp, aSig, 0 ); ! 4540: ! 4541: } ! 4542: ! 4543: floatx80 floatx80_round64( floatx80 a ) ! 4544: { ! 4545: flag aSign; ! 4546: int32 aExp; ! 4547: bits64 aSig; ! 4548: ! 4549: aSig = extractFloatx80Frac( a ); ! 4550: aExp = extractFloatx80Exp( a ); ! 4551: aSign = extractFloatx80Sign( a ); ! 4552: ! 4553: if ( aExp == 0x7FFF || aSig == 0 ) { ! 4554: return a; ! 4555: } ! 4556: if ( aExp == 0 ) { ! 4557: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4558: } ! 4559: ! 4560: return roundSigAndPackFloatx80( 64, aSign, aExp, aSig, 0 ); ! 4561: ! 4562: } ! 4563: ! 4564: floatx80 floatx80_round_to_float32( floatx80 a ) ! 4565: { ! 4566: flag aSign; ! 4567: int32 aExp; ! 4568: bits64 aSig; ! 4569: ! 4570: aSign = extractFloatx80Sign( a ); ! 4571: aSig = extractFloatx80Frac( a ); ! 4572: aExp = extractFloatx80Exp( a ); ! 4573: ! 4574: if ( aExp == 0x7FFF ) { ! 4575: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 4576: return a; ! 4577: } ! 4578: if ( aExp == 0 ) { ! 4579: if ( aSig == 0 ) return a; ! 4580: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4581: } ! 4582: ! 4583: return roundAndPackFloatx80( 32, aSign, aExp, aSig, 0 ); ! 4584: ! 4585: } ! 4586: ! 4587: floatx80 floatx80_round_to_float64( floatx80 a ) ! 4588: { ! 4589: flag aSign; ! 4590: int32 aExp; ! 4591: bits64 aSig; ! 4592: ! 4593: aSign = extractFloatx80Sign( a ); ! 4594: aSig = extractFloatx80Frac( a ); ! 4595: aExp = extractFloatx80Exp( a ); ! 4596: ! 4597: if ( aExp == 0x7FFF ) { ! 4598: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 4599: return a; ! 4600: } ! 4601: if ( aExp == 0 ) { ! 4602: if ( aSig == 0 ) return a; ! 4603: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4604: } ! 4605: ! 4606: return roundAndPackFloatx80( 64, aSign, aExp, aSig, 0 ); ! 4607: ! 4608: } ! 4609: ! 4610: floatx80 floatx80_normalize( floatx80 a ) ! 4611: { ! 4612: flag aSign; ! 4613: int16 aExp; ! 4614: bits64 aSig; ! 4615: int8 shiftCount; ! 4616: ! 4617: aSig = extractFloatx80Frac( a ); ! 4618: aExp = extractFloatx80Exp( a ); ! 4619: aSign = extractFloatx80Sign( a ); ! 4620: ! 4621: if ( aExp == 0x7FFF || aExp == 0 ) return a; ! 4622: if ( aSig == 0 ) return packFloatx80(aSign, 0, 0); ! 4623: ! 4624: shiftCount = countLeadingZeros64( aSig ); ! 4625: ! 4626: if ( shiftCount > aExp ) shiftCount = aExp; ! 4627: ! 4628: aExp -= shiftCount; ! 4629: aSig <<= shiftCount; ! 4630: ! 4631: return packFloatx80( aSign, aExp, aSig ); ! 4632: ! 4633: } ! 4634: ! 4635: floatx80 floatx80_denormalize( floatx80 a, flag eSign) ! 4636: { ! 4637: flag aSign; ! 4638: int32 aExp; ! 4639: bits64 aSig; ! 4640: int32 shiftCount; ! 4641: ! 4642: aSig = extractFloatx80Frac( a ); ! 4643: aExp = extractFloatx80Exp( a ); ! 4644: aSign = extractFloatx80Sign( a ); ! 4645: ! 4646: if ( eSign ) { ! 4647: shiftCount = 0x8000 - aExp; ! 4648: aExp = 0; ! 4649: if (shiftCount > 63) { ! 4650: aSig = 0; ! 4651: } else { ! 4652: aSig >>= shiftCount; ! 4653: } ! 4654: } ! 4655: return packFloatx80(aSign, aExp, aSig); ! 4656: ! 4657: } ! 4658: #endif // end of addition for Previous ! 4659: ! 4660: /*---------------------------------------------------------------------------- ! 4661: | Rounds the extended double-precision floating-point value `a' to an integer, ! 4662: | and returns the result as an extended quadruple-precision floating-point ! 4663: | value. The operation is performed according to the IEC/IEEE Standard for ! 4664: | Binary Floating-Point Arithmetic. ! 4665: *----------------------------------------------------------------------------*/ ! 4666: ! 4667: floatx80 floatx80_round_to_int( floatx80 a ) ! 4668: { ! 4669: flag aSign; ! 4670: int32 aExp; ! 4671: bits64 lastBitMask, roundBitsMask; ! 4672: int8 roundingMode; ! 4673: floatx80 z; ! 4674: ! 4675: aExp = extractFloatx80Exp( a ); ! 4676: if ( 0x403E <= aExp ) { ! 4677: if ( ( aExp == 0x7FFF ) && (bits64) ( extractFloatx80Frac( a )<<1 ) ) { ! 4678: return propagateFloatx80NaNOneArg( a ); ! 4679: } ! 4680: return a; ! 4681: } ! 4682: if ( aExp < 0x3FFF ) { ! 4683: if ( ( aExp == 0 ) ! 4684: #ifdef SOFTFLOAT_68K ! 4685: && ( (bits64) extractFloatx80Frac( a ) == 0 ) ) { ! 4686: #else ! 4687: && ( (bits64) ( extractFloatx80Frac( a )<<1 ) == 0 ) ) { ! 4688: #endif ! 4689: return a; ! 4690: } ! 4691: float_exception_flags |= float_flag_inexact; ! 4692: aSign = extractFloatx80Sign( a ); ! 4693: switch ( float_rounding_mode ) { ! 4694: case float_round_nearest_even: ! 4695: if ( ( aExp == 0x3FFE ) && (bits64) ( extractFloatx80Frac( a )<<1 ) ! 4696: ) { ! 4697: return ! 4698: packFloatx80( aSign, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 4699: } ! 4700: break; ! 4701: case float_round_down: ! 4702: return ! 4703: aSign ? ! 4704: packFloatx80( 1, 0x3FFF, LIT64( 0x8000000000000000 ) ) ! 4705: : packFloatx80( 0, 0, 0 ); ! 4706: case float_round_up: ! 4707: return ! 4708: aSign ? packFloatx80( 1, 0, 0 ) ! 4709: : packFloatx80( 0, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 4710: } ! 4711: return packFloatx80( aSign, 0, 0 ); ! 4712: } ! 4713: lastBitMask = 1; ! 4714: lastBitMask <<= 0x403E - aExp; ! 4715: roundBitsMask = lastBitMask - 1; ! 4716: z = a; ! 4717: roundingMode = float_rounding_mode; ! 4718: if ( roundingMode == float_round_nearest_even ) { ! 4719: z.low += lastBitMask>>1; ! 4720: if ( ( z.low & roundBitsMask ) == 0 ) z.low &= ~ lastBitMask; ! 4721: } ! 4722: else if ( roundingMode != float_round_to_zero ) { ! 4723: if ( extractFloatx80Sign( z ) ^ ( roundingMode == float_round_up ) ) { ! 4724: z.low += roundBitsMask; ! 4725: } ! 4726: } ! 4727: z.low &= ~ roundBitsMask; ! 4728: if ( z.low == 0 ) { ! 4729: ++z.high; ! 4730: z.low = LIT64( 0x8000000000000000 ); ! 4731: } ! 4732: if ( z.low != a.low ) float_exception_flags |= float_flag_inexact; ! 4733: return z; ! 4734: ! 4735: } ! 4736: ! 4737: #ifdef SOFTFLOAT_68K // 09-01-2017: Added for Previous ! 4738: floatx80 floatx80_round_to_int_toward_zero( floatx80 a ) ! 4739: { ! 4740: flag aSign; ! 4741: int32 aExp; ! 4742: bits64 lastBitMask, roundBitsMask; ! 4743: floatx80 z; ! 4744: ! 4745: aExp = extractFloatx80Exp( a ); ! 4746: if ( 0x403E <= aExp ) { ! 4747: if ( ( aExp == 0x7FFF ) && (bits64) ( extractFloatx80Frac( a )<<1 ) ) { ! 4748: return propagateFloatx80NaNOneArg( a ); ! 4749: } ! 4750: return a; ! 4751: } ! 4752: if ( aExp < 0x3FFF ) { ! 4753: if ( ( aExp == 0 ) ! 4754: #ifdef SOFTFLOAT_68K ! 4755: && ( (bits64) extractFloatx80Frac( a ) == 0 ) ) { ! 4756: #else ! 4757: && ( (bits64) ( extractFloatx80Frac( a )<<1 ) == 0 ) ) { ! 4758: #endif ! 4759: return a; ! 4760: } ! 4761: float_exception_flags |= float_flag_inexact; ! 4762: aSign = extractFloatx80Sign( a ); ! 4763: return packFloatx80( aSign, 0, 0 ); ! 4764: } ! 4765: lastBitMask = 1; ! 4766: lastBitMask <<= 0x403E - aExp; ! 4767: roundBitsMask = lastBitMask - 1; ! 4768: z = a; ! 4769: z.low &= ~ roundBitsMask; ! 4770: if ( z.low == 0 ) { ! 4771: ++z.high; ! 4772: z.low = LIT64( 0x8000000000000000 ); ! 4773: } ! 4774: if ( z.low != a.low ) float_exception_flags |= float_flag_inexact; ! 4775: return z; ! 4776: ! 4777: } ! 4778: #endif // End of addition for Previous ! 4779: ! 4780: /*---------------------------------------------------------------------------- ! 4781: | Returns the result of adding the absolute values of the extended double- ! 4782: | precision floating-point values `a' and `b'. If `zSign' is 1, the sum is ! 4783: | negated before being returned. `zSign' is ignored if the result is a NaN. ! 4784: | The addition is performed according to the IEC/IEEE Standard for Binary ! 4785: | Floating-Point Arithmetic. ! 4786: *----------------------------------------------------------------------------*/ ! 4787: ! 4788: static floatx80 addFloatx80Sigs( floatx80 a, floatx80 b, flag zSign ) ! 4789: { ! 4790: int32 aExp, bExp, zExp; ! 4791: bits64 aSig, bSig, zSig0, zSig1; ! 4792: int32 expDiff; ! 4793: ! 4794: aSig = extractFloatx80Frac( a ); ! 4795: aExp = extractFloatx80Exp( a ); ! 4796: bSig = extractFloatx80Frac( b ); ! 4797: bExp = extractFloatx80Exp( b ); ! 4798: #ifdef SOFTFLOAT_68K ! 4799: if ( aExp == 0 ) { ! 4800: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 4801: } ! 4802: if ( bExp == 0 ) { ! 4803: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 4804: } ! 4805: #endif ! 4806: expDiff = aExp - bExp; ! 4807: if ( 0 < expDiff ) { ! 4808: if ( aExp == 0x7FFF ) { ! 4809: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 4810: return a; ! 4811: } ! 4812: #ifndef SOFTFLOAT_68K ! 4813: if ( bExp == 0 ) --expDiff; ! 4814: #endif ! 4815: shift64ExtraRightJamming( bSig, 0, expDiff, &bSig, &zSig1 ); ! 4816: zExp = aExp; ! 4817: } ! 4818: else if ( expDiff < 0 ) { ! 4819: if ( bExp == 0x7FFF ) { ! 4820: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 4821: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 4822: } ! 4823: #ifndef SOFTFLOAT_68K ! 4824: if ( aExp == 0 ) ++expDiff; ! 4825: #endif ! 4826: shift64ExtraRightJamming( aSig, 0, - expDiff, &aSig, &zSig1 ); ! 4827: zExp = bExp; ! 4828: } ! 4829: else { ! 4830: if ( aExp == 0x7FFF ) { ! 4831: if ( (bits64) ( ( aSig | bSig )<<1 ) ) { ! 4832: return propagateFloatx80NaN( a, b ); ! 4833: } ! 4834: return a; ! 4835: } ! 4836: zSig1 = 0; ! 4837: zSig0 = aSig + bSig; ! 4838: #ifndef SOFTFLOAT_68K ! 4839: if ( aExp == 0 ) { ! 4840: normalizeFloatx80Subnormal( zSig0, &zExp, &zSig0 ); ! 4841: goto roundAndPack; ! 4842: } ! 4843: #endif ! 4844: zExp = aExp; ! 4845: #ifdef SOFTFLOAT_68K ! 4846: if ( aSig == 0 && bSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 4847: if ( aSig == 0 || bSig == 0 ) goto roundAndPack; ! 4848: #endif ! 4849: goto shiftRight1; ! 4850: } ! 4851: zSig0 = aSig + bSig; ! 4852: if ( (sbits64) zSig0 < 0 ) goto roundAndPack; ! 4853: shiftRight1: ! 4854: shift64ExtraRightJamming( zSig0, zSig1, 1, &zSig0, &zSig1 ); ! 4855: zSig0 |= LIT64( 0x8000000000000000 ); ! 4856: ++zExp; ! 4857: roundAndPack: ! 4858: return ! 4859: roundAndPackFloatx80( ! 4860: floatx80_rounding_precision, zSign, zExp, zSig0, zSig1 ); ! 4861: ! 4862: } ! 4863: ! 4864: /*---------------------------------------------------------------------------- ! 4865: | Returns the result of subtracting the absolute values of the extended ! 4866: | double-precision floating-point values `a' and `b'. If `zSign' is 1, the ! 4867: | difference is negated before being returned. `zSign' is ignored if the ! 4868: | result is a NaN. The subtraction is performed according to the IEC/IEEE ! 4869: | Standard for Binary Floating-Point Arithmetic. ! 4870: *----------------------------------------------------------------------------*/ ! 4871: ! 4872: static floatx80 subFloatx80Sigs( floatx80 a, floatx80 b, flag zSign ) ! 4873: { ! 4874: int32 aExp, bExp, zExp; ! 4875: bits64 aSig, bSig, zSig0, zSig1; ! 4876: int32 expDiff; ! 4877: floatx80 z; ! 4878: ! 4879: aSig = extractFloatx80Frac( a ); ! 4880: aExp = extractFloatx80Exp( a ); ! 4881: bSig = extractFloatx80Frac( b ); ! 4882: bExp = extractFloatx80Exp( b ); ! 4883: expDiff = aExp - bExp; ! 4884: if ( 0 < expDiff ) goto aExpBigger; ! 4885: if ( expDiff < 0 ) goto bExpBigger; ! 4886: if ( aExp == 0x7FFF ) { ! 4887: if ( (bits64) ( ( aSig | bSig )<<1 ) ) { ! 4888: return propagateFloatx80NaN( a, b ); ! 4889: } ! 4890: float_raise( float_flag_invalid ); ! 4891: z.low = floatx80_default_nan_low; ! 4892: z.high = floatx80_default_nan_high; ! 4893: return z; ! 4894: } ! 4895: #ifndef SOFTFLOAT_68K ! 4896: if ( aExp == 0 ) { ! 4897: aExp = 1; ! 4898: bExp = 1; ! 4899: } ! 4900: #endif ! 4901: zSig1 = 0; ! 4902: if ( bSig < aSig ) goto aBigger; ! 4903: if ( aSig < bSig ) goto bBigger; ! 4904: return packFloatx80( float_rounding_mode == float_round_down, 0, 0 ); ! 4905: bExpBigger: ! 4906: if ( bExp == 0x7FFF ) { ! 4907: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 4908: return packFloatx80( zSign ^ 1, 0x7FFF, floatx80_default_infinity_low ); ! 4909: } ! 4910: #ifndef SOFTFLOAT_68K ! 4911: if ( aExp == 0 ) ++expDiff; ! 4912: #endif ! 4913: shift128RightJamming( aSig, 0, - expDiff, &aSig, &zSig1 ); ! 4914: bBigger: ! 4915: sub128( bSig, 0, aSig, zSig1, &zSig0, &zSig1 ); ! 4916: zExp = bExp; ! 4917: zSign ^= 1; ! 4918: goto normalizeRoundAndPack; ! 4919: aExpBigger: ! 4920: if ( aExp == 0x7FFF ) { ! 4921: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 4922: return a; ! 4923: } ! 4924: #ifndef SOFTFLOAT_68K ! 4925: if ( bExp == 0 ) --expDiff; ! 4926: #endif ! 4927: shift128RightJamming( bSig, 0, expDiff, &bSig, &zSig1 ); ! 4928: aBigger: ! 4929: sub128( aSig, 0, bSig, zSig1, &zSig0, &zSig1 ); ! 4930: zExp = aExp; ! 4931: normalizeRoundAndPack: ! 4932: return ! 4933: normalizeRoundAndPackFloatx80( ! 4934: floatx80_rounding_precision, zSign, zExp, zSig0, zSig1 ); ! 4935: ! 4936: } ! 4937: ! 4938: /*---------------------------------------------------------------------------- ! 4939: | Returns the result of adding the extended double-precision floating-point ! 4940: | values `a' and `b'. The operation is performed according to the IEC/IEEE ! 4941: | Standard for Binary Floating-Point Arithmetic. ! 4942: *----------------------------------------------------------------------------*/ ! 4943: ! 4944: floatx80 floatx80_add( floatx80 a, floatx80 b ) ! 4945: { ! 4946: flag aSign, bSign; ! 4947: ! 4948: aSign = extractFloatx80Sign( a ); ! 4949: bSign = extractFloatx80Sign( b ); ! 4950: if ( aSign == bSign ) { ! 4951: return addFloatx80Sigs( a, b, aSign ); ! 4952: } ! 4953: else { ! 4954: return subFloatx80Sigs( a, b, aSign ); ! 4955: } ! 4956: ! 4957: } ! 4958: ! 4959: /*---------------------------------------------------------------------------- ! 4960: | Returns the result of subtracting the extended double-precision floating- ! 4961: | point values `a' and `b'. The operation is performed according to the ! 4962: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4963: *----------------------------------------------------------------------------*/ ! 4964: ! 4965: floatx80 floatx80_sub( floatx80 a, floatx80 b ) ! 4966: { ! 4967: flag aSign, bSign; ! 4968: ! 4969: aSign = extractFloatx80Sign( a ); ! 4970: bSign = extractFloatx80Sign( b ); ! 4971: if ( aSign == bSign ) { ! 4972: return subFloatx80Sigs( a, b, aSign ); ! 4973: } ! 4974: else { ! 4975: return addFloatx80Sigs( a, b, aSign ); ! 4976: } ! 4977: ! 4978: } ! 4979: ! 4980: /*---------------------------------------------------------------------------- ! 4981: | Returns the result of multiplying the extended double-precision floating- ! 4982: | point values `a' and `b'. The operation is performed according to the ! 4983: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 4984: *----------------------------------------------------------------------------*/ ! 4985: ! 4986: floatx80 floatx80_mul( floatx80 a, floatx80 b ) ! 4987: { ! 4988: flag aSign, bSign, zSign; ! 4989: int32 aExp, bExp, zExp; ! 4990: bits64 aSig, bSig, zSig0, zSig1; ! 4991: floatx80 z; ! 4992: ! 4993: aSig = extractFloatx80Frac( a ); ! 4994: aExp = extractFloatx80Exp( a ); ! 4995: aSign = extractFloatx80Sign( a ); ! 4996: bSig = extractFloatx80Frac( b ); ! 4997: bExp = extractFloatx80Exp( b ); ! 4998: bSign = extractFloatx80Sign( b ); ! 4999: zSign = aSign ^ bSign; ! 5000: if ( aExp == 0x7FFF ) { ! 5001: if ( (bits64) ( aSig<<1 ) ! 5002: || ( ( bExp == 0x7FFF ) && (bits64) ( bSig<<1 ) ) ) { ! 5003: return propagateFloatx80NaN( a, b ); ! 5004: } ! 5005: if ( ( bExp | bSig ) == 0 ) goto invalid; ! 5006: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5007: } ! 5008: if ( bExp == 0x7FFF ) { ! 5009: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5010: if ( ( aExp | aSig ) == 0 ) { ! 5011: invalid: ! 5012: float_raise( float_flag_invalid ); ! 5013: z.low = floatx80_default_nan_low; ! 5014: z.high = floatx80_default_nan_high; ! 5015: return z; ! 5016: } ! 5017: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5018: } ! 5019: if ( aExp == 0 ) { ! 5020: if ( aSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5021: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5022: } ! 5023: if ( bExp == 0 ) { ! 5024: if ( bSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5025: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5026: } ! 5027: zExp = aExp + bExp - 0x3FFE; ! 5028: mul64To128( aSig, bSig, &zSig0, &zSig1 ); ! 5029: if ( 0 < (sbits64) zSig0 ) { ! 5030: shortShift128Left( zSig0, zSig1, 1, &zSig0, &zSig1 ); ! 5031: --zExp; ! 5032: } ! 5033: return ! 5034: roundAndPackFloatx80( ! 5035: floatx80_rounding_precision, zSign, zExp, zSig0, zSig1 ); ! 5036: ! 5037: } ! 5038: ! 5039: #ifdef SOFTFLOAT_68K // 21-01-2017: Added for Previous ! 5040: floatx80 floatx80_sglmul( floatx80 a, floatx80 b ) ! 5041: { ! 5042: flag aSign, bSign, zSign; ! 5043: int32 aExp, bExp, zExp; ! 5044: bits64 aSig, bSig, zSig0, zSig1; ! 5045: floatx80 z; ! 5046: ! 5047: aSig = extractFloatx80Frac( a ); ! 5048: aExp = extractFloatx80Exp( a ); ! 5049: aSign = extractFloatx80Sign( a ); ! 5050: bSig = extractFloatx80Frac( b ); ! 5051: bExp = extractFloatx80Exp( b ); ! 5052: bSign = extractFloatx80Sign( b ); ! 5053: zSign = aSign ^ bSign; ! 5054: if ( aExp == 0x7FFF ) { ! 5055: if ( (bits64) ( aSig<<1 ) ! 5056: || ( ( bExp == 0x7FFF ) && (bits64) ( bSig<<1 ) ) ) { ! 5057: return propagateFloatx80NaN( a, b ); ! 5058: } ! 5059: if ( ( bExp | bSig ) == 0 ) goto invalid; ! 5060: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5061: } ! 5062: if ( bExp == 0x7FFF ) { ! 5063: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5064: if ( ( aExp | aSig ) == 0 ) { ! 5065: invalid: ! 5066: float_raise( float_flag_invalid ); ! 5067: z.low = floatx80_default_nan_low; ! 5068: z.high = floatx80_default_nan_high; ! 5069: return z; ! 5070: } ! 5071: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5072: } ! 5073: if ( aExp == 0 ) { ! 5074: if ( aSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5075: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5076: } ! 5077: if ( bExp == 0 ) { ! 5078: if ( bSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5079: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5080: } ! 5081: aSig &= LIT64( 0xFFFFFF0000000000 ); ! 5082: bSig &= LIT64( 0xFFFFFF0000000000 ); ! 5083: zExp = aExp + bExp - 0x3FFE; ! 5084: mul64To128( aSig, bSig, &zSig0, &zSig1 ); ! 5085: if ( 0 < (sbits64) zSig0 ) { ! 5086: shortShift128Left( zSig0, zSig1, 1, &zSig0, &zSig1 ); ! 5087: --zExp; ! 5088: } ! 5089: return roundSigAndPackFloatx80( 32, zSign, zExp, zSig0, zSig1 ); ! 5090: ! 5091: } ! 5092: #endif // End of addition for Previous ! 5093: ! 5094: /*---------------------------------------------------------------------------- ! 5095: | Returns the result of dividing the extended double-precision floating-point ! 5096: | value `a' by the corresponding value `b'. The operation is performed ! 5097: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 5098: *----------------------------------------------------------------------------*/ ! 5099: ! 5100: floatx80 floatx80_div( floatx80 a, floatx80 b ) ! 5101: { ! 5102: flag aSign, bSign, zSign; ! 5103: int32 aExp, bExp, zExp; ! 5104: bits64 aSig, bSig, zSig0, zSig1; ! 5105: bits64 rem0, rem1, rem2, term0, term1, term2; ! 5106: floatx80 z; ! 5107: ! 5108: aSig = extractFloatx80Frac( a ); ! 5109: aExp = extractFloatx80Exp( a ); ! 5110: aSign = extractFloatx80Sign( a ); ! 5111: bSig = extractFloatx80Frac( b ); ! 5112: bExp = extractFloatx80Exp( b ); ! 5113: bSign = extractFloatx80Sign( b ); ! 5114: zSign = aSign ^ bSign; ! 5115: if ( aExp == 0x7FFF ) { ! 5116: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5117: if ( bExp == 0x7FFF ) { ! 5118: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5119: goto invalid; ! 5120: } ! 5121: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5122: } ! 5123: if ( bExp == 0x7FFF ) { ! 5124: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5125: return packFloatx80( zSign, 0, 0 ); ! 5126: } ! 5127: if ( bExp == 0 ) { ! 5128: if ( bSig == 0 ) { ! 5129: if ( ( aExp | aSig ) == 0 ) { ! 5130: invalid: ! 5131: float_raise( float_flag_invalid ); ! 5132: z.low = floatx80_default_nan_low; ! 5133: z.high = floatx80_default_nan_high; ! 5134: return z; ! 5135: } ! 5136: float_raise( float_flag_divbyzero ); ! 5137: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5138: } ! 5139: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5140: } ! 5141: if ( aExp == 0 ) { ! 5142: if ( aSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5143: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5144: } ! 5145: zExp = aExp - bExp + 0x3FFE; ! 5146: rem1 = 0; ! 5147: if ( bSig <= aSig ) { ! 5148: shift128Right( aSig, 0, 1, &aSig, &rem1 ); ! 5149: ++zExp; ! 5150: } ! 5151: zSig0 = estimateDiv128To64( aSig, rem1, bSig ); ! 5152: mul64To128( bSig, zSig0, &term0, &term1 ); ! 5153: sub128( aSig, rem1, term0, term1, &rem0, &rem1 ); ! 5154: while ( (sbits64) rem0 < 0 ) { ! 5155: --zSig0; ! 5156: add128( rem0, rem1, 0, bSig, &rem0, &rem1 ); ! 5157: } ! 5158: zSig1 = estimateDiv128To64( rem1, 0, bSig ); ! 5159: if ( (bits64) ( zSig1<<1 ) <= 8 ) { ! 5160: mul64To128( bSig, zSig1, &term1, &term2 ); ! 5161: sub128( rem1, 0, term1, term2, &rem1, &rem2 ); ! 5162: while ( (sbits64) rem1 < 0 ) { ! 5163: --zSig1; ! 5164: add128( rem1, rem2, 0, bSig, &rem1, &rem2 ); ! 5165: } ! 5166: zSig1 |= ( ( rem1 | rem2 ) != 0 ); ! 5167: } ! 5168: return ! 5169: roundAndPackFloatx80( ! 5170: floatx80_rounding_precision, zSign, zExp, zSig0, zSig1 ); ! 5171: ! 5172: } ! 5173: ! 5174: #ifdef SOFTFLOAT_68K // 21-01-2017: Addition for Previous ! 5175: floatx80 floatx80_sgldiv( floatx80 a, floatx80 b ) ! 5176: { ! 5177: flag aSign, bSign, zSign; ! 5178: int32 aExp, bExp, zExp; ! 5179: bits64 aSig, bSig, zSig0, zSig1; ! 5180: bits64 rem0, rem1, rem2, term0, term1, term2; ! 5181: floatx80 z; ! 5182: ! 5183: aSig = extractFloatx80Frac( a ); ! 5184: aExp = extractFloatx80Exp( a ); ! 5185: aSign = extractFloatx80Sign( a ); ! 5186: bSig = extractFloatx80Frac( b ); ! 5187: bExp = extractFloatx80Exp( b ); ! 5188: bSign = extractFloatx80Sign( b ); ! 5189: zSign = aSign ^ bSign; ! 5190: if ( aExp == 0x7FFF ) { ! 5191: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5192: if ( bExp == 0x7FFF ) { ! 5193: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5194: goto invalid; ! 5195: } ! 5196: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5197: } ! 5198: if ( bExp == 0x7FFF ) { ! 5199: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5200: return packFloatx80( zSign, 0, 0 ); ! 5201: } ! 5202: if ( bExp == 0 ) { ! 5203: if ( bSig == 0 ) { ! 5204: if ( ( aExp | aSig ) == 0 ) { ! 5205: invalid: ! 5206: float_raise( float_flag_invalid ); ! 5207: z.low = floatx80_default_nan_low; ! 5208: z.high = floatx80_default_nan_high; ! 5209: return z; ! 5210: } ! 5211: float_raise( float_flag_divbyzero ); ! 5212: return packFloatx80( zSign, 0x7FFF, floatx80_default_infinity_low ); ! 5213: } ! 5214: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5215: } ! 5216: if ( aExp == 0 ) { ! 5217: if ( aSig == 0 ) return packFloatx80( zSign, 0, 0 ); ! 5218: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5219: } ! 5220: zExp = aExp - bExp + 0x3FFE; ! 5221: rem1 = 0; ! 5222: if ( bSig <= aSig ) { ! 5223: shift128Right( aSig, 0, 1, &aSig, &rem1 ); ! 5224: ++zExp; ! 5225: } ! 5226: zSig0 = estimateDiv128To64( aSig, rem1, bSig ); ! 5227: mul64To128( bSig, zSig0, &term0, &term1 ); ! 5228: sub128( aSig, rem1, term0, term1, &rem0, &rem1 ); ! 5229: while ( (sbits64) rem0 < 0 ) { ! 5230: --zSig0; ! 5231: add128( rem0, rem1, 0, bSig, &rem0, &rem1 ); ! 5232: } ! 5233: zSig1 = estimateDiv128To64( rem1, 0, bSig ); ! 5234: if ( (bits64) ( zSig1<<1 ) <= 8 ) { ! 5235: mul64To128( bSig, zSig1, &term1, &term2 ); ! 5236: sub128( rem1, 0, term1, term2, &rem1, &rem2 ); ! 5237: while ( (sbits64) rem1 < 0 ) { ! 5238: --zSig1; ! 5239: add128( rem1, rem2, 0, bSig, &rem1, &rem2 ); ! 5240: } ! 5241: zSig1 |= ( ( rem1 | rem2 ) != 0 ); ! 5242: } ! 5243: return roundSigAndPackFloatx80( 32, zSign, zExp, zSig0, zSig1 ); ! 5244: ! 5245: } ! 5246: #endif // End of addition for Previous ! 5247: ! 5248: /*---------------------------------------------------------------------------- ! 5249: | Returns the remainder of the extended double-precision floating-point value ! 5250: | `a' with respect to the corresponding value `b'. The operation is performed ! 5251: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 5252: *----------------------------------------------------------------------------*/ ! 5253: #ifndef SOFTFLOAT_68K ! 5254: floatx80 floatx80_rem( floatx80 a, floatx80 b ) ! 5255: { ! 5256: flag aSign, zSign; ! 5257: int32 aExp, bExp, expDiff; ! 5258: bits64 aSig0, aSig1, bSig; ! 5259: bits64 q, term0, term1, alternateASig0, alternateASig1; ! 5260: floatx80 z; ! 5261: ! 5262: aSig0 = extractFloatx80Frac( a ); ! 5263: aExp = extractFloatx80Exp( a ); ! 5264: aSign = extractFloatx80Sign( a ); ! 5265: bSig = extractFloatx80Frac( b ); ! 5266: bExp = extractFloatx80Exp( b ); ! 5267: // bSign = extractFloatx80Sign( b ); ! 5268: if ( aExp == 0x7FFF ) { ! 5269: if ( (bits64) ( aSig0<<1 ) ! 5270: || ( ( bExp == 0x7FFF ) && (bits64) ( bSig<<1 ) ) ) { ! 5271: return propagateFloatx80NaN( a, b ); ! 5272: } ! 5273: goto invalid; ! 5274: } ! 5275: if ( bExp == 0x7FFF ) { ! 5276: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5277: return a; ! 5278: } ! 5279: if ( bExp == 0 ) { ! 5280: if ( bSig == 0 ) { ! 5281: invalid: ! 5282: float_raise( float_flag_invalid ); ! 5283: z.low = floatx80_default_nan_low; ! 5284: z.high = floatx80_default_nan_high; ! 5285: return z; ! 5286: } ! 5287: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5288: } ! 5289: if ( aExp == 0 ) { ! 5290: if ( (bits64) ( aSig0<<1 ) == 0 ) return a; ! 5291: normalizeFloatx80Subnormal( aSig0, &aExp, &aSig0 ); ! 5292: } ! 5293: bSig |= LIT64( 0x8000000000000000 ); ! 5294: zSign = aSign; ! 5295: expDiff = aExp - bExp; ! 5296: aSig1 = 0; ! 5297: if ( expDiff < 0 ) { ! 5298: if ( expDiff < -1 ) return a; ! 5299: shift128Right( aSig0, 0, 1, &aSig0, &aSig1 ); ! 5300: expDiff = 0; ! 5301: } ! 5302: q = ( bSig <= aSig0 ); ! 5303: if ( q ) aSig0 -= bSig; ! 5304: expDiff -= 64; ! 5305: while ( 0 < expDiff ) { ! 5306: q = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5307: q = ( 2 < q ) ? q - 2 : 0; ! 5308: mul64To128( bSig, q, &term0, &term1 ); ! 5309: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5310: shortShift128Left( aSig0, aSig1, 62, &aSig0, &aSig1 ); ! 5311: expDiff -= 62; ! 5312: } ! 5313: expDiff += 64; ! 5314: if ( 0 < expDiff ) { ! 5315: q = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5316: q = ( 2 < q ) ? q - 2 : 0; ! 5317: q >>= 64 - expDiff; ! 5318: mul64To128( bSig, q<<( 64 - expDiff ), &term0, &term1 ); ! 5319: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5320: shortShift128Left( 0, bSig, 64 - expDiff, &term0, &term1 ); ! 5321: while ( le128( term0, term1, aSig0, aSig1 ) ) { ! 5322: ++q; ! 5323: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5324: } ! 5325: } ! 5326: else { ! 5327: term1 = 0; ! 5328: term0 = bSig; ! 5329: } ! 5330: sub128( term0, term1, aSig0, aSig1, &alternateASig0, &alternateASig1 ); ! 5331: if ( lt128( alternateASig0, alternateASig1, aSig0, aSig1 ) ! 5332: || ( eq128( alternateASig0, alternateASig1, aSig0, aSig1 ) ! 5333: && ( q & 1 ) ) ! 5334: ) { ! 5335: aSig0 = alternateASig0; ! 5336: aSig1 = alternateASig1; ! 5337: zSign = ! zSign; ! 5338: } ! 5339: return ! 5340: normalizeRoundAndPackFloatx80( ! 5341: 80, zSign, bExp + expDiff, aSig0, aSig1 ); ! 5342: ! 5343: } ! 5344: #else // 09-01-2017: Modified version for Previous ! 5345: floatx80 floatx80_rem( floatx80 a, floatx80 b, bits64 *q, flag *s ) ! 5346: { ! 5347: flag aSign, bSign, zSign; ! 5348: int32 aExp, bExp, expDiff; ! 5349: bits64 aSig0, aSig1, bSig; ! 5350: bits64 qTemp, term0, term1, alternateASig0, alternateASig1; ! 5351: floatx80 z; ! 5352: ! 5353: aSig0 = extractFloatx80Frac( a ); ! 5354: aExp = extractFloatx80Exp( a ); ! 5355: aSign = extractFloatx80Sign( a ); ! 5356: bSig = extractFloatx80Frac( b ); ! 5357: bExp = extractFloatx80Exp( b ); ! 5358: bSign = extractFloatx80Sign( b ); ! 5359: ! 5360: if ( aExp == 0x7FFF ) { ! 5361: if ( (bits64) ( aSig0<<1 ) ! 5362: || ( ( bExp == 0x7FFF ) && (bits64) ( bSig<<1 ) ) ) { ! 5363: return propagateFloatx80NaN( a, b ); ! 5364: } ! 5365: goto invalid; ! 5366: } ! 5367: if ( bExp == 0x7FFF ) { ! 5368: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5369: *s = (aSign != bSign); ! 5370: *q = 0; ! 5371: return a; ! 5372: } ! 5373: if ( bExp == 0 ) { ! 5374: if ( bSig == 0 ) { ! 5375: invalid: ! 5376: float_raise( float_flag_invalid ); ! 5377: z.low = floatx80_default_nan_low; ! 5378: z.high = floatx80_default_nan_high; ! 5379: return z; ! 5380: } ! 5381: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5382: } ! 5383: if ( aExp == 0 ) { ! 5384: #ifdef SOFTFLOAT_68K ! 5385: if ( aSig0 == 0 ) { ! 5386: *s = (aSign != bSign); ! 5387: *q = 0; ! 5388: return a; ! 5389: } ! 5390: #else ! 5391: if ( (bits64) ( aSig0<<1 ) == 0 ) return a; ! 5392: #endif ! 5393: normalizeFloatx80Subnormal( aSig0, &aExp, &aSig0 ); ! 5394: } ! 5395: bSig |= LIT64( 0x8000000000000000 ); ! 5396: zSign = aSign; ! 5397: expDiff = aExp - bExp; ! 5398: *s = (aSign != bSign); ! 5399: aSig1 = 0; ! 5400: if ( expDiff < 0 ) { ! 5401: if ( expDiff < -1 ) return a; ! 5402: shift128Right( aSig0, 0, 1, &aSig0, &aSig1 ); ! 5403: expDiff = 0; ! 5404: } ! 5405: qTemp = ( bSig <= aSig0 ); ! 5406: if ( qTemp ) aSig0 -= bSig; ! 5407: *q = ( expDiff > 63 ) ? 0 : ( qTemp<<expDiff ); ! 5408: expDiff -= 64; ! 5409: while ( 0 < expDiff ) { ! 5410: qTemp = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5411: qTemp = ( 2 < qTemp ) ? qTemp - 2 : 0; ! 5412: mul64To128( bSig, qTemp, &term0, &term1 ); ! 5413: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5414: shortShift128Left( aSig0, aSig1, 62, &aSig0, &aSig1 ); ! 5415: *q = ( expDiff > 63 ) ? 0 : ( qTemp<<expDiff ); ! 5416: expDiff -= 62; ! 5417: } ! 5418: expDiff += 64; ! 5419: if ( 0 < expDiff ) { ! 5420: qTemp = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5421: qTemp = ( 2 < qTemp ) ? qTemp - 2 : 0; ! 5422: qTemp >>= 64 - expDiff; ! 5423: mul64To128( bSig, qTemp<<( 64 - expDiff ), &term0, &term1 ); ! 5424: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5425: shortShift128Left( 0, bSig, 64 - expDiff, &term0, &term1 ); ! 5426: while ( le128( term0, term1, aSig0, aSig1 ) ) { ! 5427: ++qTemp; ! 5428: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5429: } ! 5430: *q += qTemp; ! 5431: } ! 5432: else { ! 5433: term1 = 0; ! 5434: term0 = bSig; ! 5435: } ! 5436: sub128( term0, term1, aSig0, aSig1, &alternateASig0, &alternateASig1 ); ! 5437: if ( lt128( alternateASig0, alternateASig1, aSig0, aSig1 ) ! 5438: || ( eq128( alternateASig0, alternateASig1, aSig0, aSig1 ) ! 5439: && ( qTemp & 1 ) ) ! 5440: ) { ! 5441: aSig0 = alternateASig0; ! 5442: aSig1 = alternateASig1; ! 5443: zSign = ! zSign; ! 5444: ++*q; ! 5445: } ! 5446: return ! 5447: normalizeRoundAndPackFloatx80( ! 5448: 80, zSign, bExp + expDiff, aSig0, aSig1 ); ! 5449: ! 5450: } ! 5451: #endif // End of modification ! 5452: ! 5453: #ifdef SOFTFLOAT_68K // 08-01-2017: Added for Previous ! 5454: /*---------------------------------------------------------------------------- ! 5455: | Returns the modulo remainder of the extended double-precision floating-point ! 5456: | value `a' with respect to the corresponding value `b'. ! 5457: *----------------------------------------------------------------------------*/ ! 5458: ! 5459: floatx80 floatx80_mod( floatx80 a, floatx80 b, bits64 *q, flag *s ) ! 5460: { ! 5461: flag aSign, bSign, zSign; ! 5462: int32 aExp, bExp, expDiff; ! 5463: bits64 aSig0, aSig1, bSig; ! 5464: bits64 qTemp, term0, term1; ! 5465: floatx80 z; ! 5466: ! 5467: aSig0 = extractFloatx80Frac( a ); ! 5468: aExp = extractFloatx80Exp( a ); ! 5469: aSign = extractFloatx80Sign( a ); ! 5470: bSig = extractFloatx80Frac( b ); ! 5471: bExp = extractFloatx80Exp( b ); ! 5472: bSign = extractFloatx80Sign( b ); ! 5473: ! 5474: if ( aExp == 0x7FFF ) { ! 5475: if ( (bits64) ( aSig0<<1 ) ! 5476: || ( ( bExp == 0x7FFF ) && (bits64) ( bSig<<1 ) ) ) { ! 5477: return propagateFloatx80NaN( a, b ); ! 5478: } ! 5479: goto invalid; ! 5480: } ! 5481: if ( bExp == 0x7FFF ) { ! 5482: if ( (bits64) ( bSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5483: *s = (aSign != bSign); ! 5484: *q = 0; ! 5485: return a; ! 5486: } ! 5487: if ( bExp == 0 ) { ! 5488: if ( bSig == 0 ) { ! 5489: invalid: ! 5490: float_raise( float_flag_invalid ); ! 5491: z.low = floatx80_default_nan_low; ! 5492: z.high = floatx80_default_nan_high; ! 5493: return z; ! 5494: } ! 5495: normalizeFloatx80Subnormal( bSig, &bExp, &bSig ); ! 5496: } ! 5497: if ( aExp == 0 ) { ! 5498: #ifdef SOFTFLOAT_68K ! 5499: if ( aSig0 == 0 ) { ! 5500: *s = (aSign != bSign); ! 5501: *q = 0; ! 5502: return a; ! 5503: } ! 5504: #else ! 5505: if ( (bits64) ( aSig0<<1 ) == 0 ) return a; ! 5506: #endif ! 5507: normalizeFloatx80Subnormal( aSig0, &aExp, &aSig0 ); ! 5508: } ! 5509: bSig |= LIT64( 0x8000000000000000 ); ! 5510: zSign = aSign; ! 5511: expDiff = aExp - bExp; ! 5512: *s = (aSign != bSign); ! 5513: aSig1 = 0; ! 5514: if ( expDiff < 0 ) return a; ! 5515: qTemp = ( bSig <= aSig0 ); ! 5516: if ( qTemp ) aSig0 -= bSig; ! 5517: *q = ( expDiff > 63 ) ? 0 : ( qTemp<<expDiff ); ! 5518: expDiff -= 64; ! 5519: while ( 0 < expDiff ) { ! 5520: qTemp = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5521: qTemp = ( 2 < qTemp ) ? qTemp - 2 : 0; ! 5522: mul64To128( bSig, qTemp, &term0, &term1 ); ! 5523: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5524: shortShift128Left( aSig0, aSig1, 62, &aSig0, &aSig1 ); ! 5525: *q = ( expDiff > 63 ) ? 0 : ( qTemp<<expDiff ); ! 5526: expDiff -= 62; ! 5527: } ! 5528: expDiff += 64; ! 5529: if ( 0 < expDiff ) { ! 5530: qTemp = estimateDiv128To64( aSig0, aSig1, bSig ); ! 5531: qTemp = ( 2 < qTemp ) ? qTemp - 2 : 0; ! 5532: qTemp >>= 64 - expDiff; ! 5533: mul64To128( bSig, qTemp<<( 64 - expDiff ), &term0, &term1 ); ! 5534: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5535: shortShift128Left( 0, bSig, 64 - expDiff, &term0, &term1 ); ! 5536: while ( le128( term0, term1, aSig0, aSig1 ) ) { ! 5537: ++qTemp; ! 5538: sub128( aSig0, aSig1, term0, term1, &aSig0, &aSig1 ); ! 5539: } ! 5540: *q += qTemp; ! 5541: } ! 5542: return ! 5543: normalizeRoundAndPackFloatx80( ! 5544: 80, zSign, bExp + expDiff, aSig0, aSig1 ); ! 5545: ! 5546: } ! 5547: #endif // end of addition for Previous ! 5548: ! 5549: /*---------------------------------------------------------------------------- ! 5550: | Returns the square root of the extended double-precision floating-point ! 5551: | value `a'. The operation is performed according to the IEC/IEEE Standard ! 5552: | for Binary Floating-Point Arithmetic. ! 5553: *----------------------------------------------------------------------------*/ ! 5554: ! 5555: floatx80 floatx80_sqrt( floatx80 a ) ! 5556: { ! 5557: flag aSign; ! 5558: int32 aExp, zExp; ! 5559: bits64 aSig0, aSig1, zSig0, zSig1, doubleZSig0; ! 5560: bits64 rem0, rem1, rem2, rem3, term0, term1, term2, term3; ! 5561: floatx80 z; ! 5562: ! 5563: aSig0 = extractFloatx80Frac( a ); ! 5564: aExp = extractFloatx80Exp( a ); ! 5565: aSign = extractFloatx80Sign( a ); ! 5566: if ( aExp == 0x7FFF ) { ! 5567: if ( (bits64) ( aSig0<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 5568: if ( ! aSign ) return a; ! 5569: goto invalid; ! 5570: } ! 5571: if ( aSign ) { ! 5572: if ( ( aExp | aSig0 ) == 0 ) return a; ! 5573: invalid: ! 5574: float_raise( float_flag_invalid ); ! 5575: z.low = floatx80_default_nan_low; ! 5576: z.high = floatx80_default_nan_high; ! 5577: return z; ! 5578: } ! 5579: if ( aExp == 0 ) { ! 5580: if ( aSig0 == 0 ) return packFloatx80( 0, 0, 0 ); ! 5581: normalizeFloatx80Subnormal( aSig0, &aExp, &aSig0 ); ! 5582: } ! 5583: zExp = ( ( aExp - 0x3FFF )>>1 ) + 0x3FFF; ! 5584: zSig0 = estimateSqrt32( aExp, aSig0>>32 ); ! 5585: shift128Right( aSig0, 0, 2 + ( aExp & 1 ), &aSig0, &aSig1 ); ! 5586: zSig0 = estimateDiv128To64( aSig0, aSig1, zSig0<<32 ) + ( zSig0<<30 ); ! 5587: doubleZSig0 = zSig0<<1; ! 5588: mul64To128( zSig0, zSig0, &term0, &term1 ); ! 5589: sub128( aSig0, aSig1, term0, term1, &rem0, &rem1 ); ! 5590: while ( (sbits64) rem0 < 0 ) { ! 5591: --zSig0; ! 5592: doubleZSig0 -= 2; ! 5593: add128( rem0, rem1, zSig0>>63, doubleZSig0 | 1, &rem0, &rem1 ); ! 5594: } ! 5595: zSig1 = estimateDiv128To64( rem1, 0, doubleZSig0 ); ! 5596: if ( ( zSig1 & LIT64( 0x3FFFFFFFFFFFFFFF ) ) <= 5 ) { ! 5597: if ( zSig1 == 0 ) zSig1 = 1; ! 5598: mul64To128( doubleZSig0, zSig1, &term1, &term2 ); ! 5599: sub128( rem1, 0, term1, term2, &rem1, &rem2 ); ! 5600: mul64To128( zSig1, zSig1, &term2, &term3 ); ! 5601: sub192( rem1, rem2, 0, 0, term2, term3, &rem1, &rem2, &rem3 ); ! 5602: while ( (sbits64) rem1 < 0 ) { ! 5603: --zSig1; ! 5604: shortShift128Left( 0, zSig1, 1, &term2, &term3 ); ! 5605: term3 |= 1; ! 5606: term2 |= doubleZSig0; ! 5607: add192( rem1, rem2, rem3, 0, term2, term3, &rem1, &rem2, &rem3 ); ! 5608: } ! 5609: zSig1 |= ( ( rem1 | rem2 | rem3 ) != 0 ); ! 5610: } ! 5611: shortShift128Left( 0, zSig1, 1, &zSig0, &zSig1 ); ! 5612: zSig0 |= doubleZSig0; ! 5613: return ! 5614: roundAndPackFloatx80( ! 5615: floatx80_rounding_precision, 0, zExp, zSig0, zSig1 ); ! 5616: ! 5617: } ! 5618: ! 5619: #ifdef SOFTFLOAT_68K // 07-01-2017: Added for Previous ! 5620: /*---------------------------------------------------------------------------- ! 5621: | Returns the mantissa of the extended double-precision floating-point ! 5622: | value `a'. ! 5623: *----------------------------------------------------------------------------*/ ! 5624: ! 5625: floatx80 floatx80_getman( floatx80 a ) ! 5626: { ! 5627: flag aSign; ! 5628: int32 aExp; ! 5629: bits64 aSig; ! 5630: ! 5631: aSig = extractFloatx80Frac( a ); ! 5632: aExp = extractFloatx80Exp( a ); ! 5633: aSign = extractFloatx80Sign( a ); ! 5634: ! 5635: if ( aExp == 0x7FFF ) { ! 5636: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 5637: float_raise( float_flag_invalid ); ! 5638: a.low = floatx80_default_nan_low; ! 5639: a.high = floatx80_default_nan_high; ! 5640: return a; ! 5641: } ! 5642: ! 5643: if ( aExp == 0 ) { ! 5644: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 5645: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5646: } ! 5647: ! 5648: return roundAndPackFloatx80(floatx80_rounding_precision, aSign, 0x3FFF, aSig, 0); ! 5649: } ! 5650: ! 5651: /*---------------------------------------------------------------------------- ! 5652: | Returns the exponent of the extended double-precision floating-point ! 5653: | value `a' as an extended double-precision value. ! 5654: *----------------------------------------------------------------------------*/ ! 5655: ! 5656: floatx80 floatx80_getexp( floatx80 a ) ! 5657: { ! 5658: flag aSign; ! 5659: int32 aExp; ! 5660: bits64 aSig; ! 5661: ! 5662: aSig = extractFloatx80Frac( a ); ! 5663: aExp = extractFloatx80Exp( a ); ! 5664: aSign = extractFloatx80Sign( a ); ! 5665: ! 5666: if ( aExp == 0x7FFF ) { ! 5667: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 5668: float_raise( float_flag_invalid ); ! 5669: a.low = floatx80_default_nan_low; ! 5670: a.high = floatx80_default_nan_high; ! 5671: return a; ! 5672: } ! 5673: ! 5674: if ( aExp == 0 ) { ! 5675: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 5676: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5677: } ! 5678: ! 5679: return int32_to_floatx80(aExp - 0x3FFF); ! 5680: } ! 5681: ! 5682: /*---------------------------------------------------------------------------- ! 5683: | Scales extended double-precision floating-point value in operand `a' by ! 5684: | value `b'. The function truncates the value in the second operand 'b' to ! 5685: | an integral value and adds that value to the exponent of the operand 'a'. ! 5686: | The operation performed according to the IEC/IEEE Standard for Binary ! 5687: | Floating-Point Arithmetic. ! 5688: *----------------------------------------------------------------------------*/ ! 5689: ! 5690: floatx80 floatx80_scale(floatx80 a, floatx80 b) ! 5691: { ! 5692: flag aSign, bSign; ! 5693: int32 aExp, bExp, shiftCount; ! 5694: bits64 aSig, bSig; ! 5695: ! 5696: aSig = extractFloatx80Frac(a); ! 5697: aExp = extractFloatx80Exp(a); ! 5698: aSign = extractFloatx80Sign(a); ! 5699: bSig = extractFloatx80Frac(b); ! 5700: bExp = extractFloatx80Exp(b); ! 5701: bSign = extractFloatx80Sign(b); ! 5702: ! 5703: if ( bExp == 0x7FFF ) { ! 5704: if ( (bits64) ( bSig<<1 ) || ! 5705: ( ( aExp == 0x7FFF ) && (bits64) ( aSig<<1 ) ) ) { ! 5706: return propagateFloatx80NaN( a, b ); ! 5707: } ! 5708: float_raise( float_flag_invalid ); ! 5709: a.low = floatx80_default_nan_low; ! 5710: a.high = floatx80_default_nan_high; ! 5711: return a; ! 5712: } ! 5713: if ( aExp == 0x7FFF ) { ! 5714: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaN( a, b ); ! 5715: return packFloatx80( aSign, 0x7FFF, floatx80_default_infinity_low ); ! 5716: } ! 5717: if ( aExp == 0 ) { ! 5718: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0); ! 5719: if ( bExp < 0x3FFF ) return a; ! 5720: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5721: } ! 5722: ! 5723: if ( bExp < 0x3FFF ) return a; ! 5724: ! 5725: if ( 0x400F < bExp ) { ! 5726: aExp = bSign ? -0x6001 : 0xE000; ! 5727: return roundAndPackFloatx80( ! 5728: floatx80_rounding_precision, aSign, aExp, aSig, 0 ); ! 5729: } ! 5730: ! 5731: shiftCount = 0x403E - bExp; ! 5732: bSig >>= shiftCount; ! 5733: aExp = bSign ? ( aExp - bSig ) : ( aExp + bSig ); ! 5734: ! 5735: return roundAndPackFloatx80( ! 5736: floatx80_rounding_precision, aSign, aExp, aSig, 0); ! 5737: ! 5738: } ! 5739: ! 5740: /*----------------------------------------------------------------------------- ! 5741: | Calculates the absolute value of the extended double-precision floating-point ! 5742: | value `a'. The operation is performed according to the IEC/IEEE Standard ! 5743: | for Binary Floating-Point Arithmetic. ! 5744: *----------------------------------------------------------------------------*/ ! 5745: ! 5746: floatx80 floatx80_abs(floatx80 a) ! 5747: { ! 5748: int32 aExp; ! 5749: bits64 aSig; ! 5750: ! 5751: aSig = extractFloatx80Frac(a); ! 5752: aExp = extractFloatx80Exp(a); ! 5753: ! 5754: if ( aExp == 0x7FFF && (bits64) ( aSig<<1 ) ) { ! 5755: return propagateFloatx80NaNOneArg( a ); ! 5756: } ! 5757: ! 5758: if ( aExp == 0 ) { ! 5759: if ( aSig == 0 ) return packFloatx80( 0, 0, 0 ); ! 5760: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5761: } ! 5762: ! 5763: return roundAndPackFloatx80( ! 5764: floatx80_rounding_precision, 0, aExp, aSig, 0 ); ! 5765: ! 5766: } ! 5767: ! 5768: /*----------------------------------------------------------------------------- ! 5769: | Changes the sign of the extended double-precision floating-point value 'a'. ! 5770: | The operation is performed according to the IEC/IEEE Standard for Binary ! 5771: | Floating-Point Arithmetic. ! 5772: *----------------------------------------------------------------------------*/ ! 5773: ! 5774: floatx80 floatx80_neg(floatx80 a) ! 5775: { ! 5776: flag aSign; ! 5777: int32 aExp; ! 5778: bits64 aSig; ! 5779: ! 5780: aSig = extractFloatx80Frac(a); ! 5781: aExp = extractFloatx80Exp(a); ! 5782: aSign = extractFloatx80Sign(a); ! 5783: ! 5784: if ( aExp == 0x7FFF && (bits64) ( aSig<<1 ) ) { ! 5785: return propagateFloatx80NaNOneArg( a ); ! 5786: } ! 5787: ! 5788: aSign = !aSign; ! 5789: ! 5790: if ( aExp == 0 ) { ! 5791: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 5792: normalizeFloatx80Subnormal( aSig, &aExp, &aSig ); ! 5793: } ! 5794: ! 5795: return roundAndPackFloatx80( ! 5796: floatx80_rounding_precision, aSign, aExp, aSig, 0 ); ! 5797: ! 5798: } ! 5799: ! 5800: /*---------------------------------------------------------------------------- ! 5801: | Returns the result of comparing the extended double-precision floating- ! 5802: | point values `a' and `b'. The result is abstracted for matching the ! 5803: | corresponding condition codes. ! 5804: *----------------------------------------------------------------------------*/ ! 5805: ! 5806: floatx80 floatx80_cmp( floatx80 a, floatx80 b ) ! 5807: { ! 5808: flag aSign, bSign; ! 5809: int32 aExp, bExp; ! 5810: bits64 aSig, bSig; ! 5811: ! 5812: aSig = extractFloatx80Frac( a ); ! 5813: aExp = extractFloatx80Exp( a ); ! 5814: aSign = extractFloatx80Sign( a ); ! 5815: bSig = extractFloatx80Frac( b ); ! 5816: bExp = extractFloatx80Exp( b ); ! 5817: bSign = extractFloatx80Sign( b ); ! 5818: ! 5819: if ( ( aExp == 0x7FFF && (bits64) ( aSig<<1 ) ) || ! 5820: ( bExp == 0x7FFF && (bits64) ( bSig<<1 ) ) ) { ! 5821: return propagateFloatx80NaN( packFloatx80( 0, aExp, aSig ), ! 5822: packFloatx80( 0, bExp, bSig ) ); ! 5823: } ! 5824: ! 5825: if ( bExp < aExp ) return packFloatx80( aSign, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5826: if ( aExp < bExp ) return packFloatx80( bSign ^ 1, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5827: ! 5828: if ( aExp == 0x7FFF ) { ! 5829: if ( aSign == bSign ) return packFloatx80( aSign, 0, 0 ); ! 5830: return packFloatx80( aSign, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5831: } ! 5832: ! 5833: if ( bSig < aSig ) return packFloatx80( aSign, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5834: if ( aSig < bSig ) return packFloatx80( bSign ^ 1, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5835: ! 5836: if ( aSig == 0 ) return packFloatx80( aSign, 0, 0 ); ! 5837: ! 5838: if ( aSign == bSign ) return packFloatx80( 0, 0, 0 ); ! 5839: ! 5840: return packFloatx80( aSign, 0x3FFF, LIT64( 0x8000000000000000 ) ); ! 5841: ! 5842: } ! 5843: ! 5844: floatx80 floatx80_tst( floatx80 a ) ! 5845: { ! 5846: int32 aExp; ! 5847: bits64 aSig; ! 5848: ! 5849: aSig = extractFloatx80Frac( a ); ! 5850: aExp = extractFloatx80Exp( a ); ! 5851: ! 5852: if ( aExp == 0x7FFF && (bits64) ( aSig<<1 ) ) ! 5853: return propagateFloatx80NaNOneArg( a ); ! 5854: ! 5855: return a; ! 5856: } ! 5857: ! 5858: floatx80 floatx80_move( floatx80 a ) ! 5859: { ! 5860: flag aSign; ! 5861: int32 aExp; ! 5862: bits64 aSig; ! 5863: ! 5864: aSig = extractFloatx80Frac( a ); ! 5865: aExp = extractFloatx80Exp( a ); ! 5866: aSign = extractFloatx80Sign( a ); ! 5867: ! 5868: if ( aExp == 0x7FFF ) { ! 5869: if ( (bits64) ( aSig<<1 ) ) return propagateFloatx80NaNOneArg( a ); ! 5870: return a; ! 5871: } ! 5872: if ( aExp == 0 ) { ! 5873: if ( aSig == 0 ) return a; ! 5874: normalizeRoundAndPackFloatx80( floatx80_rounding_precision, aSign, aExp, aSig, 0 ); ! 5875: } ! 5876: return roundAndPackFloatx80( floatx80_rounding_precision, aSign, aExp, aSig, 0 ); ! 5877: ! 5878: } ! 5879: ! 5880: #endif // End of addition for Previous ! 5881: ! 5882: /*---------------------------------------------------------------------------- ! 5883: | Returns 1 if the extended double-precision floating-point value `a' is ! 5884: | equal to the corresponding value `b', and 0 otherwise. The comparison is ! 5885: | performed according to the IEC/IEEE Standard for Binary Floating-Point ! 5886: | Arithmetic. ! 5887: *----------------------------------------------------------------------------*/ ! 5888: ! 5889: flag floatx80_eq( floatx80 a, floatx80 b ) ! 5890: { ! 5891: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 5892: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 5893: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 5894: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 5895: ) { ! 5896: if ( floatx80_is_signaling_nan( a ) ! 5897: || floatx80_is_signaling_nan( b ) ) { ! 5898: float_raise( float_flag_invalid ); ! 5899: } ! 5900: return 0; ! 5901: } ! 5902: return ! 5903: ( a.low == b.low ) ! 5904: && ( ( a.high == b.high ) ! 5905: || ( ( a.low == 0 ) ! 5906: && ( (bits16) ( ( a.high | b.high )<<1 ) == 0 ) ) ! 5907: ); ! 5908: ! 5909: } ! 5910: ! 5911: /*---------------------------------------------------------------------------- ! 5912: | Returns 1 if the extended double-precision floating-point value `a' is ! 5913: | less than or equal to the corresponding value `b', and 0 otherwise. The ! 5914: | comparison is performed according to the IEC/IEEE Standard for Binary ! 5915: | Floating-Point Arithmetic. ! 5916: *----------------------------------------------------------------------------*/ ! 5917: ! 5918: flag floatx80_le( floatx80 a, floatx80 b ) ! 5919: { ! 5920: flag aSign, bSign; ! 5921: ! 5922: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 5923: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 5924: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 5925: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 5926: ) { ! 5927: float_raise( float_flag_invalid ); ! 5928: return 0; ! 5929: } ! 5930: aSign = extractFloatx80Sign( a ); ! 5931: bSign = extractFloatx80Sign( b ); ! 5932: if ( aSign != bSign ) { ! 5933: return ! 5934: aSign ! 5935: || ( ( ( (bits16) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 5936: == 0 ); ! 5937: } ! 5938: return ! 5939: aSign ? le128( b.high, b.low, a.high, a.low ) ! 5940: : le128( a.high, a.low, b.high, b.low ); ! 5941: ! 5942: } ! 5943: ! 5944: /*---------------------------------------------------------------------------- ! 5945: | Returns 1 if the extended double-precision floating-point value `a' is ! 5946: | less than the corresponding value `b', and 0 otherwise. The comparison ! 5947: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 5948: | Arithmetic. ! 5949: *----------------------------------------------------------------------------*/ ! 5950: ! 5951: flag floatx80_lt( floatx80 a, floatx80 b ) ! 5952: { ! 5953: flag aSign, bSign; ! 5954: ! 5955: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 5956: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 5957: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 5958: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 5959: ) { ! 5960: float_raise( float_flag_invalid ); ! 5961: return 0; ! 5962: } ! 5963: aSign = extractFloatx80Sign( a ); ! 5964: bSign = extractFloatx80Sign( b ); ! 5965: if ( aSign != bSign ) { ! 5966: return ! 5967: aSign ! 5968: && ( ( ( (bits16) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 5969: != 0 ); ! 5970: } ! 5971: return ! 5972: aSign ? lt128( b.high, b.low, a.high, a.low ) ! 5973: : lt128( a.high, a.low, b.high, b.low ); ! 5974: ! 5975: } ! 5976: ! 5977: /*---------------------------------------------------------------------------- ! 5978: | Returns 1 if the extended double-precision floating-point value `a' is equal ! 5979: | to the corresponding value `b', and 0 otherwise. The invalid exception is ! 5980: | raised if either operand is a NaN. Otherwise, the comparison is performed ! 5981: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 5982: *----------------------------------------------------------------------------*/ ! 5983: ! 5984: flag floatx80_eq_signaling( floatx80 a, floatx80 b ) ! 5985: { ! 5986: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 5987: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 5988: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 5989: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 5990: ) { ! 5991: float_raise( float_flag_invalid ); ! 5992: return 0; ! 5993: } ! 5994: return ! 5995: ( a.low == b.low ) ! 5996: && ( ( a.high == b.high ) ! 5997: || ( ( a.low == 0 ) ! 5998: && ( (bits16) ( ( a.high | b.high )<<1 ) == 0 ) ) ! 5999: ); ! 6000: ! 6001: } ! 6002: ! 6003: /*---------------------------------------------------------------------------- ! 6004: | Returns 1 if the extended double-precision floating-point value `a' is less ! 6005: | than or equal to the corresponding value `b', and 0 otherwise. Quiet NaNs ! 6006: | do not cause an exception. Otherwise, the comparison is performed according ! 6007: | to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 6008: *----------------------------------------------------------------------------*/ ! 6009: ! 6010: flag floatx80_le_quiet( floatx80 a, floatx80 b ) ! 6011: { ! 6012: flag aSign, bSign; ! 6013: ! 6014: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 6015: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 6016: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 6017: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 6018: ) { ! 6019: if ( floatx80_is_signaling_nan( a ) ! 6020: || floatx80_is_signaling_nan( b ) ) { ! 6021: float_raise( float_flag_invalid ); ! 6022: } ! 6023: return 0; ! 6024: } ! 6025: aSign = extractFloatx80Sign( a ); ! 6026: bSign = extractFloatx80Sign( b ); ! 6027: if ( aSign != bSign ) { ! 6028: return ! 6029: aSign ! 6030: || ( ( ( (bits16) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 6031: == 0 ); ! 6032: } ! 6033: return ! 6034: aSign ? le128( b.high, b.low, a.high, a.low ) ! 6035: : le128( a.high, a.low, b.high, b.low ); ! 6036: ! 6037: } ! 6038: ! 6039: /*---------------------------------------------------------------------------- ! 6040: | Returns 1 if the extended double-precision floating-point value `a' is less ! 6041: | than the corresponding value `b', and 0 otherwise. Quiet NaNs do not cause ! 6042: | an exception. Otherwise, the comparison is performed according to the ! 6043: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 6044: *----------------------------------------------------------------------------*/ ! 6045: ! 6046: flag floatx80_lt_quiet( floatx80 a, floatx80 b ) ! 6047: { ! 6048: flag aSign, bSign; ! 6049: ! 6050: if ( ( ( extractFloatx80Exp( a ) == 0x7FFF ) ! 6051: && (bits64) ( extractFloatx80Frac( a )<<1 ) ) ! 6052: || ( ( extractFloatx80Exp( b ) == 0x7FFF ) ! 6053: && (bits64) ( extractFloatx80Frac( b )<<1 ) ) ! 6054: ) { ! 6055: if ( floatx80_is_signaling_nan( a ) ! 6056: || floatx80_is_signaling_nan( b ) ) { ! 6057: float_raise( float_flag_invalid ); ! 6058: } ! 6059: return 0; ! 6060: } ! 6061: aSign = extractFloatx80Sign( a ); ! 6062: bSign = extractFloatx80Sign( b ); ! 6063: if ( aSign != bSign ) { ! 6064: return ! 6065: aSign ! 6066: && ( ( ( (bits16) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 6067: != 0 ); ! 6068: } ! 6069: return ! 6070: aSign ? lt128( b.high, b.low, a.high, a.low ) ! 6071: : lt128( a.high, a.low, b.high, b.low ); ! 6072: ! 6073: } ! 6074: ! 6075: #endif ! 6076: ! 6077: #ifdef FLOAT128 ! 6078: ! 6079: /*---------------------------------------------------------------------------- ! 6080: | Returns the result of converting the quadruple-precision floating-point ! 6081: | value `a' to the 32-bit two's complement integer format. The conversion ! 6082: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6083: | Arithmetic---which means in particular that the conversion is rounded ! 6084: | according to the current rounding mode. If `a' is a NaN, the largest ! 6085: | positive integer is returned. Otherwise, if the conversion overflows, the ! 6086: | largest integer with the same sign as `a' is returned. ! 6087: *----------------------------------------------------------------------------*/ ! 6088: ! 6089: int32 float128_to_int32( float128 a ) ! 6090: { ! 6091: flag aSign; ! 6092: int32 aExp, shiftCount; ! 6093: bits64 aSig0, aSig1; ! 6094: ! 6095: aSig1 = extractFloat128Frac1( a ); ! 6096: aSig0 = extractFloat128Frac0( a ); ! 6097: aExp = extractFloat128Exp( a ); ! 6098: aSign = extractFloat128Sign( a ); ! 6099: if ( ( aExp == 0x7FFF ) && ( aSig0 | aSig1 ) ) aSign = 0; ! 6100: if ( aExp ) aSig0 |= LIT64( 0x0001000000000000 ); ! 6101: aSig0 |= ( aSig1 != 0 ); ! 6102: shiftCount = 0x4028 - aExp; ! 6103: if ( 0 < shiftCount ) shift64RightJamming( aSig0, shiftCount, &aSig0 ); ! 6104: return roundAndPackInt32( aSign, aSig0 ); ! 6105: ! 6106: } ! 6107: ! 6108: /*---------------------------------------------------------------------------- ! 6109: | Returns the result of converting the quadruple-precision floating-point ! 6110: | value `a' to the 32-bit two's complement integer format. The conversion ! 6111: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6112: | Arithmetic, except that the conversion is always rounded toward zero. If ! 6113: | `a' is a NaN, the largest positive integer is returned. Otherwise, if the ! 6114: | conversion overflows, the largest integer with the same sign as `a' is ! 6115: | returned. ! 6116: *----------------------------------------------------------------------------*/ ! 6117: ! 6118: int32 float128_to_int32_round_to_zero( float128 a ) ! 6119: { ! 6120: flag aSign; ! 6121: int32 aExp, shiftCount; ! 6122: bits64 aSig0, aSig1, savedASig; ! 6123: int32 z; ! 6124: ! 6125: aSig1 = extractFloat128Frac1( a ); ! 6126: aSig0 = extractFloat128Frac0( a ); ! 6127: aExp = extractFloat128Exp( a ); ! 6128: aSign = extractFloat128Sign( a ); ! 6129: aSig0 |= ( aSig1 != 0 ); ! 6130: if ( 0x401E < aExp ) { ! 6131: if ( ( aExp == 0x7FFF ) && aSig0 ) aSign = 0; ! 6132: goto invalid; ! 6133: } ! 6134: else if ( aExp < 0x3FFF ) { ! 6135: if ( aExp || aSig0 ) float_exception_flags |= float_flag_inexact; ! 6136: return 0; ! 6137: } ! 6138: aSig0 |= LIT64( 0x0001000000000000 ); ! 6139: shiftCount = 0x402F - aExp; ! 6140: savedASig = aSig0; ! 6141: aSig0 >>= shiftCount; ! 6142: z = aSig0; ! 6143: if ( aSign ) z = - z; ! 6144: z = (sbits32) z; ! 6145: if ( ( z < 0 ) ^ aSign ) { ! 6146: invalid: ! 6147: float_raise( float_flag_invalid ); ! 6148: return aSign ? (sbits32) 0x80000000 : 0x7FFFFFFF; ! 6149: } ! 6150: if ( ( aSig0<<shiftCount ) != savedASig ) { ! 6151: float_exception_flags |= float_flag_inexact; ! 6152: } ! 6153: return z; ! 6154: ! 6155: } ! 6156: ! 6157: /*---------------------------------------------------------------------------- ! 6158: | Returns the result of converting the quadruple-precision floating-point ! 6159: | value `a' to the 64-bit two's complement integer format. The conversion ! 6160: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6161: | Arithmetic---which means in particular that the conversion is rounded ! 6162: | according to the current rounding mode. If `a' is a NaN, the largest ! 6163: | positive integer is returned. Otherwise, if the conversion overflows, the ! 6164: | largest integer with the same sign as `a' is returned. ! 6165: *----------------------------------------------------------------------------*/ ! 6166: ! 6167: int64 float128_to_int64( float128 a ) ! 6168: { ! 6169: flag aSign; ! 6170: int32 aExp, shiftCount; ! 6171: bits64 aSig0, aSig1; ! 6172: ! 6173: aSig1 = extractFloat128Frac1( a ); ! 6174: aSig0 = extractFloat128Frac0( a ); ! 6175: aExp = extractFloat128Exp( a ); ! 6176: aSign = extractFloat128Sign( a ); ! 6177: if ( aExp ) aSig0 |= LIT64( 0x0001000000000000 ); ! 6178: shiftCount = 0x402F - aExp; ! 6179: if ( shiftCount <= 0 ) { ! 6180: if ( 0x403E < aExp ) { ! 6181: float_raise( float_flag_invalid ); ! 6182: if ( ! aSign ! 6183: || ( ( aExp == 0x7FFF ) ! 6184: && ( aSig1 || ( aSig0 != LIT64( 0x0001000000000000 ) ) ) ! 6185: ) ! 6186: ) { ! 6187: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 6188: } ! 6189: return (sbits64) LIT64( 0x8000000000000000 ); ! 6190: } ! 6191: shortShift128Left( aSig0, aSig1, - shiftCount, &aSig0, &aSig1 ); ! 6192: } ! 6193: else { ! 6194: shift64ExtraRightJamming( aSig0, aSig1, shiftCount, &aSig0, &aSig1 ); ! 6195: } ! 6196: return roundAndPackInt64( aSign, aSig0, aSig1 ); ! 6197: ! 6198: } ! 6199: ! 6200: /*---------------------------------------------------------------------------- ! 6201: | Returns the result of converting the quadruple-precision floating-point ! 6202: | value `a' to the 64-bit two's complement integer format. The conversion ! 6203: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6204: | Arithmetic, except that the conversion is always rounded toward zero. ! 6205: | If `a' is a NaN, the largest positive integer is returned. Otherwise, if ! 6206: | the conversion overflows, the largest integer with the same sign as `a' is ! 6207: | returned. ! 6208: *----------------------------------------------------------------------------*/ ! 6209: ! 6210: int64 float128_to_int64_round_to_zero( float128 a ) ! 6211: { ! 6212: flag aSign; ! 6213: int32 aExp, shiftCount; ! 6214: bits64 aSig0, aSig1; ! 6215: int64 z; ! 6216: ! 6217: aSig1 = extractFloat128Frac1( a ); ! 6218: aSig0 = extractFloat128Frac0( a ); ! 6219: aExp = extractFloat128Exp( a ); ! 6220: aSign = extractFloat128Sign( a ); ! 6221: if ( aExp ) aSig0 |= LIT64( 0x0001000000000000 ); ! 6222: shiftCount = aExp - 0x402F; ! 6223: if ( 0 < shiftCount ) { ! 6224: if ( 0x403E <= aExp ) { ! 6225: aSig0 &= LIT64( 0x0000FFFFFFFFFFFF ); ! 6226: if ( ( a.high == LIT64( 0xC03E000000000000 ) ) ! 6227: && ( aSig1 < LIT64( 0x0002000000000000 ) ) ) { ! 6228: if ( aSig1 ) float_exception_flags |= float_flag_inexact; ! 6229: } ! 6230: else { ! 6231: float_raise( float_flag_invalid ); ! 6232: if ( ! aSign || ( ( aExp == 0x7FFF ) && ( aSig0 | aSig1 ) ) ) { ! 6233: return LIT64( 0x7FFFFFFFFFFFFFFF ); ! 6234: } ! 6235: } ! 6236: return (sbits64) LIT64( 0x8000000000000000 ); ! 6237: } ! 6238: z = ( aSig0<<shiftCount ) | ( aSig1>>( ( - shiftCount ) & 63 ) ); ! 6239: if ( (bits64) ( aSig1<<shiftCount ) ) { ! 6240: float_exception_flags |= float_flag_inexact; ! 6241: } ! 6242: } ! 6243: else { ! 6244: if ( aExp < 0x3FFF ) { ! 6245: if ( aExp | aSig0 | aSig1 ) { ! 6246: float_exception_flags |= float_flag_inexact; ! 6247: } ! 6248: return 0; ! 6249: } ! 6250: z = aSig0>>( - shiftCount ); ! 6251: if ( aSig1 ! 6252: || ( shiftCount && (bits64) ( aSig0<<( shiftCount & 63 ) ) ) ) { ! 6253: float_exception_flags |= float_flag_inexact; ! 6254: } ! 6255: } ! 6256: if ( aSign ) z = - z; ! 6257: return z; ! 6258: ! 6259: } ! 6260: ! 6261: /*---------------------------------------------------------------------------- ! 6262: | Returns the result of converting the quadruple-precision floating-point ! 6263: | value `a' to the single-precision floating-point format. The conversion ! 6264: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6265: | Arithmetic. ! 6266: *----------------------------------------------------------------------------*/ ! 6267: ! 6268: float32 float128_to_float32( float128 a ) ! 6269: { ! 6270: flag aSign; ! 6271: int32 aExp; ! 6272: bits64 aSig0, aSig1; ! 6273: bits32 zSig; ! 6274: ! 6275: aSig1 = extractFloat128Frac1( a ); ! 6276: aSig0 = extractFloat128Frac0( a ); ! 6277: aExp = extractFloat128Exp( a ); ! 6278: aSign = extractFloat128Sign( a ); ! 6279: if ( aExp == 0x7FFF ) { ! 6280: if ( aSig0 | aSig1 ) { ! 6281: return commonNaNToFloat32( float128ToCommonNaN( a ) ); ! 6282: } ! 6283: return packFloat32( aSign, 0xFF, 0 ); ! 6284: } ! 6285: aSig0 |= ( aSig1 != 0 ); ! 6286: shift64RightJamming( aSig0, 18, &aSig0 ); ! 6287: zSig = aSig0; ! 6288: if ( aExp || zSig ) { ! 6289: zSig |= 0x40000000; ! 6290: aExp -= 0x3F81; ! 6291: } ! 6292: return roundAndPackFloat32( aSign, aExp, zSig ); ! 6293: ! 6294: } ! 6295: ! 6296: /*---------------------------------------------------------------------------- ! 6297: | Returns the result of converting the quadruple-precision floating-point ! 6298: | value `a' to the double-precision floating-point format. The conversion ! 6299: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 6300: | Arithmetic. ! 6301: *----------------------------------------------------------------------------*/ ! 6302: ! 6303: float64 float128_to_float64( float128 a ) ! 6304: { ! 6305: flag aSign; ! 6306: int32 aExp; ! 6307: bits64 aSig0, aSig1; ! 6308: ! 6309: aSig1 = extractFloat128Frac1( a ); ! 6310: aSig0 = extractFloat128Frac0( a ); ! 6311: aExp = extractFloat128Exp( a ); ! 6312: aSign = extractFloat128Sign( a ); ! 6313: if ( aExp == 0x7FFF ) { ! 6314: if ( aSig0 | aSig1 ) { ! 6315: return commonNaNToFloat64( float128ToCommonNaN( a ) ); ! 6316: } ! 6317: return packFloat64( aSign, 0x7FF, 0 ); ! 6318: } ! 6319: shortShift128Left( aSig0, aSig1, 14, &aSig0, &aSig1 ); ! 6320: aSig0 |= ( aSig1 != 0 ); ! 6321: if ( aExp || aSig0 ) { ! 6322: aSig0 |= LIT64( 0x4000000000000000 ); ! 6323: aExp -= 0x3C01; ! 6324: } ! 6325: return roundAndPackFloat64( aSign, aExp, aSig0 ); ! 6326: ! 6327: } ! 6328: ! 6329: #ifdef FLOATX80 ! 6330: ! 6331: /*---------------------------------------------------------------------------- ! 6332: | Returns the result of converting the quadruple-precision floating-point ! 6333: | value `a' to the extended double-precision floating-point format. The ! 6334: | conversion is performed according to the IEC/IEEE Standard for Binary ! 6335: | Floating-Point Arithmetic. ! 6336: *----------------------------------------------------------------------------*/ ! 6337: ! 6338: floatx80 float128_to_floatx80( float128 a ) ! 6339: { ! 6340: flag aSign; ! 6341: int32 aExp; ! 6342: bits64 aSig0, aSig1; ! 6343: ! 6344: aSig1 = extractFloat128Frac1( a ); ! 6345: aSig0 = extractFloat128Frac0( a ); ! 6346: aExp = extractFloat128Exp( a ); ! 6347: aSign = extractFloat128Sign( a ); ! 6348: if ( aExp == 0x7FFF ) { ! 6349: if ( aSig0 | aSig1 ) { ! 6350: return commonNaNToFloatx80( float128ToCommonNaN( a ) ); ! 6351: } ! 6352: return packFloatx80( aSign, 0x7FFF, floatx80_default_infinity_low ); ! 6353: } ! 6354: if ( aExp == 0 ) { ! 6355: if ( ( aSig0 | aSig1 ) == 0 ) return packFloatx80( aSign, 0, 0 ); ! 6356: normalizeFloat128Subnormal( aSig0, aSig1, &aExp, &aSig0, &aSig1 ); ! 6357: } ! 6358: else { ! 6359: aSig0 |= LIT64( 0x0001000000000000 ); ! 6360: } ! 6361: shortShift128Left( aSig0, aSig1, 15, &aSig0, &aSig1 ); ! 6362: return roundAndPackFloatx80( 80, aSign, aExp, aSig0, aSig1 ); ! 6363: ! 6364: } ! 6365: ! 6366: #endif ! 6367: ! 6368: /*---------------------------------------------------------------------------- ! 6369: | Rounds the quadruple-precision floating-point value `a' to an integer, and ! 6370: | returns the result as a quadruple-precision floating-point value. The ! 6371: | operation is performed according to the IEC/IEEE Standard for Binary ! 6372: | Floating-Point Arithmetic. ! 6373: *----------------------------------------------------------------------------*/ ! 6374: ! 6375: float128 float128_round_to_int( float128 a ) ! 6376: { ! 6377: flag aSign; ! 6378: int32 aExp; ! 6379: bits64 lastBitMask, roundBitsMask; ! 6380: int8 roundingMode; ! 6381: float128 z; ! 6382: ! 6383: aExp = extractFloat128Exp( a ); ! 6384: if ( 0x402F <= aExp ) { ! 6385: if ( 0x406F <= aExp ) { ! 6386: if ( ( aExp == 0x7FFF ) ! 6387: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ! 6388: ) { ! 6389: return propagateFloat128NaN( a, a ); ! 6390: } ! 6391: return a; ! 6392: } ! 6393: lastBitMask = 1; ! 6394: lastBitMask = ( lastBitMask<<( 0x406E - aExp ) )<<1; ! 6395: roundBitsMask = lastBitMask - 1; ! 6396: z = a; ! 6397: roundingMode = float_rounding_mode; ! 6398: if ( roundingMode == float_round_nearest_even ) { ! 6399: if ( lastBitMask ) { ! 6400: add128( z.high, z.low, 0, lastBitMask>>1, &z.high, &z.low ); ! 6401: if ( ( z.low & roundBitsMask ) == 0 ) z.low &= ~ lastBitMask; ! 6402: } ! 6403: else { ! 6404: if ( (sbits64) z.low < 0 ) { ! 6405: ++z.high; ! 6406: if ( (bits64) ( z.low<<1 ) == 0 ) z.high &= ~1; ! 6407: } ! 6408: } ! 6409: } ! 6410: else if ( roundingMode != float_round_to_zero ) { ! 6411: if ( extractFloat128Sign( z ) ! 6412: ^ ( roundingMode == float_round_up ) ) { ! 6413: add128( z.high, z.low, 0, roundBitsMask, &z.high, &z.low ); ! 6414: } ! 6415: } ! 6416: z.low &= ~ roundBitsMask; ! 6417: } ! 6418: else { ! 6419: if ( aExp < 0x3FFF ) { ! 6420: if ( ( ( (bits64) ( a.high<<1 ) ) | a.low ) == 0 ) return a; ! 6421: float_exception_flags |= float_flag_inexact; ! 6422: aSign = extractFloat128Sign( a ); ! 6423: switch ( float_rounding_mode ) { ! 6424: case float_round_nearest_even: ! 6425: if ( ( aExp == 0x3FFE ) ! 6426: && ( extractFloat128Frac0( a ) ! 6427: | extractFloat128Frac1( a ) ) ! 6428: ) { ! 6429: return packFloat128( aSign, 0x3FFF, 0, 0 ); ! 6430: } ! 6431: break; ! 6432: case float_round_down: ! 6433: return ! 6434: aSign ? packFloat128( 1, 0x3FFF, 0, 0 ) ! 6435: : packFloat128( 0, 0, 0, 0 ); ! 6436: case float_round_up: ! 6437: return ! 6438: aSign ? packFloat128( 1, 0, 0, 0 ) ! 6439: : packFloat128( 0, 0x3FFF, 0, 0 ); ! 6440: } ! 6441: return packFloat128( aSign, 0, 0, 0 ); ! 6442: } ! 6443: lastBitMask = 1; ! 6444: lastBitMask <<= 0x402F - aExp; ! 6445: roundBitsMask = lastBitMask - 1; ! 6446: z.low = 0; ! 6447: z.high = a.high; ! 6448: roundingMode = float_rounding_mode; ! 6449: if ( roundingMode == float_round_nearest_even ) { ! 6450: z.high += lastBitMask>>1; ! 6451: if ( ( ( z.high & roundBitsMask ) | a.low ) == 0 ) { ! 6452: z.high &= ~ lastBitMask; ! 6453: } ! 6454: } ! 6455: else if ( roundingMode != float_round_to_zero ) { ! 6456: if ( extractFloat128Sign( z ) ! 6457: ^ ( roundingMode == float_round_up ) ) { ! 6458: z.high |= ( a.low != 0 ); ! 6459: z.high += roundBitsMask; ! 6460: } ! 6461: } ! 6462: z.high &= ~ roundBitsMask; ! 6463: } ! 6464: if ( ( z.low != a.low ) || ( z.high != a.high ) ) { ! 6465: float_exception_flags |= float_flag_inexact; ! 6466: } ! 6467: return z; ! 6468: ! 6469: } ! 6470: ! 6471: /*---------------------------------------------------------------------------- ! 6472: | Returns the result of adding the absolute values of the quadruple-precision ! 6473: | floating-point values `a' and `b'. If `zSign' is 1, the sum is negated ! 6474: | before being returned. `zSign' is ignored if the result is a NaN. ! 6475: | The addition is performed according to the IEC/IEEE Standard for Binary ! 6476: | Floating-Point Arithmetic. ! 6477: *----------------------------------------------------------------------------*/ ! 6478: ! 6479: static float128 addFloat128Sigs( float128 a, float128 b, flag zSign ) ! 6480: { ! 6481: int32 aExp, bExp, zExp; ! 6482: bits64 aSig0, aSig1, bSig0, bSig1, zSig0, zSig1, zSig2; ! 6483: int32 expDiff; ! 6484: ! 6485: aSig1 = extractFloat128Frac1( a ); ! 6486: aSig0 = extractFloat128Frac0( a ); ! 6487: aExp = extractFloat128Exp( a ); ! 6488: bSig1 = extractFloat128Frac1( b ); ! 6489: bSig0 = extractFloat128Frac0( b ); ! 6490: bExp = extractFloat128Exp( b ); ! 6491: expDiff = aExp - bExp; ! 6492: if ( 0 < expDiff ) { ! 6493: if ( aExp == 0x7FFF ) { ! 6494: if ( aSig0 | aSig1 ) return propagateFloat128NaN( a, b ); ! 6495: return a; ! 6496: } ! 6497: if ( bExp == 0 ) { ! 6498: --expDiff; ! 6499: } ! 6500: else { ! 6501: bSig0 |= LIT64( 0x0001000000000000 ); ! 6502: } ! 6503: shift128ExtraRightJamming( ! 6504: bSig0, bSig1, 0, expDiff, &bSig0, &bSig1, &zSig2 ); ! 6505: zExp = aExp; ! 6506: } ! 6507: else if ( expDiff < 0 ) { ! 6508: if ( bExp == 0x7FFF ) { ! 6509: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6510: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 6511: } ! 6512: if ( aExp == 0 ) { ! 6513: ++expDiff; ! 6514: } ! 6515: else { ! 6516: aSig0 |= LIT64( 0x0001000000000000 ); ! 6517: } ! 6518: shift128ExtraRightJamming( ! 6519: aSig0, aSig1, 0, - expDiff, &aSig0, &aSig1, &zSig2 ); ! 6520: zExp = bExp; ! 6521: } ! 6522: else { ! 6523: if ( aExp == 0x7FFF ) { ! 6524: if ( aSig0 | aSig1 | bSig0 | bSig1 ) { ! 6525: return propagateFloat128NaN( a, b ); ! 6526: } ! 6527: return a; ! 6528: } ! 6529: add128( aSig0, aSig1, bSig0, bSig1, &zSig0, &zSig1 ); ! 6530: if ( aExp == 0 ) return packFloat128( zSign, 0, zSig0, zSig1 ); ! 6531: zSig2 = 0; ! 6532: zSig0 |= LIT64( 0x0002000000000000 ); ! 6533: zExp = aExp; ! 6534: goto shiftRight1; ! 6535: } ! 6536: aSig0 |= LIT64( 0x0001000000000000 ); ! 6537: add128( aSig0, aSig1, bSig0, bSig1, &zSig0, &zSig1 ); ! 6538: --zExp; ! 6539: if ( zSig0 < LIT64( 0x0002000000000000 ) ) goto roundAndPack; ! 6540: ++zExp; ! 6541: shiftRight1: ! 6542: shift128ExtraRightJamming( ! 6543: zSig0, zSig1, zSig2, 1, &zSig0, &zSig1, &zSig2 ); ! 6544: roundAndPack: ! 6545: return roundAndPackFloat128( zSign, zExp, zSig0, zSig1, zSig2 ); ! 6546: ! 6547: } ! 6548: ! 6549: /*---------------------------------------------------------------------------- ! 6550: | Returns the result of subtracting the absolute values of the quadruple- ! 6551: | precision floating-point values `a' and `b'. If `zSign' is 1, the ! 6552: | difference is negated before being returned. `zSign' is ignored if the ! 6553: | result is a NaN. The subtraction is performed according to the IEC/IEEE ! 6554: | Standard for Binary Floating-Point Arithmetic. ! 6555: *----------------------------------------------------------------------------*/ ! 6556: ! 6557: static float128 subFloat128Sigs( float128 a, float128 b, flag zSign ) ! 6558: { ! 6559: int32 aExp, bExp, zExp; ! 6560: bits64 aSig0, aSig1, bSig0, bSig1, zSig0, zSig1; ! 6561: int32 expDiff; ! 6562: float128 z; ! 6563: ! 6564: aSig1 = extractFloat128Frac1( a ); ! 6565: aSig0 = extractFloat128Frac0( a ); ! 6566: aExp = extractFloat128Exp( a ); ! 6567: bSig1 = extractFloat128Frac1( b ); ! 6568: bSig0 = extractFloat128Frac0( b ); ! 6569: bExp = extractFloat128Exp( b ); ! 6570: expDiff = aExp - bExp; ! 6571: shortShift128Left( aSig0, aSig1, 14, &aSig0, &aSig1 ); ! 6572: shortShift128Left( bSig0, bSig1, 14, &bSig0, &bSig1 ); ! 6573: if ( 0 < expDiff ) goto aExpBigger; ! 6574: if ( expDiff < 0 ) goto bExpBigger; ! 6575: if ( aExp == 0x7FFF ) { ! 6576: if ( aSig0 | aSig1 | bSig0 | bSig1 ) { ! 6577: return propagateFloat128NaN( a, b ); ! 6578: } ! 6579: float_raise( float_flag_invalid ); ! 6580: z.low = float128_default_nan_low; ! 6581: z.high = float128_default_nan_high; ! 6582: return z; ! 6583: } ! 6584: if ( aExp == 0 ) { ! 6585: aExp = 1; ! 6586: bExp = 1; ! 6587: } ! 6588: if ( bSig0 < aSig0 ) goto aBigger; ! 6589: if ( aSig0 < bSig0 ) goto bBigger; ! 6590: if ( bSig1 < aSig1 ) goto aBigger; ! 6591: if ( aSig1 < bSig1 ) goto bBigger; ! 6592: return packFloat128( float_rounding_mode == float_round_down, 0, 0, 0 ); ! 6593: bExpBigger: ! 6594: if ( bExp == 0x7FFF ) { ! 6595: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6596: return packFloat128( zSign ^ 1, 0x7FFF, 0, 0 ); ! 6597: } ! 6598: if ( aExp == 0 ) { ! 6599: ++expDiff; ! 6600: } ! 6601: else { ! 6602: aSig0 |= LIT64( 0x4000000000000000 ); ! 6603: } ! 6604: shift128RightJamming( aSig0, aSig1, - expDiff, &aSig0, &aSig1 ); ! 6605: bSig0 |= LIT64( 0x4000000000000000 ); ! 6606: bBigger: ! 6607: sub128( bSig0, bSig1, aSig0, aSig1, &zSig0, &zSig1 ); ! 6608: zExp = bExp; ! 6609: zSign ^= 1; ! 6610: goto normalizeRoundAndPack; ! 6611: aExpBigger: ! 6612: if ( aExp == 0x7FFF ) { ! 6613: if ( aSig0 | aSig1 ) return propagateFloat128NaN( a, b ); ! 6614: return a; ! 6615: } ! 6616: if ( bExp == 0 ) { ! 6617: --expDiff; ! 6618: } ! 6619: else { ! 6620: bSig0 |= LIT64( 0x4000000000000000 ); ! 6621: } ! 6622: shift128RightJamming( bSig0, bSig1, expDiff, &bSig0, &bSig1 ); ! 6623: aSig0 |= LIT64( 0x4000000000000000 ); ! 6624: aBigger: ! 6625: sub128( aSig0, aSig1, bSig0, bSig1, &zSig0, &zSig1 ); ! 6626: zExp = aExp; ! 6627: normalizeRoundAndPack: ! 6628: --zExp; ! 6629: return normalizeRoundAndPackFloat128( zSign, zExp - 14, zSig0, zSig1 ); ! 6630: ! 6631: } ! 6632: ! 6633: /*---------------------------------------------------------------------------- ! 6634: | Returns the result of adding the quadruple-precision floating-point values ! 6635: | `a' and `b'. The operation is performed according to the IEC/IEEE Standard ! 6636: | for Binary Floating-Point Arithmetic. ! 6637: *----------------------------------------------------------------------------*/ ! 6638: ! 6639: float128 float128_add( float128 a, float128 b ) ! 6640: { ! 6641: flag aSign, bSign; ! 6642: ! 6643: aSign = extractFloat128Sign( a ); ! 6644: bSign = extractFloat128Sign( b ); ! 6645: if ( aSign == bSign ) { ! 6646: return addFloat128Sigs( a, b, aSign ); ! 6647: } ! 6648: else { ! 6649: return subFloat128Sigs( a, b, aSign ); ! 6650: } ! 6651: ! 6652: } ! 6653: ! 6654: /*---------------------------------------------------------------------------- ! 6655: | Returns the result of subtracting the quadruple-precision floating-point ! 6656: | values `a' and `b'. The operation is performed according to the IEC/IEEE ! 6657: | Standard for Binary Floating-Point Arithmetic. ! 6658: *----------------------------------------------------------------------------*/ ! 6659: ! 6660: float128 float128_sub( float128 a, float128 b ) ! 6661: { ! 6662: flag aSign, bSign; ! 6663: ! 6664: aSign = extractFloat128Sign( a ); ! 6665: bSign = extractFloat128Sign( b ); ! 6666: if ( aSign == bSign ) { ! 6667: return subFloat128Sigs( a, b, aSign ); ! 6668: } ! 6669: else { ! 6670: return addFloat128Sigs( a, b, aSign ); ! 6671: } ! 6672: ! 6673: } ! 6674: ! 6675: /*---------------------------------------------------------------------------- ! 6676: | Returns the result of multiplying the quadruple-precision floating-point ! 6677: | values `a' and `b'. The operation is performed according to the IEC/IEEE ! 6678: | Standard for Binary Floating-Point Arithmetic. ! 6679: *----------------------------------------------------------------------------*/ ! 6680: ! 6681: float128 float128_mul( float128 a, float128 b ) ! 6682: { ! 6683: flag aSign, bSign, zSign; ! 6684: int32 aExp, bExp, zExp; ! 6685: bits64 aSig0, aSig1, bSig0, bSig1, zSig0, zSig1, zSig2, zSig3; ! 6686: float128 z; ! 6687: ! 6688: aSig1 = extractFloat128Frac1( a ); ! 6689: aSig0 = extractFloat128Frac0( a ); ! 6690: aExp = extractFloat128Exp( a ); ! 6691: aSign = extractFloat128Sign( a ); ! 6692: bSig1 = extractFloat128Frac1( b ); ! 6693: bSig0 = extractFloat128Frac0( b ); ! 6694: bExp = extractFloat128Exp( b ); ! 6695: bSign = extractFloat128Sign( b ); ! 6696: zSign = aSign ^ bSign; ! 6697: if ( aExp == 0x7FFF ) { ! 6698: if ( ( aSig0 | aSig1 ) ! 6699: || ( ( bExp == 0x7FFF ) && ( bSig0 | bSig1 ) ) ) { ! 6700: return propagateFloat128NaN( a, b ); ! 6701: } ! 6702: if ( ( bExp | bSig0 | bSig1 ) == 0 ) goto invalid; ! 6703: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 6704: } ! 6705: if ( bExp == 0x7FFF ) { ! 6706: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6707: if ( ( aExp | aSig0 | aSig1 ) == 0 ) { ! 6708: invalid: ! 6709: float_raise( float_flag_invalid ); ! 6710: z.low = float128_default_nan_low; ! 6711: z.high = float128_default_nan_high; ! 6712: return z; ! 6713: } ! 6714: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 6715: } ! 6716: if ( aExp == 0 ) { ! 6717: if ( ( aSig0 | aSig1 ) == 0 ) return packFloat128( zSign, 0, 0, 0 ); ! 6718: normalizeFloat128Subnormal( aSig0, aSig1, &aExp, &aSig0, &aSig1 ); ! 6719: } ! 6720: if ( bExp == 0 ) { ! 6721: if ( ( bSig0 | bSig1 ) == 0 ) return packFloat128( zSign, 0, 0, 0 ); ! 6722: normalizeFloat128Subnormal( bSig0, bSig1, &bExp, &bSig0, &bSig1 ); ! 6723: } ! 6724: zExp = aExp + bExp - 0x4000; ! 6725: aSig0 |= LIT64( 0x0001000000000000 ); ! 6726: shortShift128Left( bSig0, bSig1, 16, &bSig0, &bSig1 ); ! 6727: mul128To256( aSig0, aSig1, bSig0, bSig1, &zSig0, &zSig1, &zSig2, &zSig3 ); ! 6728: add128( zSig0, zSig1, aSig0, aSig1, &zSig0, &zSig1 ); ! 6729: zSig2 |= ( zSig3 != 0 ); ! 6730: if ( LIT64( 0x0002000000000000 ) <= zSig0 ) { ! 6731: shift128ExtraRightJamming( ! 6732: zSig0, zSig1, zSig2, 1, &zSig0, &zSig1, &zSig2 ); ! 6733: ++zExp; ! 6734: } ! 6735: return roundAndPackFloat128( zSign, zExp, zSig0, zSig1, zSig2 ); ! 6736: ! 6737: } ! 6738: ! 6739: /*---------------------------------------------------------------------------- ! 6740: | Returns the result of dividing the quadruple-precision floating-point value ! 6741: | `a' by the corresponding value `b'. The operation is performed according to ! 6742: | the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 6743: *----------------------------------------------------------------------------*/ ! 6744: ! 6745: float128 float128_div( float128 a, float128 b ) ! 6746: { ! 6747: flag aSign, bSign, zSign; ! 6748: int32 aExp, bExp, zExp; ! 6749: bits64 aSig0, aSig1, bSig0, bSig1, zSig0, zSig1, zSig2; ! 6750: bits64 rem0, rem1, rem2, rem3, term0, term1, term2, term3; ! 6751: float128 z; ! 6752: ! 6753: aSig1 = extractFloat128Frac1( a ); ! 6754: aSig0 = extractFloat128Frac0( a ); ! 6755: aExp = extractFloat128Exp( a ); ! 6756: aSign = extractFloat128Sign( a ); ! 6757: bSig1 = extractFloat128Frac1( b ); ! 6758: bSig0 = extractFloat128Frac0( b ); ! 6759: bExp = extractFloat128Exp( b ); ! 6760: bSign = extractFloat128Sign( b ); ! 6761: zSign = aSign ^ bSign; ! 6762: if ( aExp == 0x7FFF ) { ! 6763: if ( aSig0 | aSig1 ) return propagateFloat128NaN( a, b ); ! 6764: if ( bExp == 0x7FFF ) { ! 6765: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6766: goto invalid; ! 6767: } ! 6768: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 6769: } ! 6770: if ( bExp == 0x7FFF ) { ! 6771: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6772: return packFloat128( zSign, 0, 0, 0 ); ! 6773: } ! 6774: if ( bExp == 0 ) { ! 6775: if ( ( bSig0 | bSig1 ) == 0 ) { ! 6776: if ( ( aExp | aSig0 | aSig1 ) == 0 ) { ! 6777: invalid: ! 6778: float_raise( float_flag_invalid ); ! 6779: z.low = float128_default_nan_low; ! 6780: z.high = float128_default_nan_high; ! 6781: return z; ! 6782: } ! 6783: float_raise( float_flag_divbyzero ); ! 6784: return packFloat128( zSign, 0x7FFF, 0, 0 ); ! 6785: } ! 6786: normalizeFloat128Subnormal( bSig0, bSig1, &bExp, &bSig0, &bSig1 ); ! 6787: } ! 6788: if ( aExp == 0 ) { ! 6789: if ( ( aSig0 | aSig1 ) == 0 ) return packFloat128( zSign, 0, 0, 0 ); ! 6790: normalizeFloat128Subnormal( aSig0, aSig1, &aExp, &aSig0, &aSig1 ); ! 6791: } ! 6792: zExp = aExp - bExp + 0x3FFD; ! 6793: shortShift128Left( ! 6794: aSig0 | LIT64( 0x0001000000000000 ), aSig1, 15, &aSig0, &aSig1 ); ! 6795: shortShift128Left( ! 6796: bSig0 | LIT64( 0x0001000000000000 ), bSig1, 15, &bSig0, &bSig1 ); ! 6797: if ( le128( bSig0, bSig1, aSig0, aSig1 ) ) { ! 6798: shift128Right( aSig0, aSig1, 1, &aSig0, &aSig1 ); ! 6799: ++zExp; ! 6800: } ! 6801: zSig0 = estimateDiv128To64( aSig0, aSig1, bSig0 ); ! 6802: mul128By64To192( bSig0, bSig1, zSig0, &term0, &term1, &term2 ); ! 6803: sub192( aSig0, aSig1, 0, term0, term1, term2, &rem0, &rem1, &rem2 ); ! 6804: while ( (sbits64) rem0 < 0 ) { ! 6805: --zSig0; ! 6806: add192( rem0, rem1, rem2, 0, bSig0, bSig1, &rem0, &rem1, &rem2 ); ! 6807: } ! 6808: zSig1 = estimateDiv128To64( rem1, rem2, bSig0 ); ! 6809: if ( ( zSig1 & 0x3FFF ) <= 4 ) { ! 6810: mul128By64To192( bSig0, bSig1, zSig1, &term1, &term2, &term3 ); ! 6811: sub192( rem1, rem2, 0, term1, term2, term3, &rem1, &rem2, &rem3 ); ! 6812: while ( (sbits64) rem1 < 0 ) { ! 6813: --zSig1; ! 6814: add192( rem1, rem2, rem3, 0, bSig0, bSig1, &rem1, &rem2, &rem3 ); ! 6815: } ! 6816: zSig1 |= ( ( rem1 | rem2 | rem3 ) != 0 ); ! 6817: } ! 6818: shift128ExtraRightJamming( zSig0, zSig1, 0, 15, &zSig0, &zSig1, &zSig2 ); ! 6819: return roundAndPackFloat128( zSign, zExp, zSig0, zSig1, zSig2 ); ! 6820: ! 6821: } ! 6822: ! 6823: /*---------------------------------------------------------------------------- ! 6824: | Returns the remainder of the quadruple-precision floating-point value `a' ! 6825: | with respect to the corresponding value `b'. The operation is performed ! 6826: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 6827: *----------------------------------------------------------------------------*/ ! 6828: ! 6829: float128 float128_rem( float128 a, float128 b ) ! 6830: { ! 6831: flag aSign, zSign; ! 6832: int32 aExp, bExp, expDiff; ! 6833: bits64 aSig0, aSig1, bSig0, bSig1, q, term0, term1, term2; ! 6834: bits64 allZero, alternateASig0, alternateASig1, sigMean1; ! 6835: sbits64 sigMean0; ! 6836: float128 z; ! 6837: ! 6838: aSig1 = extractFloat128Frac1( a ); ! 6839: aSig0 = extractFloat128Frac0( a ); ! 6840: aExp = extractFloat128Exp( a ); ! 6841: aSign = extractFloat128Sign( a ); ! 6842: bSig1 = extractFloat128Frac1( b ); ! 6843: bSig0 = extractFloat128Frac0( b ); ! 6844: bExp = extractFloat128Exp( b ); ! 6845: // bSign = extractFloat128Sign( b ); ! 6846: if ( aExp == 0x7FFF ) { ! 6847: if ( ( aSig0 | aSig1 ) ! 6848: || ( ( bExp == 0x7FFF ) && ( bSig0 | bSig1 ) ) ) { ! 6849: return propagateFloat128NaN( a, b ); ! 6850: } ! 6851: goto invalid; ! 6852: } ! 6853: if ( bExp == 0x7FFF ) { ! 6854: if ( bSig0 | bSig1 ) return propagateFloat128NaN( a, b ); ! 6855: return a; ! 6856: } ! 6857: if ( bExp == 0 ) { ! 6858: if ( ( bSig0 | bSig1 ) == 0 ) { ! 6859: invalid: ! 6860: float_raise( float_flag_invalid ); ! 6861: z.low = float128_default_nan_low; ! 6862: z.high = float128_default_nan_high; ! 6863: return z; ! 6864: } ! 6865: normalizeFloat128Subnormal( bSig0, bSig1, &bExp, &bSig0, &bSig1 ); ! 6866: } ! 6867: if ( aExp == 0 ) { ! 6868: if ( ( aSig0 | aSig1 ) == 0 ) return a; ! 6869: normalizeFloat128Subnormal( aSig0, aSig1, &aExp, &aSig0, &aSig1 ); ! 6870: } ! 6871: expDiff = aExp - bExp; ! 6872: if ( expDiff < -1 ) return a; ! 6873: shortShift128Left( ! 6874: aSig0 | LIT64( 0x0001000000000000 ), ! 6875: aSig1, ! 6876: 15 - ( expDiff < 0 ), ! 6877: &aSig0, ! 6878: &aSig1 ! 6879: ); ! 6880: shortShift128Left( ! 6881: bSig0 | LIT64( 0x0001000000000000 ), bSig1, 15, &bSig0, &bSig1 ); ! 6882: q = le128( bSig0, bSig1, aSig0, aSig1 ); ! 6883: if ( q ) sub128( aSig0, aSig1, bSig0, bSig1, &aSig0, &aSig1 ); ! 6884: expDiff -= 64; ! 6885: while ( 0 < expDiff ) { ! 6886: q = estimateDiv128To64( aSig0, aSig1, bSig0 ); ! 6887: q = ( 4 < q ) ? q - 4 : 0; ! 6888: mul128By64To192( bSig0, bSig1, q, &term0, &term1, &term2 ); ! 6889: shortShift192Left( term0, term1, term2, 61, &term1, &term2, &allZero ); ! 6890: shortShift128Left( aSig0, aSig1, 61, &aSig0, &allZero ); ! 6891: sub128( aSig0, 0, term1, term2, &aSig0, &aSig1 ); ! 6892: expDiff -= 61; ! 6893: } ! 6894: if ( -64 < expDiff ) { ! 6895: q = estimateDiv128To64( aSig0, aSig1, bSig0 ); ! 6896: q = ( 4 < q ) ? q - 4 : 0; ! 6897: q >>= - expDiff; ! 6898: shift128Right( bSig0, bSig1, 12, &bSig0, &bSig1 ); ! 6899: expDiff += 52; ! 6900: if ( expDiff < 0 ) { ! 6901: shift128Right( aSig0, aSig1, - expDiff, &aSig0, &aSig1 ); ! 6902: } ! 6903: else { ! 6904: shortShift128Left( aSig0, aSig1, expDiff, &aSig0, &aSig1 ); ! 6905: } ! 6906: mul128By64To192( bSig0, bSig1, q, &term0, &term1, &term2 ); ! 6907: sub128( aSig0, aSig1, term1, term2, &aSig0, &aSig1 ); ! 6908: } ! 6909: else { ! 6910: shift128Right( aSig0, aSig1, 12, &aSig0, &aSig1 ); ! 6911: shift128Right( bSig0, bSig1, 12, &bSig0, &bSig1 ); ! 6912: } ! 6913: do { ! 6914: alternateASig0 = aSig0; ! 6915: alternateASig1 = aSig1; ! 6916: ++q; ! 6917: sub128( aSig0, aSig1, bSig0, bSig1, &aSig0, &aSig1 ); ! 6918: } while ( 0 <= (sbits64) aSig0 ); ! 6919: add128( ! 6920: aSig0, aSig1, alternateASig0, alternateASig1, (bits64 *)&sigMean0, &sigMean1 ); ! 6921: if ( ( sigMean0 < 0 ) ! 6922: || ( ( ( sigMean0 | sigMean1 ) == 0 ) && ( q & 1 ) ) ) { ! 6923: aSig0 = alternateASig0; ! 6924: aSig1 = alternateASig1; ! 6925: } ! 6926: zSign = ( (sbits64) aSig0 < 0 ); ! 6927: if ( zSign ) sub128( 0, 0, aSig0, aSig1, &aSig0, &aSig1 ); ! 6928: return ! 6929: normalizeRoundAndPackFloat128( aSign ^ zSign, bExp - 4, aSig0, aSig1 ); ! 6930: ! 6931: } ! 6932: ! 6933: /*---------------------------------------------------------------------------- ! 6934: | Returns the square root of the quadruple-precision floating-point value `a'. ! 6935: | The operation is performed according to the IEC/IEEE Standard for Binary ! 6936: | Floating-Point Arithmetic. ! 6937: *----------------------------------------------------------------------------*/ ! 6938: ! 6939: float128 float128_sqrt( float128 a ) ! 6940: { ! 6941: flag aSign; ! 6942: int32 aExp, zExp; ! 6943: bits64 aSig0, aSig1, zSig0, zSig1, zSig2, doubleZSig0; ! 6944: bits64 rem0, rem1, rem2, rem3, term0, term1, term2, term3; ! 6945: float128 z; ! 6946: ! 6947: aSig1 = extractFloat128Frac1( a ); ! 6948: aSig0 = extractFloat128Frac0( a ); ! 6949: aExp = extractFloat128Exp( a ); ! 6950: aSign = extractFloat128Sign( a ); ! 6951: if ( aExp == 0x7FFF ) { ! 6952: if ( aSig0 | aSig1 ) return propagateFloat128NaN( a, a ); ! 6953: if ( ! aSign ) return a; ! 6954: goto invalid; ! 6955: } ! 6956: if ( aSign ) { ! 6957: if ( ( aExp | aSig0 | aSig1 ) == 0 ) return a; ! 6958: invalid: ! 6959: float_raise( float_flag_invalid ); ! 6960: z.low = float128_default_nan_low; ! 6961: z.high = float128_default_nan_high; ! 6962: return z; ! 6963: } ! 6964: if ( aExp == 0 ) { ! 6965: if ( ( aSig0 | aSig1 ) == 0 ) return packFloat128( 0, 0, 0, 0 ); ! 6966: normalizeFloat128Subnormal( aSig0, aSig1, &aExp, &aSig0, &aSig1 ); ! 6967: } ! 6968: zExp = ( ( aExp - 0x3FFF )>>1 ) + 0x3FFE; ! 6969: aSig0 |= LIT64( 0x0001000000000000 ); ! 6970: zSig0 = estimateSqrt32( aExp, aSig0>>17 ); ! 6971: shortShift128Left( aSig0, aSig1, 13 - ( aExp & 1 ), &aSig0, &aSig1 ); ! 6972: zSig0 = estimateDiv128To64( aSig0, aSig1, zSig0<<32 ) + ( zSig0<<30 ); ! 6973: doubleZSig0 = zSig0<<1; ! 6974: mul64To128( zSig0, zSig0, &term0, &term1 ); ! 6975: sub128( aSig0, aSig1, term0, term1, &rem0, &rem1 ); ! 6976: while ( (sbits64) rem0 < 0 ) { ! 6977: --zSig0; ! 6978: doubleZSig0 -= 2; ! 6979: add128( rem0, rem1, zSig0>>63, doubleZSig0 | 1, &rem0, &rem1 ); ! 6980: } ! 6981: zSig1 = estimateDiv128To64( rem1, 0, doubleZSig0 ); ! 6982: if ( ( zSig1 & 0x1FFF ) <= 5 ) { ! 6983: if ( zSig1 == 0 ) zSig1 = 1; ! 6984: mul64To128( doubleZSig0, zSig1, &term1, &term2 ); ! 6985: sub128( rem1, 0, term1, term2, &rem1, &rem2 ); ! 6986: mul64To128( zSig1, zSig1, &term2, &term3 ); ! 6987: sub192( rem1, rem2, 0, 0, term2, term3, &rem1, &rem2, &rem3 ); ! 6988: while ( (sbits64) rem1 < 0 ) { ! 6989: --zSig1; ! 6990: shortShift128Left( 0, zSig1, 1, &term2, &term3 ); ! 6991: term3 |= 1; ! 6992: term2 |= doubleZSig0; ! 6993: add192( rem1, rem2, rem3, 0, term2, term3, &rem1, &rem2, &rem3 ); ! 6994: } ! 6995: zSig1 |= ( ( rem1 | rem2 | rem3 ) != 0 ); ! 6996: } ! 6997: shift128ExtraRightJamming( zSig0, zSig1, 0, 14, &zSig0, &zSig1, &zSig2 ); ! 6998: return roundAndPackFloat128( 0, zExp, zSig0, zSig1, zSig2 ); ! 6999: ! 7000: } ! 7001: ! 7002: /*---------------------------------------------------------------------------- ! 7003: | Returns 1 if the quadruple-precision floating-point value `a' is equal to ! 7004: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 7005: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 7006: *----------------------------------------------------------------------------*/ ! 7007: ! 7008: flag float128_eq( float128 a, float128 b ) ! 7009: { ! 7010: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7011: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7012: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7013: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7014: ) { ! 7015: if ( float128_is_signaling_nan( a ) ! 7016: || float128_is_signaling_nan( b ) ) { ! 7017: float_raise( float_flag_invalid ); ! 7018: } ! 7019: return 0; ! 7020: } ! 7021: return ! 7022: ( a.low == b.low ) ! 7023: && ( ( a.high == b.high ) ! 7024: || ( ( a.low == 0 ) ! 7025: && ( (bits64) ( ( a.high | b.high )<<1 ) == 0 ) ) ! 7026: ); ! 7027: ! 7028: } ! 7029: ! 7030: /*---------------------------------------------------------------------------- ! 7031: | Returns 1 if the quadruple-precision floating-point value `a' is less than ! 7032: | or equal to the corresponding value `b', and 0 otherwise. The comparison ! 7033: | is performed according to the IEC/IEEE Standard for Binary Floating-Point ! 7034: | Arithmetic. ! 7035: *----------------------------------------------------------------------------*/ ! 7036: ! 7037: flag float128_le( float128 a, float128 b ) ! 7038: { ! 7039: flag aSign, bSign; ! 7040: ! 7041: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7042: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7043: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7044: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7045: ) { ! 7046: float_raise( float_flag_invalid ); ! 7047: return 0; ! 7048: } ! 7049: aSign = extractFloat128Sign( a ); ! 7050: bSign = extractFloat128Sign( b ); ! 7051: if ( aSign != bSign ) { ! 7052: return ! 7053: aSign ! 7054: || ( ( ( (bits64) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 7055: == 0 ); ! 7056: } ! 7057: return ! 7058: aSign ? le128( b.high, b.low, a.high, a.low ) ! 7059: : le128( a.high, a.low, b.high, b.low ); ! 7060: ! 7061: } ! 7062: ! 7063: /*---------------------------------------------------------------------------- ! 7064: | Returns 1 if the quadruple-precision floating-point value `a' is less than ! 7065: | the corresponding value `b', and 0 otherwise. The comparison is performed ! 7066: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 7067: *----------------------------------------------------------------------------*/ ! 7068: ! 7069: flag float128_lt( float128 a, float128 b ) ! 7070: { ! 7071: flag aSign, bSign; ! 7072: ! 7073: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7074: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7075: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7076: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7077: ) { ! 7078: float_raise( float_flag_invalid ); ! 7079: return 0; ! 7080: } ! 7081: aSign = extractFloat128Sign( a ); ! 7082: bSign = extractFloat128Sign( b ); ! 7083: if ( aSign != bSign ) { ! 7084: return ! 7085: aSign ! 7086: && ( ( ( (bits64) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 7087: != 0 ); ! 7088: } ! 7089: return ! 7090: aSign ? lt128( b.high, b.low, a.high, a.low ) ! 7091: : lt128( a.high, a.low, b.high, b.low ); ! 7092: ! 7093: } ! 7094: ! 7095: /*---------------------------------------------------------------------------- ! 7096: | Returns 1 if the quadruple-precision floating-point value `a' is equal to ! 7097: | the corresponding value `b', and 0 otherwise. The invalid exception is ! 7098: | raised if either operand is a NaN. Otherwise, the comparison is performed ! 7099: | according to the IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 7100: *----------------------------------------------------------------------------*/ ! 7101: ! 7102: flag float128_eq_signaling( float128 a, float128 b ) ! 7103: { ! 7104: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7105: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7106: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7107: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7108: ) { ! 7109: float_raise( float_flag_invalid ); ! 7110: return 0; ! 7111: } ! 7112: return ! 7113: ( a.low == b.low ) ! 7114: && ( ( a.high == b.high ) ! 7115: || ( ( a.low == 0 ) ! 7116: && ( (bits64) ( ( a.high | b.high )<<1 ) == 0 ) ) ! 7117: ); ! 7118: ! 7119: } ! 7120: ! 7121: /*---------------------------------------------------------------------------- ! 7122: | Returns 1 if the quadruple-precision floating-point value `a' is less than ! 7123: | or equal to the corresponding value `b', and 0 otherwise. Quiet NaNs do not ! 7124: | cause an exception. Otherwise, the comparison is performed according to the ! 7125: | IEC/IEEE Standard for Binary Floating-Point Arithmetic. ! 7126: *----------------------------------------------------------------------------*/ ! 7127: ! 7128: flag float128_le_quiet( float128 a, float128 b ) ! 7129: { ! 7130: flag aSign, bSign; ! 7131: ! 7132: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7133: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7134: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7135: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7136: ) { ! 7137: if ( float128_is_signaling_nan( a ) ! 7138: || float128_is_signaling_nan( b ) ) { ! 7139: float_raise( float_flag_invalid ); ! 7140: } ! 7141: return 0; ! 7142: } ! 7143: aSign = extractFloat128Sign( a ); ! 7144: bSign = extractFloat128Sign( b ); ! 7145: if ( aSign != bSign ) { ! 7146: return ! 7147: aSign ! 7148: || ( ( ( (bits64) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 7149: == 0 ); ! 7150: } ! 7151: return ! 7152: aSign ? le128( b.high, b.low, a.high, a.low ) ! 7153: : le128( a.high, a.low, b.high, b.low ); ! 7154: ! 7155: } ! 7156: ! 7157: /*---------------------------------------------------------------------------- ! 7158: | Returns 1 if the quadruple-precision floating-point value `a' is less than ! 7159: | the corresponding value `b', and 0 otherwise. Quiet NaNs do not cause an ! 7160: | exception. Otherwise, the comparison is performed according to the IEC/IEEE ! 7161: | Standard for Binary Floating-Point Arithmetic. ! 7162: *----------------------------------------------------------------------------*/ ! 7163: ! 7164: flag float128_lt_quiet( float128 a, float128 b ) ! 7165: { ! 7166: flag aSign, bSign; ! 7167: ! 7168: if ( ( ( extractFloat128Exp( a ) == 0x7FFF ) ! 7169: && ( extractFloat128Frac0( a ) | extractFloat128Frac1( a ) ) ) ! 7170: || ( ( extractFloat128Exp( b ) == 0x7FFF ) ! 7171: && ( extractFloat128Frac0( b ) | extractFloat128Frac1( b ) ) ) ! 7172: ) { ! 7173: if ( float128_is_signaling_nan( a ) ! 7174: || float128_is_signaling_nan( b ) ) { ! 7175: float_raise( float_flag_invalid ); ! 7176: } ! 7177: return 0; ! 7178: } ! 7179: aSign = extractFloat128Sign( a ); ! 7180: bSign = extractFloat128Sign( b ); ! 7181: if ( aSign != bSign ) { ! 7182: return ! 7183: aSign ! 7184: && ( ( ( (bits64) ( ( a.high | b.high )<<1 ) ) | a.low | b.low ) ! 7185: != 0 ); ! 7186: } ! 7187: return ! 7188: aSign ? lt128( b.high, b.low, a.high, a.low ) ! 7189: : lt128( a.high, a.low, b.high, b.low ); ! 7190: ! 7191: } ! 7192: ! 7193: #endif
This archive runs on limited infrastructure. Preserving old code on modern bandwidth. Automated agents are requested to crawl responsibly.