Skip to content

ieee_754

IEEE 754 formats can be mathematically characterized by four integers:

name radix precision emin emax
binary32 2 24 -126 127
binary64 2 53 -1022 1023
binary128 2 113 -16382 16383
decimal64 10 16 -383 384
decimal128 10 34 -6143 6144

the significand always has a non-zero leading digit (except in the subnormal range). but in formats with radix == 2, the leading digit of the significand is always 1, since that is the only non-zero leading digit possible. so they can afford one extra digit of precision by implicitly assuming the leading bit is 1.

derived

name epsilon normal_min normal_max subnormal_min subnormal_max
binary32 1 * 2 ^ -23 1 * 2 ^ -126 16777215 * 2 ^ 104 1 * 2 ^ -149 8388607 * 2 ^ -149
binary64 1 * 2 ^ -52 1 * 2 ^ -1022 9007199254740991 * 2 ^ 971 1 * 2 ^ -1074 4503599627370495 * 2 ^ -1074
binary128 1 * 2 ^ -112 1 * 2 ^ -16382 10384593717069655257060992658440191 * 2 ^ 16271 1 * 2 ^ -16494 5192296858534827628530496329220095 * 2 ^ -16494
decimal64 1 * 10 ^ -15 1 * 10 ^ -383 9999999999999999 * 10 ^ 369 1 * 10 ^ -398 999999999999999 * 10 ^ -398
decimal128 1 * 10 ^ -33 1 * 10 ^ -6143 9999999999999999999999999999999999 * 10 ^ 6111 1 * 10 ^ -6176 999999999999999999999999999999999 * 10 ^ -6176

specials

nan: .nan # any payload
pos_inf: +.inf
neg_inf: -.inf
pos_zero: +0.0
neg_zero: -0.0

these special values are not mathematical, and are only pragmatic marker objects with special behaviour. they shall be stored under each format's namespace in their native datatype (if available)

sources

these constants are generated from:

#from daatypes import Float
import sys

sys.set_int_max_str_digits(10000)

fact_table = [[]]
derived_table = [[]]

for format_name, radix, prec, emin, emax in [
        # IEEE 754 basic formats only. no interchange formats.
        #('binary16', 2, 11, -14, 15),
        ('binary32', 2, 24, -126, 127),
        ('binary64', 2, 53, -1022, 1023),
        ('binary128', 2, 113, -16382, 16383),
        #('binary256', 2, 237, -262142, 262143),
        #('decimal32', 10, 7, -95, 96),
        ('decimal64', 10, 16, -383, 384),
        ('decimal128', 10, 34, -6143, 6144)]:
    print(f'\n{format_name}:')
    print(f'  nan: .nan')
    print(f'  pos_inf: +.inf')
    print(f'  neg_inf: -.inf')
    print(f'  pos_zero: +0.0')
    print(f'  neg_zero: -0.0')
    print(f'  radix:', radix)
    print(f'  precision:', prec)
    print(f'  emin:', emin)
    print(f'  emax:', emax)

    fact_table.append([format_name, radix, prec, emin, emax])

    derived_row = [format_name]

    for constant_name, constant in {
        'epsilon': (1, radix, 1 - prec),
        'normal_min': (1, radix, emin), 
        'normal_max': (radix ** prec - 1, radix, emax - prec + 1), 
        'subnormal_min': (1, radix, emin - prec + 1), 
        'subnormal_max': (radix ** (prec - 1) - 1, radix, emin - prec + 1), 
    }.items():
        significand, radix, exponent = constant
        print(f'  {constant_name}:')
        print(f'    significand: {significand}')
        print(f'    radix: {radix}')
        print(f'    exponent: {exponent}')

        derived_row.append(f'{significand} * {radix} ^ {exponent}')

    derived_table.append(derived_row)

print('| format | radix | precision | emin   | emax   |')
print('| ------ | ----- | --------- | ------ | ------ |')
for row in fact_table:
    print('| ' + ' | '.join(map(str, row)) + ' |')

print('| format | epsilon      | normal_min     | normal_max                                      | subnormal_min  | subnormal_max                                   |')
print('| ------ | ------------ | -------------- | ----------------------------------------------- | -------------- | ----------------------------------------------- |')
for row in derived_table:
    print('| ' + ' | '.join(map(str, row)) + ' |')

yaml

binary32:
  nan: .nan
  pos_inf: +.inf
  neg_inf: -.inf
  pos_zero: +0.0
  neg_zero: -0.0
  radix: 2
  precision: 24
  emin: -126
  emax: 127
  epsilon:
    significand: 1
    radix: 2
    exponent: -23
  normal_min:
    significand: 1
    radix: 2
    exponent: -126
  normal_max:
    significand: 16777215
    radix: 2
    exponent: 104
  subnormal_min:
    significand: 1
    radix: 2
    exponent: -149
  subnormal_max:
    significand: 8388607
    radix: 2
    exponent: -149

binary64:
  nan: .nan
  pos_inf: +.inf
  neg_inf: -.inf
  pos_zero: +0.0
  neg_zero: -0.0
  radix: 2
  precision: 53
  emin: -1022
  emax: 1023
  epsilon:
    significand: 1
    radix: 2
    exponent: -52
  normal_min:
    significand: 1
    radix: 2
    exponent: -1022
  normal_max:
    significand: 9007199254740991
    radix: 2
    exponent: 971
  subnormal_min:
    significand: 1
    radix: 2
    exponent: -1074
  subnormal_max:
    significand: 4503599627370495
    radix: 2
    exponent: -1074

binary128:
  nan: .nan
  pos_inf: +.inf
  neg_inf: -.inf
  pos_zero: +0.0
  neg_zero: -0.0
  radix: 2
  precision: 113
  emin: -16382
  emax: 16383
  epsilon:
    significand: 1
    radix: 2
    exponent: -112
  normal_min:
    significand: 1
    radix: 2
    exponent: -16382
  normal_max:
    significand: 10384593717069655257060992658440191
    radix: 2
    exponent: 16271
  subnormal_min:
    significand: 1
    radix: 2
    exponent: -16494
  subnormal_max:
    significand: 5192296858534827628530496329220095
    radix: 2
    exponent: -16494

decimal64:
  nan: .nan
  pos_inf: +.inf
  neg_inf: -.inf
  pos_zero: +0.0
  neg_zero: -0.0
  radix: 10
  precision: 16
  emin: -383
  emax: 384
  epsilon:
    significand: 1
    radix: 10
    exponent: -15
  normal_min:
    significand: 1
    radix: 10
    exponent: -383
  normal_max:
    significand: 9999999999999999
    radix: 10
    exponent: 369
  subnormal_min:
    significand: 1
    radix: 10
    exponent: -398
  subnormal_max:
    significand: 999999999999999
    radix: 10
    exponent: -398

decimal128:
  nan: .nan
  pos_inf: +.inf
  neg_inf: -.inf
  pos_zero: +0.0
  neg_zero: -0.0
  radix: 10
  precision: 34
  emin: -6143
  emax: 6144
  epsilon:
    significand: 1
    radix: 10
    exponent: -33
  normal_min:
    significand: 1
    radix: 10
    exponent: -6143
  normal_max:
    significand: 9999999999999999999999999999999999
    radix: 10
    exponent: 6111
  subnormal_min:
    significand: 1
    radix: 10
    exponent: -6176
  subnormal_max:
    significand: 999999999999999999999999999999999
    radix: 10
    exponent: -6176