| author | |
| committer | |
| log | 327482c3a41a30d5ea63eedd3bfb13553c8c2d00 |
| tree | ef668f8342632858ee20e6d9a05d9c93d8e5b704 |
| parent | 68dcdf1c867a918aa0bb7b5c4cb42df185e1c8f9 |
| parent | 353419f82d3575dc45631750a8cf08aa4826ec4c |
| signature |
Default to strict IEEE floating point18 files changed, 17 insertions(+), 61 deletions(-)
doc/langref.html.in+5-5| ... | ... | @@ -744,19 +744,19 @@ const yet_another_hex_float = 0x103.70P-5; |
| 744 | 744 | {#code_end#} |
| 745 | 745 | {#header_close#} |
| 746 | 746 | {#header_open|Floating Point Operations#} |
| 747 | <p>By default floating point operations use <code>Optimized</code> mode, | |
| 748 | but you can switch to <code>Strict</code> mode on a per-block basis:</p> | |
| 747 | <p>By default floating point operations use <code>Strict</code> mode, | |
| 748 | but you can switch to <code>Optimized</code> mode on a per-block basis:</p> | |
| 749 | 749 | {#code_begin|obj|foo#} |
| 750 | 750 | {#code_release_fast#} |
| 751 | 751 | const builtin = @import("builtin"); |
| 752 | 752 | const big = f64(1 << 40); |
| 753 | 753 | |
| 754 | 754 | export fn foo_strict(x: f64) f64 { |
| 755 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 756 | 755 | return x + big - big; |
| 757 | 756 | } |
| 758 | 757 | |
| 759 | 758 | export fn foo_optimized(x: f64) f64 { |
| 759 | @setFloatMode(this, builtin.FloatMode.Optimized); | |
| 760 | 760 | return x + big - big; |
| 761 | 761 | } |
| 762 | 762 | {#code_end#} |
| ... | ... | @@ -5948,7 +5948,7 @@ pub const FloatMode = enum { |
| 5948 | 5948 | {#code_end#} |
| 5949 | 5949 | <ul> |
| 5950 | 5950 | <li> |
| 5951 | <code>Optimized</code> (default) - Floating point operations may do all of the following: | |
| 5951 | <code>Optimized</code> - Floating point operations may do all of the following: | |
| 5952 | 5952 | <ul> |
| 5953 | 5953 | <li>Assume the arguments and result are not NaN. Optimizations are required to retain defined behavior over NaNs, but the value of the result is undefined.</li> |
| 5954 | 5954 | <li>Assume the arguments and result are not +/-Inf. Optimizations are required to retain defined behavior over +/-Inf, but the value of the result is undefined.</li> |
| ... | ... | @@ -5960,7 +5960,7 @@ pub const FloatMode = enum { |
| 5960 | 5960 | This is equivalent to <code>-ffast-math</code> in GCC. |
| 5961 | 5961 | </li> |
| 5962 | 5962 | <li> |
| 5963 | <code>Strict</code> - Floating point operations follow strict IEEE compliance. | |
| 5963 | <code>Strict</code> (default) - Floating point operations follow strict IEEE compliance. | |
| 5964 | 5964 | </li> |
| 5965 | 5965 | </ul> |
| 5966 | 5966 | {#see_also|Floating Point Operations#} |
src/all_types.hpp+2-2| ... | ... | @@ -1852,7 +1852,7 @@ struct ScopeDecls { |
| 1852 | 1852 | HashMap<Buf *, Tld *, buf_hash, buf_eql_buf> decl_table; |
| 1853 | 1853 | bool safety_off; |
| 1854 | 1854 | AstNode *safety_set_node; |
| 1855 | bool fast_math_off; | |
| 1855 | bool fast_math_on; | |
| 1856 | 1856 | AstNode *fast_math_set_node; |
| 1857 | 1857 | ImportTableEntry *import; |
| 1858 | 1858 | // If this is a scope from a container, this is the type entry, otherwise null |
| ... | ... | @@ -1872,7 +1872,7 @@ struct ScopeBlock { |
| 1872 | 1872 | |
| 1873 | 1873 | bool safety_off; |
| 1874 | 1874 | AstNode *safety_set_node; |
| 1875 | bool fast_math_off; | |
| 1875 | bool fast_math_on; | |
| 1876 | 1876 | AstNode *fast_math_set_node; |
| 1877 | 1877 | }; |
| 1878 | 1878 |
src/codegen.cpp+3-3| ... | ... | @@ -829,15 +829,15 @@ static bool ir_want_fast_math(CodeGen *g, IrInstruction *instruction) { |
| 829 | 829 | if (scope->id == ScopeIdBlock) { |
| 830 | 830 | ScopeBlock *block_scope = (ScopeBlock *)scope; |
| 831 | 831 | if (block_scope->fast_math_set_node) |
| 832 | return !block_scope->fast_math_off; | |
| 832 | return block_scope->fast_math_on; | |
| 833 | 833 | } else if (scope->id == ScopeIdDecls) { |
| 834 | 834 | ScopeDecls *decls_scope = (ScopeDecls *)scope; |
| 835 | 835 | if (decls_scope->fast_math_set_node) |
| 836 | return !decls_scope->fast_math_off; | |
| 836 | return decls_scope->fast_math_on; | |
| 837 | 837 | } |
| 838 | 838 | scope = scope->parent; |
| 839 | 839 | } |
| 840 | return true; | |
| 840 | return false; | |
| 841 | 841 | } |
| 842 | 842 | |
| 843 | 843 | static bool ir_want_runtime_safety(CodeGen *g, IrInstruction *instruction) { |
src/ir.cpp+5-5| ... | ... | @@ -15200,17 +15200,17 @@ static TypeTableEntry *ir_analyze_instruction_set_float_mode(IrAnalyze *ira, |
| 15200 | 15200 | return ira->codegen->builtin_types.entry_void; |
| 15201 | 15201 | } |
| 15202 | 15202 | |
| 15203 | bool *fast_math_off_ptr; | |
| 15203 | bool *fast_math_on_ptr; | |
| 15204 | 15204 | AstNode **fast_math_set_node_ptr; |
| 15205 | 15205 | if (target_type->id == TypeTableEntryIdBlock) { |
| 15206 | 15206 | ScopeBlock *block_scope = (ScopeBlock *)target_val->data.x_block; |
| 15207 | fast_math_off_ptr = &block_scope->fast_math_off; | |
| 15207 | fast_math_on_ptr = &block_scope->fast_math_on; | |
| 15208 | 15208 | fast_math_set_node_ptr = &block_scope->fast_math_set_node; |
| 15209 | 15209 | } else if (target_type->id == TypeTableEntryIdFn) { |
| 15210 | 15210 | assert(target_val->data.x_ptr.special == ConstPtrSpecialFunction); |
| 15211 | 15211 | FnTableEntry *target_fn = target_val->data.x_ptr.data.fn.fn_entry; |
| 15212 | 15212 | assert(target_fn->def_scope); |
| 15213 | fast_math_off_ptr = &target_fn->def_scope->fast_math_off; | |
| 15213 | fast_math_on_ptr = &target_fn->def_scope->fast_math_on; | |
| 15214 | 15214 | fast_math_set_node_ptr = &target_fn->def_scope->fast_math_set_node; |
| 15215 | 15215 | } else if (target_type->id == TypeTableEntryIdMetaType) { |
| 15216 | 15216 | ScopeDecls *decls_scope; |
| ... | ... | @@ -15226,7 +15226,7 @@ static TypeTableEntry *ir_analyze_instruction_set_float_mode(IrAnalyze *ira, |
| 15226 | 15226 | buf_sprintf("expected scope reference, found type '%s'", buf_ptr(&type_arg->name))); |
| 15227 | 15227 | return ira->codegen->builtin_types.entry_invalid; |
| 15228 | 15228 | } |
| 15229 | fast_math_off_ptr = &decls_scope->fast_math_off; | |
| 15229 | fast_math_on_ptr = &decls_scope->fast_math_on; | |
| 15230 | 15230 | fast_math_set_node_ptr = &decls_scope->fast_math_set_node; |
| 15231 | 15231 | } else { |
| 15232 | 15232 | ir_add_error_node(ira, target_instruction->source_node, |
| ... | ... | @@ -15248,7 +15248,7 @@ static TypeTableEntry *ir_analyze_instruction_set_float_mode(IrAnalyze *ira, |
| 15248 | 15248 | return ira->codegen->builtin_types.entry_invalid; |
| 15249 | 15249 | } |
| 15250 | 15250 | *fast_math_set_node_ptr = source_node; |
| 15251 | *fast_math_off_ptr = (float_mode_scalar == FloatModeStrict); | |
| 15251 | *fast_math_on_ptr = (float_mode_scalar == FloatModeOptimized); | |
| 15252 | 15252 | |
| 15253 | 15253 | ir_build_const_from(ira, &instruction->base); |
| 15254 | 15254 | return ira->codegen->builtin_types.entry_void; |
std/fmt/errol/index.zig-4| ... | ... | @@ -253,11 +253,7 @@ fn gethi(in: f64) f64 { |
| 253 | 253 | /// Normalize the number by factoring in the error. |
| 254 | 254 | /// @hp: The float pair. |
| 255 | 255 | fn hpNormalize(hp: *HP) void { |
| 256 | // Required to avoid segfaults causing buffer overrun during errol3 digit output termination. | |
| 257 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 258 | ||
| 259 | 256 | const val = hp.val; |
| 260 | ||
| 261 | 257 | hp.val += hp.off; |
| 262 | 258 | hp.off += val - hp.val; |
| 263 | 259 | } |
std/math/ceil.zig-2| ... | ... | @@ -61,10 +61,8 @@ fn ceil64(x: f64) f64 { |
| 61 | 61 | } |
| 62 | 62 | |
| 63 | 63 | if (u >> 63 != 0) { |
| 64 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 65 | 64 | y = x - math.f64_toint + math.f64_toint - x; |
| 66 | 65 | } else { |
| 67 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 68 | 66 | y = x + math.f64_toint - math.f64_toint - x; |
| 69 | 67 | } |
| 70 | 68 |
std/math/complex/exp.zig-2| ... | ... | @@ -17,8 +17,6 @@ pub fn exp(z: var) @typeOf(z) { |
| 17 | 17 | } |
| 18 | 18 | |
| 19 | 19 | fn exp32(z: Complex(f32)) Complex(f32) { |
| 20 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 21 | ||
| 22 | 20 | const exp_overflow = 0x42b17218; // max_exp * ln2 ~= 88.72283955 |
| 23 | 21 | const cexp_overflow = 0x43400074; // (max_exp - min_denom_exp) * ln2 |
| 24 | 22 |
std/math/cos.zig-2| ... | ... | @@ -37,8 +37,6 @@ const C5 = 4.16666666666665929218E-2; |
| 37 | 37 | // |
| 38 | 38 | // This may have slight differences on some edge cases and may need to replaced if so. |
| 39 | 39 | fn cos32(x_: f32) f32 { |
| 40 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 41 | ||
| 42 | 40 | const pi4a = 7.85398125648498535156e-1; |
| 43 | 41 | const pi4b = 3.77489470793079817668E-8; |
| 44 | 42 | const pi4c = 2.69515142907905952645E-15; |
std/math/exp.zig-4| ... | ... | @@ -18,8 +18,6 @@ pub fn exp(x: var) @typeOf(x) { |
| 18 | 18 | } |
| 19 | 19 | |
| 20 | 20 | fn exp32(x_: f32) f32 { |
| 21 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 22 | ||
| 23 | 21 | const half = []f32{ 0.5, -0.5 }; |
| 24 | 22 | const ln2hi = 6.9314575195e-1; |
| 25 | 23 | const ln2lo = 1.4286067653e-6; |
| ... | ... | @@ -95,8 +93,6 @@ fn exp32(x_: f32) f32 { |
| 95 | 93 | } |
| 96 | 94 | |
| 97 | 95 | fn exp64(x_: f64) f64 { |
| 98 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 99 | ||
| 100 | 96 | const half = []const f64{ 0.5, -0.5 }; |
| 101 | 97 | const ln2hi: f64 = 6.93147180369123816490e-01; |
| 102 | 98 | const ln2lo: f64 = 1.90821492927058770002e-10; |
std/math/exp2.zig-4| ... | ... | @@ -36,8 +36,6 @@ const exp2ft = []const f64{ |
| 36 | 36 | }; |
| 37 | 37 | |
| 38 | 38 | fn exp2_32(x: f32) f32 { |
| 39 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 40 | ||
| 41 | 39 | const tblsiz = @intCast(u32, exp2ft.len); |
| 42 | 40 | const redux: f32 = 0x1.8p23 / @intToFloat(f32, tblsiz); |
| 43 | 41 | const P1: f32 = 0x1.62e430p-1; |
| ... | ... | @@ -353,8 +351,6 @@ const exp2dt = []f64{ |
| 353 | 351 | }; |
| 354 | 352 | |
| 355 | 353 | fn exp2_64(x: f64) f64 { |
| 356 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 357 | ||
| 358 | 354 | const tblsiz = @intCast(u32, exp2dt.len / 2); |
| 359 | 355 | const redux: f64 = 0x1.8p52 / @intToFloat(f64, tblsiz); |
| 360 | 356 | const P1: f64 = 0x1.62e42fefa39efp-1; |
std/math/expm1.zig-4| ... | ... | @@ -19,8 +19,6 @@ pub fn expm1(x: var) @typeOf(x) { |
| 19 | 19 | } |
| 20 | 20 | |
| 21 | 21 | fn expm1_32(x_: f32) f32 { |
| 22 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 23 | ||
| 24 | 22 | if (math.isNan(x_)) |
| 25 | 23 | return math.nan(f32); |
| 26 | 24 | |
| ... | ... | @@ -149,8 +147,6 @@ fn expm1_32(x_: f32) f32 { |
| 149 | 147 | } |
| 150 | 148 | |
| 151 | 149 | fn expm1_64(x_: f64) f64 { |
| 152 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 153 | ||
| 154 | 150 | if (math.isNan(x_)) |
| 155 | 151 | return math.nan(f64); |
| 156 | 152 |
std/math/floor.zig-2| ... | ... | @@ -97,10 +97,8 @@ fn floor64(x: f64) f64 { |
| 97 | 97 | } |
| 98 | 98 | |
| 99 | 99 | if (u >> 63 != 0) { |
| 100 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 101 | 100 | y = x - math.f64_toint + math.f64_toint - x; |
| 102 | 101 | } else { |
| 103 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 104 | 102 | y = x + math.f64_toint - math.f64_toint - x; |
| 105 | 103 | } |
| 106 | 104 |
std/math/ln.zig-4| ... | ... | @@ -35,8 +35,6 @@ pub fn ln(x: var) @typeOf(x) { |
| 35 | 35 | } |
| 36 | 36 | |
| 37 | 37 | pub fn ln_32(x_: f32) f32 { |
| 38 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 39 | ||
| 40 | 38 | const ln2_hi: f32 = 6.9313812256e-01; |
| 41 | 39 | const ln2_lo: f32 = 9.0580006145e-06; |
| 42 | 40 | const Lg1: f32 = 0xaaaaaa.0p-24; |
| ... | ... | @@ -89,8 +87,6 @@ pub fn ln_32(x_: f32) f32 { |
| 89 | 87 | } |
| 90 | 88 | |
| 91 | 89 | pub fn ln_64(x_: f64) f64 { |
| 92 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 93 | ||
| 94 | 90 | const ln2_hi: f64 = 6.93147180369123816490e-01; |
| 95 | 91 | const ln2_lo: f64 = 1.90821492927058770002e-10; |
| 96 | 92 | const Lg1: f64 = 6.666666666666735130e-01; |
std/math/pow.zig-2| ... | ... | @@ -28,8 +28,6 @@ const assert = std.debug.assert; |
| 28 | 28 | |
| 29 | 29 | // This implementation is taken from the go stlib, musl is a bit more complex. |
| 30 | 30 | pub fn pow(comptime T: type, x: T, y: T) T { |
| 31 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 32 | ||
| 33 | 31 | if (T != f32 and T != f64) { |
| 34 | 32 | @compileError("pow not implemented for " ++ @typeName(T)); |
| 35 | 33 | } |
std/math/round.zig+2-10| ... | ... | @@ -35,11 +35,7 @@ fn round32(x_: f32) f32 { |
| 35 | 35 | return 0 * @bitCast(f32, u); |
| 36 | 36 | } |
| 37 | 37 | |
| 38 | { | |
| 39 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 40 | y = x + math.f32_toint - math.f32_toint - x; | |
| 41 | } | |
| 42 | ||
| 38 | y = x + math.f32_toint - math.f32_toint - x; | |
| 43 | 39 | if (y > 0.5) { |
| 44 | 40 | y = y + x - 1; |
| 45 | 41 | } else if (y <= -0.5) { |
| ... | ... | @@ -72,11 +68,7 @@ fn round64(x_: f64) f64 { |
| 72 | 68 | return 0 * @bitCast(f64, u); |
| 73 | 69 | } |
| 74 | 70 | |
| 75 | { | |
| 76 | @setFloatMode(this, builtin.FloatMode.Strict); | |
| 77 | y = x + math.f64_toint - math.f64_toint - x; | |
| 78 | } | |
| 79 | ||
| 71 | y = x + math.f64_toint - math.f64_toint - x; | |
| 80 | 72 | if (y > 0.5) { |
| 81 | 73 | y = y + x - 1; |
| 82 | 74 | } else if (y <= -0.5) { |
std/math/sin.zig-2| ... | ... | @@ -38,8 +38,6 @@ const C5 = 4.16666666666665929218E-2; |
| 38 | 38 | // |
| 39 | 39 | // This may have slight differences on some edge cases and may need to replaced if so. |
| 40 | 40 | fn sin32(x_: f32) f32 { |
| 41 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 42 | ||
| 43 | 41 | const pi4a = 7.85398125648498535156e-1; |
| 44 | 42 | const pi4b = 3.77489470793079817668E-8; |
| 45 | 43 | const pi4c = 2.69515142907905952645E-15; |
std/math/sinh.zig-2| ... | ... | @@ -54,8 +54,6 @@ fn sinh32(x: f32) f32 { |
| 54 | 54 | } |
| 55 | 55 | |
| 56 | 56 | fn sinh64(x: f64) f64 { |
| 57 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 58 | ||
| 59 | 57 | const u = @bitCast(u64, x); |
| 60 | 58 | const w = @intCast(u32, u >> 32); |
| 61 | 59 | const ax = @bitCast(f64, u & (@maxValue(u64) >> 1)); |
std/math/tan.zig-2| ... | ... | @@ -31,8 +31,6 @@ const Tq4 = -5.38695755929454629881E7; |
| 31 | 31 | // |
| 32 | 32 | // This may have slight differences on some edge cases and may need to replaced if so. |
| 33 | 33 | fn tan32(x_: f32) f32 { |
| 34 | @setFloatMode(this, @import("builtin").FloatMode.Strict); | |
| 35 | ||
| 36 | 34 | const pi4a = 7.85398125648498535156e-1; |
| 37 | 35 | const pi4b = 3.77489470793079817668E-8; |
| 38 | 36 | const pi4c = 2.69515142907905952645E-15; |