
=== 1. the version ===
tinygrad commit 06e57e3 2026-10-03 | python 3.13.16

=== 2. the c that tinygrad writes for relu(x*y+1) on 4 values, with its default optimisations (DEV=CPU DEBUG=4) ===
typedef float float4 __attribute__((aligned(16),ext_vector_type(4)));
void E_4(float* restrict data0_4, float* restrict data1_4, float* restrict data2_4) {
  float4 val0 = (*((float4*)((data1_4+0))));
  float4 val1 = (*((float4*)((data2_4+0))));
  float alu0 = ((val0[0]*val1[0])+1.0f);
  float alu1 = ((val0[1]*val1[1])+1.0f);
  float alu2 = ((val0[2]*val1[2])+1.0f);
  float alu3 = ((val0[3]*val1[3])+1.0f);
  float alu4 = ((alu0<0.0f)?0.0f:alu0);
  float alu5 = ((alu1<0.0f)?0.0f:alu1);
  float alu6 = ((alu2<0.0f)?0.0f:alu2);
  float alu7 = ((alu3<0.0f)?0.0f:alu3);
  *((float4*)((data0_4+0))) = (float4){alu4,alu5,alu6,alu7};
}
result: [11.0, 41.0, 91.0, 161.0]

=== 3. the same with optimisations off (NOOPT=1) ===
void E_4(float* restrict data0_4, float* restrict data1_4, float* restrict data2_4) {
  for (int Lidx0 = 0; Lidx0 < 4; Lidx0++) {
    float val0 = (*(data1_4+Lidx0));
    float val1 = (*(data2_4+Lidx0));
    float alu0 = ((val0*val1)+1.0f);
    float alu1 = ((alu0<0.0f)?0.0f:alu0);
    *(data0_4+Lidx0) = alu1;
  }
}

=== 4. relu(x*y+1) when x is nan ===
tinygrad, x = [nan, 1, -2, 0], y = 1: [nan, 2.0, 0.0, 1.0]

=== 5. the same simplifications on a float parameter with no range ===
f + 0.0            -> f
f * 0.0            -> const 0.0
f - f              -> const 0.0
f * 1.0            -> f
(f * 3.0) * 5.0    -> (f*15.0)

=== 6. constants as graph nodes ===
const 0.0 is const -0.0 : False
const nan is const nan  : True
const 1 is const True   : False

=== 7. signed zero through real tensors: -0.0 + 0.0 is +0.0 in IEEE ===
tinygrad, tensor + tensor of zeros : [0.0, 1.0]   sign of first: 1.0
tinygrad, tensor + the scalar 0.0  : [-0.0, 1.0]   sign of first: -1.0
numpy, float32 array + 0.0         : [0.0, 1.0]   sign of first: 1.0
