t27.aiРусский

XOR needs a bend

You will learn

Why a network needs a hidden ReLU layer to compute XOR, and what one missing bend does.

Recording pending: it waits on tri test and tri mutate plant from gHashTag/t27#7400, the two commands the recording runs, and until then the widget below is a placeholder that shows no run. XOR is 1 when exactly one input is 1, and a comment in gft_xornet.t27 says a single linear model cannot separate it. on_comb is a 2-layer net: two hidden sums, z0 and z1, each through relu, then a weighted sum of h0 and h1. It does not train: the tests pass hand-set weights, the analytic solution W=[[1,1],[1,1]], c=[0,-1], v=[1,-2]. The browser skips all 4 tests because its runner does not know assert_eq yet; the native t27c runs all 4, all pass, none vacuous. The recording removes relu from h1 alone, and exactly one test fails, x00: at (0,0) z1 is -1.0, and the output becomes 2.0 instead of 0. Every byte in the recording was printed by the command; only the typing is staged.

Try it

In the recording, find the changed line and the test that fails; then in the spec frame work out z1 for x10, x01 and x11 and see why relu changes nothing there.

Open the interactive lesson →

gft_xornet.t27: recording pending
gft_xornet.t27: recording pending ↗

Recording pending: waits on tri test and tri mutate plant from gHashTag/t27#7400. Until then this page is a placeholder and shows no run.

specs/ternary/gft_xornet.t27

module GftXorNet;
// #1764 + GF-T: a GF-T SGD weight update -- w' = w - eta * g, the final brick of an
// on-device training step (forward softmax -> loss -> gradient g -> THIS update).
// eta is the (positive) learning rate; g the gradient (signed); w the weight (signed).
// Composes the verified primitives: signed multiply (smul over the RNE magnitude
// mul) + subtract (sadd + neg). Bit-exact to the integer oracle; accuracy is to
// GF-T16 precision (<=1 ULP; ~0.03 abs at the largest magnitudes).
//
// Inputs: w, g, eta signed GF-T16 (u32). Output: updated weight w' GF-T16 (u32).

fn magadd(a: i32, b: i32) -> i32 {
    var ao : i32 = a >> 9; var am : i32 = a & 511;
    var bo : i32 = b >> 9; var bm : i32 = b & 511;
    var ho : i32 = bo; var hm : i32 = bm; var lo : i32 = ao; var lm : i32 = am;
    if (ao >= bo) { ho = ao; hm = am; lo = bo; lm = bm; }
    var hs : i32 = 512 + hm; var ls : i32 = 512 + lm;
    var d : i32 = ho - lo; if (d > 11) { d = 11; }
    var losh : i32 = ls >> d; var rem : i32 = ls - (losh << d);
    var s : i32 = hs + losh; var off : i32 = ho; var mant : i32 = s - 512;
    if (s >= 1024) {
        var g : i32 = s & 1; var pre : i32 = s >> 1; mant = pre - 512;
        if (g == 1) { if (rem > 0) { mant = mant + 1; } else { if ((pre & 1) == 1) { mant = mant + 1; } } }
        off = ho + 1; if (off >= 80) { off = 80; }
    } else {
        var t : i32 = rem << 1; var hf : i32 = 1 << d;
        if (t > hf) { mant = mant + 1; } else { if (t == hf) { if ((s & 1) == 1) { mant = mant + 1; } } }
    }
    if (mant >= 512) { mant = 0; off = off + 1; if (off >= 80) { off = 80; } }
    return (off << 9) | mant;
}

fn magsub(hi: i32, lo: i32) -> i32 {
    if (hi == lo) { return 0; }
    var ho : i32 = hi >> 9; var hm : i32 = hi & 511;
    var lo_o : i32 = lo >> 9; var lm : i32 = lo & 511;
    var d : i32 = ho - lo_o; var hs : i32 = (512 + hm) << 14;
    var la : i32 = 0; var sticky : i32 = 0;
    if (d >= 26) { la = 0; sticky = 1; }
    else { var ls : i32 = (512 + lm) << 14; la = ls >> d; if ((ls - (la << d)) > 0) { sticky = 1; } }
    var diff : i32 = hs - la; var off : i32 = ho;
    var cap : i32 = 12; if (off - 1 < cap) { cap = off - 1; } if (cap < 0) { cap = 0; }
    var sh : i32 = 0;
    if (diff != 0) {
        var t : i32 = diff;
        if (t < 65536) { if (sh + 8 <= cap) { t = t << 8; sh = sh + 8; } }
        if (t < 1048576) { if (sh + 4 <= cap) { t = t << 4; sh = sh + 4; } }
        if (t < 4194304) { if (sh + 2 <= cap) { t = t << 2; sh = sh + 2; } }
        if (t < 8388608) { if (sh + 1 <= cap) { t = t << 1; sh = sh + 1; } }
    }
    diff = diff << sh; off = off - sh;
    var q : i32 = diff >> 14; var rem : i32 = diff - (q << 14); var half : i32 = 8192; var mant : i32 = q - 512;
    if (rem > half) { mant = mant + 1; }
    else { if (rem == half) { if (sticky == 1) { mant = mant + 1; } else { if ((q & 1) == 1) { mant = mant + 1; } } } }
    if (mant >= 512) { mant = 0; off = off + 1; if (off >= 80) { off = 80; } }
    return (off << 9) | mant;
}

fn sadd(a: u32, b: u32) -> u32 {
    if (a == 0) { return b; }
    if (b == 0) { return a; }
    var sa : i32 = (a >> 16) as i32; var ma : i32 = (a & 65535) as i32;
    var sb : i32 = (b >> 16) as i32; var mb : i32 = (b & 65535) as i32;
    if (sa == sb) { return ((sa << 16) | magadd(ma, mb)) as u32; }
    var bsign : i32 = sa;
    var r : i32 = magsub(ma, mb);
    if (ma < mb) { r = magsub(mb, ma); bsign = sb; }
    if (r == 0) { return 0; }
    return ((bsign << 16) | r) as u32;
}

fn neg(v: u32) -> u32 {
    if (v == 0) { return 0; }
    return v ^ 65536;
}

fn magmul(a16: i32, b16: i32) -> i32 {
    var ao : i32 = a16 >> 9; var am : i32 = a16 & 511;
    var bo : i32 = b16 >> 9; var bm : i32 = b16 & 511;
    var prod : i32 = (512 + am) * (512 + bm);
    var carry : i32 = 0; if (prod >= 524288) { carry = 1; }
    var q : i32 = prod >> 9; var r : i32 = prod & 511; var half : i32 = 256;
    if (carry == 1) { q = prod >> 10; r = prod & 1023; half = 512; }
    var mant : i32 = q - 512;
    if (r > half) { mant = mant + 1; }
    if (r == half) { if ((q & 1) == 1) { mant = mant + 1; } }
    var sm : i32 = ao + bo + carry;
    var out_off : i32 = 0;
    if (sm >= 40) { var res : i32 = sm - 40; if (res >= 80) { out_off = 80; } else { out_off = res; } }
    if (mant >= 512) { mant = 0; out_off = out_off + 1; if (out_off >= 80) { out_off = 80; } }
    return (out_off << 9) | mant;
}

// softmax: p_sel = 2^(l_sel - M) / sum_i 2^(l_i - M), M = max logit.

// signed GF-T multiply: sign = xor of signs, magnitude = RNE magnitude mul.
fn smul(a: u32, b: u32) -> u32 {
    if (a == 0) { return 0; }
    if (b == 0) { return 0; }
    var sgn : i32 = ((a >> 16) & 1) as i32;
    var sb : i32 = ((b >> 16) & 1) as i32;
    if (sgn != sb) { sgn = 1; } else { sgn = 0; }
    var mag : i32 = magmul((a & 65535) as i32, (b & 65535) as i32);
    if (mag == 0) { return 0; }
    return ((sgn << 16) | mag) as u32;
}

// GF-T ReLU: max(0,z) — zero if z is zero or has the sign bit set, else z.
fn relu(z: u32) -> u32 {
    if (z == 0) { return 0; }
    if (((z >> 16) & 1) == 1) { return 0; }
    return z;
}
// ReLU derivative as a GF-T gate: 1.0 (=20480) if z>0, else 0.
fn relu_prime(z: u32) -> u32 {
    if (z == 0) { return 0; }
    if (((z >> 16) & 1) == 1) { return 0; }
    return 20480;
}
// 2-layer ReLU network forward pass (the canonical XOR net):
//   h = relu(W*x + c) ; y = v.h + b.  A single linear model cannot separate XOR;
//   the hidden ReLU layer makes it nonlinearly separable. Weights are baked in the
//   wrapper (analytic XOR: W=[[1,1],[1,1]], c=[0,-1], v=[1,-2]). Returns y (GF-T16).
fn on_comb(wh00: u32, wh01: u32, bh0: u32, wh10: u32, wh11: u32, bh1: u32,
           v0: u32, v1: u32, bo: u32, x0: u32, x1: u32) -> u32 {
    var z0 : u32 = sadd(sadd(smul(wh00, x0), smul(wh01, x1)), bh0);
    var z1 : u32 = sadd(sadd(smul(wh10, x0), smul(wh11, x1)), bh1);
    var h0 : u32 = relu(z0);
    var h1 : u32 = relu(z1);
    return sadd(sadd(smul(v0, h0), smul(v1, h1)), bo);
}
// analytic XOR weights: W=[[1,1],[1,1]], c=[0,-1], v=[1,-2], b=0.
// (0,0)->0, (1,0)->1.0(20480), (0,1)->1.0, (1,1)->0.
test x00 { assert_eq(on_comb(20480,20480,0, 20480,20480,86016, 20480,86528,0, 0,0), 0); }
test x10 { assert_eq(on_comb(20480,20480,0, 20480,20480,86016, 20480,86528,0, 20480,0), 20480); }
test x01 { assert_eq(on_comb(20480,20480,0, 20480,20480,86016, 20480,86528,0, 0,20480), 20480); }
test x11 { assert_eq(on_comb(20480,20480,0, 20480,20480,86016, 20480,86528,0, 20480,20480), 0); }

Open the lesson's spec in the player ↗

All lessons