From be3bd1d95944f1d9d13ed5450f7ecc8a93e51efb Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:33:48 +0300 Subject: [PATCH 1/7] train: add full-parameter training API --- Core_CPP/niyah_train_full.h | 38 +++++++++++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) create mode 100644 Core_CPP/niyah_train_full.h diff --git a/Core_CPP/niyah_train_full.h b/Core_CPP/niyah_train_full.h new file mode 100644 index 0000000..cdbf2e7 --- /dev/null +++ b/Core_CPP/niyah_train_full.h @@ -0,0 +1,38 @@ +#ifndef NIYAH_TRAIN_FULL_H +#define NIYAH_TRAIN_FULL_H + +#include "niyah_core.h" + +#ifdef __cplusplus +extern "C" { +#endif + +/* + * Deterministic non-zero initialization for a fresh model. + * Allocation itself remains zero-initialized so loading a checkpoint keeps + * its existing semantics. Call this exactly once before training from scratch. + */ +void niyah_init_weights(NiyahModel *m, uint64_t seed); + +/* + * Full-parameter autoregressive training step. + * + * Updates token embeddings, every attention/FFN projection, RMSNorm scales, + * and the LM head with AdamW. The causal KV cache is a deliberate truncated + * backpropagation boundary: a token receives gradients through its own Q/K/V + * path, while future losses do not backpropagate into earlier cached K/V + * states. This keeps the C trainer bounded and deterministic without claiming + * exact full-sequence BPTT. + * + * Returns mean next-token cross-entropy. Returns NAN for invalid token ids, + * an oversized sequence, invalid pointers, or non-finite arithmetic. + * Returns 0 for n < 2. + */ +float niyah_full_train_step(NiyahModel *m, NiyahAdam *opt, + const uint32_t *tokens, uint32_t n); + +#ifdef __cplusplus +} +#endif + +#endif /* NIYAH_TRAIN_FULL_H */ From 2577a0efd02d1eccca97a83b461bfe0bcd1494ec Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:35:02 +0300 Subject: [PATCH 2/7] train: implement full-parameter truncated-BPTT step --- Core_CPP/niyah_train_full.c | 633 ++++++++++++++++++++++++++++++++++++ 1 file changed, 633 insertions(+) create mode 100644 Core_CPP/niyah_train_full.c diff --git a/Core_CPP/niyah_train_full.c b/Core_CPP/niyah_train_full.c new file mode 100644 index 0000000..069dab5 --- /dev/null +++ b/Core_CPP/niyah_train_full.c @@ -0,0 +1,633 @@ +#include "niyah_train_full.h" + +#include +#include +#include +#include +#include + +static void *train_calloc(size_t n, size_t size) +{ + if (size != 0u && n > SIZE_MAX / size) return NULL; + return calloc(n, size); +} + +static float train_dot(const float *a, const float *b, uint32_t n) +{ + float sum = 0.0f; + uint32_t i; + for (i = 0u; i < n; ++i) sum += a[i] * b[i]; + return sum; +} + +static void train_matvec(float *out, const float *matrix, const float *in, + uint32_t rows, uint32_t cols) +{ + uint32_t r; + for (r = 0u; r < rows; ++r) { + const float *row = matrix + (size_t)r * cols; + out[r] = train_dot(row, in, cols); + } +} + +static void train_matvec_t_add(float *out, const float *matrix, const float *grad_out, + uint32_t rows, uint32_t cols) +{ + uint32_t r; + for (r = 0u; r < rows; ++r) { + const float g = grad_out[r]; + const float *row = matrix + (size_t)r * cols; + uint32_t c; + for (c = 0u; c < cols; ++c) out[c] += row[c] * g; + } +} + +static void train_outer_add(float *grad_matrix, const float *grad_out, const float *in, + uint32_t rows, uint32_t cols) +{ + uint32_t r; + for (r = 0u; r < rows; ++r) { + const float g = grad_out[r]; + float *row = grad_matrix + (size_t)r * cols; + uint32_t c; + for (c = 0u; c < cols; ++c) row[c] += g * in[c]; + } +} + +static void train_rmsnorm(float *out, const float *x, const float *weight, + uint32_t n, float eps) +{ + float ss = 0.0f; + float scale; + uint32_t i; + for (i = 0u; i < n; ++i) ss += x[i] * x[i]; + scale = 1.0f / sqrtf(ss / (float)n + eps); + for (i = 0u; i < n; ++i) out[i] = x[i] * weight[i] * scale; +} + +static void train_rmsnorm_backward(float *grad_x, float *grad_weight, + const float *x, const float *weight, + const float *grad_out, uint32_t n, float eps) +{ + float ss = 0.0f; + float weighted_dot = 0.0f; + float scale; + float common; + uint32_t i; + + for (i = 0u; i < n; ++i) ss += x[i] * x[i]; + scale = 1.0f / sqrtf(ss / (float)n + eps); + for (i = 0u; i < n; ++i) { + weighted_dot += grad_out[i] * weight[i] * x[i]; + grad_weight[i] += grad_out[i] * x[i] * scale; + } + common = (scale * scale * scale) * weighted_dot / (float)n; + for (i = 0u; i < n; ++i) { + grad_x[i] += grad_out[i] * weight[i] * scale - x[i] * common; + } +} + +static void train_rope(float *x, uint32_t pos, uint32_t head_dim, float theta) +{ + uint32_t i; + for (i = 0u; i + 1u < head_dim; i += 2u) { + const float angle = (float)pos / powf(theta, (float)i / (float)head_dim); + const float c = cosf(angle); + const float s = sinf(angle); + const float x0 = x[i]; + const float x1 = x[i + 1u]; + x[i] = x0 * c - x1 * s; + x[i + 1u] = x0 * s + x1 * c; + } +} + +static void train_rope_backward(float *grad, uint32_t pos, uint32_t head_dim, float theta) +{ + uint32_t i; + for (i = 0u; i + 1u < head_dim; i += 2u) { + const float angle = (float)pos / powf(theta, (float)i / (float)head_dim); + const float c = cosf(angle); + const float s = sinf(angle); + const float g0 = grad[i]; + const float g1 = grad[i + 1u]; + grad[i] = g0 * c + g1 * s; + grad[i + 1u] = -g0 * s + g1 * c; + } +} + +static float train_silu(float x) +{ + return x / (1.0f + expf(-x)); +} + +static float train_silu_grad(float x) +{ + const float sig = 1.0f / (1.0f + expf(-x)); + return sig * (1.0f + x * (1.0f - sig)); +} + +static uint64_t train_splitmix64(uint64_t *state) +{ + uint64_t z; + *state += UINT64_C(0x9E3779B97F4A7C15); + z = *state; + z = (z ^ (z >> 30)) * UINT64_C(0xBF58476D1CE4E5B9); + z = (z ^ (z >> 27)) * UINT64_C(0x94D049BB133111EB); + return z ^ (z >> 31); +} + +static float train_random_signed(uint64_t *state) +{ + const uint64_t bits = train_splitmix64(state); + const uint32_t mantissa = (uint32_t)((bits >> 40) & UINT64_C(0xFFFFFF)); + return ((float)mantissa / 8388608.0f) - 1.0f; +} + +static void train_init_matrix(float *matrix, uint32_t rows, uint32_t cols, uint64_t *state) +{ + const float scale = sqrtf(6.0f / ((float)rows + (float)cols)); + size_t count = (size_t)rows * cols; + size_t i; + for (i = 0u; i < count; ++i) matrix[i] = train_random_signed(state) * scale; +} + +void niyah_init_weights(NiyahModel *m, uint64_t seed) +{ + uint64_t state = seed; + uint32_t l; + const uint32_t d = m ? m->cfg.embed_dim : 0u; + + if (!m || !m->_pool) return; + + memset(m->_pool, 0, niyah_param_count(m) * sizeof(float)); + + for (l = 0u; l < m->cfg.n_layers; ++l) { + NiyahLayer *layer = &m->layers[l]; + uint32_t i; + train_init_matrix(layer->wq, d, d, &state); + train_init_matrix(layer->wk, m->kv_dim, d, &state); + train_init_matrix(layer->wv, m->kv_dim, d, &state); + train_init_matrix(layer->wo, d, d, &state); + train_init_matrix(layer->w_gate, m->ffn_dim, d, &state); + train_init_matrix(layer->w_up, m->ffn_dim, d, &state); + train_init_matrix(layer->w_down, d, m->ffn_dim, &state); + for (i = 0u; i < d; ++i) { + layer->rms_att[i] = 1.0f; + layer->rms_ffn[i] = 1.0f; + } + } + + { + const float embed_scale = 1.0f / sqrtf((float)d); + size_t count = (size_t)m->cfg.vocab_size * d; + size_t i; + for (i = 0u; i < count; ++i) { + m->token_embed[i] = train_random_signed(&state) * embed_scale; + } + for (i = 0u; i < d; ++i) m->rms_final[i] = 1.0f; + train_init_matrix(m->lm_head, m->cfg.vocab_size, d, &state); + } + + { + const size_t kv_each = (size_t)m->cfg.n_layers * m->cfg.n_kv_heads + * m->cfg.ctx_len * m->head_dim; + memset(m->kv_k, 0, kv_each * sizeof(float)); + memset(m->kv_v, 0, kv_each * sizeof(float)); + } +} + +static float *train_grad_ptr(float *grad, float *base, float *weight) +{ + return grad + (size_t)(weight - base); +} + +float niyah_full_train_step(NiyahModel *m, NiyahAdam *opt, + const uint32_t *tokens, uint32_t n) +{ + uint32_t t; + uint32_t l; + const uint32_t steps = (n > 0u) ? n - 1u : 0u; + uint32_t d; + uint32_t kd; + uint32_t f; + uint32_t nh; + uint32_t nkv; + uint32_t hd; + uint32_t ctx; + uint32_t vocab; + size_t nw; + float *base; + float *grad = NULL; + float *x_in = NULL; + float *att_norm = NULL; + float *q = NULL; + float *k = NULL; + float *v = NULL; + float *att_ctx = NULL; + float *x1 = NULL; + float *ffn_norm = NULL; + float *gate = NULL; + float *up = NULL; + float *sw = NULL; + float *probs = NULL; + float *xcur = NULL; + float *final_x = NULL; + float *final_norm = NULL; + float *logits = NULL; + float *dlogits = NULL; + float *dx = NULL; + float *d_x1 = NULL; + float *d_xin = NULL; + float *d_ctx = NULL; + float *d_q = NULL; + float *d_k = NULL; + float *d_v = NULL; + float *d_a = NULL; + float *d_b = NULL; + float *d_gate = NULL; + float *d_up = NULL; + float *d_sw = NULL; + float *tmp_d = NULL; + float *dp = NULL; + float loss = 0.0f; + float inv_steps; + int failed = 0; + + if (!m || !opt || !tokens) return NAN; + if (n < 2u) return 0.0f; + if (steps > m->cfg.ctx_len) return NAN; + if (opt->n_weights != niyah_param_count(m)) return NAN; + for (t = 0u; t < n; ++t) { + if (tokens[t] >= m->cfg.vocab_size) return NAN; + } + + d = m->cfg.embed_dim; + kd = m->kv_dim; + f = m->ffn_dim; + nh = m->cfg.n_heads; + nkv = m->cfg.n_kv_heads; + hd = m->head_dim; + ctx = m->cfg.ctx_len; + vocab = m->cfg.vocab_size; + nw = niyah_param_count(m); + base = (float *)m->_pool; + inv_steps = 1.0f / (float)steps; + +#define ALLOC_FLOATS(name, count) \ + do { \ + name = train_calloc((count), sizeof(float)); \ + if (!(name)) failed = 1; \ + } while (0) + + ALLOC_FLOATS(grad, nw); + ALLOC_FLOATS(x_in, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(att_norm, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(q, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(k, (size_t)m->cfg.n_layers * kd); + ALLOC_FLOATS(v, (size_t)m->cfg.n_layers * kd); + ALLOC_FLOATS(att_ctx, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(x1, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(ffn_norm, (size_t)m->cfg.n_layers * d); + ALLOC_FLOATS(gate, (size_t)m->cfg.n_layers * f); + ALLOC_FLOATS(up, (size_t)m->cfg.n_layers * f); + ALLOC_FLOATS(sw, (size_t)m->cfg.n_layers * f); + ALLOC_FLOATS(probs, (size_t)m->cfg.n_layers * nh * ctx); + ALLOC_FLOATS(xcur, d); + ALLOC_FLOATS(final_x, d); + ALLOC_FLOATS(final_norm, d); + ALLOC_FLOATS(logits, vocab); + ALLOC_FLOATS(dlogits, vocab); + ALLOC_FLOATS(dx, d); + ALLOC_FLOATS(d_x1, d); + ALLOC_FLOATS(d_xin, d); + ALLOC_FLOATS(d_ctx, d); + ALLOC_FLOATS(d_q, d); + ALLOC_FLOATS(d_k, kd); + ALLOC_FLOATS(d_v, kd); + ALLOC_FLOATS(d_a, d); + ALLOC_FLOATS(d_b, d); + ALLOC_FLOATS(d_gate, f); + ALLOC_FLOATS(d_up, f); + ALLOC_FLOATS(d_sw, f); + ALLOC_FLOATS(tmp_d, d); + ALLOC_FLOATS(dp, ctx); + +#undef ALLOC_FLOATS + + if (failed) goto cleanup; + + { + const size_t kv_each = (size_t)m->cfg.n_layers * nkv * ctx * hd; + memset(m->kv_k, 0, kv_each * sizeof(float)); + memset(m->kv_v, 0, kv_each * sizeof(float)); + } + + for (t = 0u; t < steps; ++t) { + const uint32_t token = tokens[t]; + const uint32_t target = tokens[t + 1u]; + const float inv_sqrt_hd = 1.0f / sqrtf((float)hd); + float max_logit; + float exp_sum = 0.0f; + float logsumexp; + uint32_t i; + + memcpy(xcur, m->token_embed + (size_t)token * d, d * sizeof(float)); + memset(probs, 0, (size_t)m->cfg.n_layers * nh * ctx * sizeof(float)); + + for (l = 0u; l < m->cfg.n_layers; ++l) { + NiyahLayer *layer = &m->layers[l]; + float *lx = x_in + (size_t)l * d; + float *la = att_norm + (size_t)l * d; + float *lq = q + (size_t)l * d; + float *lk = k + (size_t)l * kd; + float *lv = v + (size_t)l * kd; + float *lc = att_ctx + (size_t)l * d; + float *lx1 = x1 + (size_t)l * d; + float *lb = ffn_norm + (size_t)l * d; + float *lg = gate + (size_t)l * f; + float *lu = up + (size_t)l * f; + float *lsw = sw + (size_t)l * f; + float *layer_probs = probs + (size_t)l * nh * ctx; + const size_t layer_stride = (size_t)nkv * ctx * hd; + float *kc = m->kv_k + (size_t)l * layer_stride; + float *vc = m->kv_v + (size_t)l * layer_stride; + uint32_t h; + + memcpy(lx, xcur, d * sizeof(float)); + train_rmsnorm(la, lx, layer->rms_att, d, m->cfg.rms_eps); + train_matvec(lq, layer->wq, la, d, d); + train_matvec(lk, layer->wk, la, kd, d); + train_matvec(lv, layer->wv, la, kd, d); + + for (h = 0u; h < nh; ++h) train_rope(lq + h * hd, t, hd, m->cfg.rope_theta); + for (h = 0u; h < nkv; ++h) train_rope(lk + h * hd, t, hd, m->cfg.rope_theta); + + for (h = 0u; h < nkv; ++h) { + memcpy(kc + (size_t)h * ctx * hd + (size_t)t * hd, + lk + h * hd, hd * sizeof(float)); + memcpy(vc + (size_t)h * ctx * hd + (size_t)t * hd, + lv + h * hd, hd * sizeof(float)); + } + + memset(lc, 0, d * sizeof(float)); + for (h = 0u; h < nh; ++h) { + const uint32_t kvh = (h * nkv) / nh; + const float *qh = lq + h * hd; + float *ph = layer_probs + (size_t)h * ctx; + float max_score = -INFINITY; + float den = 0.0f; + uint32_t s; + + for (s = 0u; s <= t; ++s) { + const float *ks = kc + (size_t)kvh * ctx * hd + (size_t)s * hd; + const float score = train_dot(qh, ks, hd) * inv_sqrt_hd; + ph[s] = score; + if (score > max_score) max_score = score; + } + for (s = 0u; s <= t; ++s) { + ph[s] = expf(ph[s] - max_score); + den += ph[s]; + } + if (!(den > 0.0f) || !isfinite(den)) { + failed = 1; + goto cleanup; + } + for (s = 0u; s <= t; ++s) { + const float p = ph[s] / den; + const float *vs = vc + (size_t)kvh * ctx * hd + (size_t)s * hd; + uint32_t j; + ph[s] = p; + for (j = 0u; j < hd; ++j) lc[h * hd + j] += p * vs[j]; + } + } + + train_matvec(tmp_d, layer->wo, lc, d, d); + for (i = 0u; i < d; ++i) lx1[i] = lx[i] + tmp_d[i]; + + train_rmsnorm(lb, lx1, layer->rms_ffn, d, m->cfg.rms_eps); + train_matvec(lg, layer->w_gate, lb, f, d); + train_matvec(lu, layer->w_up, lb, f, d); + for (i = 0u; i < f; ++i) lsw[i] = train_silu(lg[i]) * lu[i]; + train_matvec(tmp_d, layer->w_down, lsw, d, f); + for (i = 0u; i < d; ++i) xcur[i] = lx1[i] + tmp_d[i]; + } + + memcpy(final_x, xcur, d * sizeof(float)); + train_rmsnorm(final_norm, final_x, m->rms_final, d, m->cfg.rms_eps); + train_matvec(logits, m->lm_head, final_norm, vocab, d); + + max_logit = logits[0]; + for (i = 1u; i < vocab; ++i) if (logits[i] > max_logit) max_logit = logits[i]; + for (i = 0u; i < vocab; ++i) exp_sum += expf(logits[i] - max_logit); + if (!(exp_sum > 0.0f) || !isfinite(exp_sum)) { + failed = 1; + goto cleanup; + } + logsumexp = logf(exp_sum) + max_logit; + loss += (logsumexp - logits[target]) * inv_steps; + + for (i = 0u; i < vocab; ++i) { + dlogits[i] = expf(logits[i] - logsumexp) * inv_steps; + } + dlogits[target] -= inv_steps; + + { + float *grad_lm = train_grad_ptr(grad, base, m->lm_head); + float *grad_final_rms = train_grad_ptr(grad, base, m->rms_final); + memset(dx, 0, d * sizeof(float)); + train_outer_add(grad_lm, dlogits, final_norm, vocab, d); + train_matvec_t_add(dx, m->lm_head, dlogits, vocab, d); + memset(tmp_d, 0, d * sizeof(float)); + train_rmsnorm_backward(tmp_d, grad_final_rms, final_x, m->rms_final, + dx, d, m->cfg.rms_eps); + memcpy(dx, tmp_d, d * sizeof(float)); + } + + for (l = m->cfg.n_layers; l-- > 0u;) { + NiyahLayer *layer = &m->layers[l]; + const float *lx = x_in + (size_t)l * d; + const float *la = att_norm + (size_t)l * d; + const float *lq = q + (size_t)l * d; + const float *lc = att_ctx + (size_t)l * d; + const float *lx1 = x1 + (size_t)l * d; + const float *lb = ffn_norm + (size_t)l * d; + const float *lg = gate + (size_t)l * f; + const float *lu = up + (size_t)l * f; + const float *lsw = sw + (size_t)l * f; + const float *layer_probs = probs + (size_t)l * nh * ctx; + const size_t layer_stride = (size_t)nkv * ctx * hd; + const float *kc = m->kv_k + (size_t)l * layer_stride; + const float *vc = m->kv_v + (size_t)l * layer_stride; + float *g_wq = train_grad_ptr(grad, base, layer->wq); + float *g_wk = train_grad_ptr(grad, base, layer->wk); + float *g_wv = train_grad_ptr(grad, base, layer->wv); + float *g_wo = train_grad_ptr(grad, base, layer->wo); + float *g_gate = train_grad_ptr(grad, base, layer->w_gate); + float *g_up = train_grad_ptr(grad, base, layer->w_up); + float *g_down = train_grad_ptr(grad, base, layer->w_down); + float *g_rms_att = train_grad_ptr(grad, base, layer->rms_att); + float *g_rms_ffn = train_grad_ptr(grad, base, layer->rms_ffn); + uint32_t h; + + memcpy(d_x1, dx, d * sizeof(float)); + train_outer_add(g_down, dx, lsw, d, f); + memset(d_sw, 0, f * sizeof(float)); + train_matvec_t_add(d_sw, layer->w_down, dx, d, f); + + for (i = 0u; i < f; ++i) { + d_gate[i] = d_sw[i] * lu[i] * train_silu_grad(lg[i]); + d_up[i] = d_sw[i] * train_silu(lg[i]); + } + train_outer_add(g_gate, d_gate, lb, f, d); + train_outer_add(g_up, d_up, lb, f, d); + memset(d_b, 0, d * sizeof(float)); + train_matvec_t_add(d_b, layer->w_gate, d_gate, f, d); + train_matvec_t_add(d_b, layer->w_up, d_up, f, d); + memset(tmp_d, 0, d * sizeof(float)); + train_rmsnorm_backward(tmp_d, g_rms_ffn, lx1, layer->rms_ffn, + d_b, d, m->cfg.rms_eps); + for (i = 0u; i < d; ++i) d_x1[i] += tmp_d[i]; + + memcpy(d_xin, d_x1, d * sizeof(float)); + train_outer_add(g_wo, d_x1, lc, d, d); + memset(d_ctx, 0, d * sizeof(float)); + train_matvec_t_add(d_ctx, layer->wo, d_x1, d, d); + + memset(d_q, 0, d * sizeof(float)); + memset(d_k, 0, kd * sizeof(float)); + memset(d_v, 0, kd * sizeof(float)); + + for (h = 0u; h < nh; ++h) { + const uint32_t kvh = (h * nkv) / nh; + const float *qh = lq + h * hd; + const float *ph = layer_probs + (size_t)h * ctx; + const float *dch = d_ctx + h * hd; + float weighted = 0.0f; + uint32_t s; + + for (s = 0u; s <= t; ++s) { + const float *vs = vc + (size_t)kvh * ctx * hd + (size_t)s * hd; + dp[s] = train_dot(dch, vs, hd); + weighted += ph[s] * dp[s]; + } + for (s = 0u; s <= t; ++s) { + const float ds = ph[s] * (dp[s] - weighted) * inv_sqrt_hd; + const float *ks = kc + (size_t)kvh * ctx * hd + (size_t)s * hd; + uint32_t j; + for (j = 0u; j < hd; ++j) d_q[h * hd + j] += ds * ks[j]; + if (s == t) { + for (j = 0u; j < hd; ++j) d_k[kvh * hd + j] += ds * qh[j]; + } + } + for (i = 0u; i < hd; ++i) { + d_v[kvh * hd + i] += ph[t] * dch[i]; + } + } + + for (h = 0u; h < nh; ++h) { + train_rope_backward(d_q + h * hd, t, hd, m->cfg.rope_theta); + } + for (h = 0u; h < nkv; ++h) { + train_rope_backward(d_k + h * hd, t, hd, m->cfg.rope_theta); + } + + train_outer_add(g_wq, d_q, la, d, d); + train_outer_add(g_wk, d_k, la, kd, d); + train_outer_add(g_wv, d_v, la, kd, d); + memset(d_a, 0, d * sizeof(float)); + train_matvec_t_add(d_a, layer->wq, d_q, d, d); + train_matvec_t_add(d_a, layer->wk, d_k, kd, d); + train_matvec_t_add(d_a, layer->wv, d_v, kd, d); + memset(tmp_d, 0, d * sizeof(float)); + train_rmsnorm_backward(tmp_d, g_rms_att, lx, layer->rms_att, + d_a, d, m->cfg.rms_eps); + for (i = 0u; i < d; ++i) d_xin[i] += tmp_d[i]; + memcpy(dx, d_xin, d * sizeof(float)); + } + + { + float *g_embed = train_grad_ptr(grad, base, m->token_embed) + + (size_t)token * d; + for (i = 0u; i < d; ++i) g_embed[i] += dx[i]; + } + } + + if (!isfinite(loss)) { + failed = 1; + goto cleanup; + } + + { + double sumsq = 0.0; + float clip = 1.0f; + float bc1; + float bc2; + size_t i; + + for (i = 0u; i < nw; ++i) sumsq += (double)grad[i] * (double)grad[i]; + if (!isfinite(sumsq)) { + failed = 1; + goto cleanup; + } + if (sumsq > 1.0) clip = (float)(1.0 / sqrt(sumsq)); + + ++opt->step; + bc1 = 1.0f - powf(opt->beta1, (float)opt->step); + bc2 = 1.0f - powf(opt->beta2, (float)opt->step); + if (!(bc1 > 0.0f) || !(bc2 > 0.0f)) { + failed = 1; + goto cleanup; + } + + for (i = 0u; i < nw; ++i) { + const float g = grad[i] * clip + opt->wd * base[i]; + float mh; + float vh; + opt->m[i] = opt->beta1 * opt->m[i] + (1.0f - opt->beta1) * g; + opt->v[i] = opt->beta2 * opt->v[i] + (1.0f - opt->beta2) * g * g; + mh = opt->m[i] / bc1; + vh = opt->v[i] / bc2; + if (!isfinite(mh) || !isfinite(vh)) { + failed = 1; + goto cleanup; + } + base[i] -= opt->lr * mh / (sqrtf(vh) + opt->eps); + } + } + +cleanup: + free(grad); + free(x_in); + free(att_norm); + free(q); + free(k); + free(v); + free(att_ctx); + free(x1); + free(ffn_norm); + free(gate); + free(up); + free(sw); + free(probs); + free(xcur); + free(final_x); + free(final_norm); + free(logits); + free(dlogits); + free(dx); + free(d_x1); + free(d_xin); + free(d_ctx); + free(d_q); + free(d_k); + free(d_v); + free(d_a); + free(d_b); + free(d_gate); + free(d_up); + free(d_sw); + free(tmp_d); + free(dp); + + return failed ? NAN : loss; +} From a96d5e74a92b714d6dfe56ea5fc58b5da6a429ee Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:35:40 +0300 Subject: [PATCH 3/7] train: use tokenizer vocabulary and full-parameter step --- Core_CPP/niyah_train.c | 258 +++++++++++++++++++++++++++-------------- 1 file changed, 169 insertions(+), 89 deletions(-) diff --git a/Core_CPP/niyah_train.c b/Core_CPP/niyah_train.c index b5365cb..c884c0b 100644 --- a/Core_CPP/niyah_train.c +++ b/Core_CPP/niyah_train.c @@ -1,147 +1,222 @@ #include "niyah_core.h" +#include "niyah_train_full.h" +#include "tokenizer.h" + +#include +#include +#include #include #include #include -#include #include -#include -#include - -void tokenizer_init(void); -uint32_t tokenizer_encode(const char *text, uint32_t *tokens, uint32_t max_len); -void tokenizer_free(void); -static FILE *open_training_data_path(const char *path) { +static FILE *open_training_data_path(const char *path) +{ if (path && path[0]) { FILE *f = fopen(path, "r"); if (f) return f; } - const char *candidates[] = { - "Data_Training/sovereign_knowledge.txt", - "sovereign_knowledge_data.txt", - "sovereign_knowledge.txt" - }; - for (size_t i = 0; i < sizeof(candidates) / sizeof(candidates[0]); ++i) { - FILE *f = fopen(candidates[i], "r"); - if (f) return f; + + { + const char *candidates[] = { + "Data_Training/sovereign_knowledge.txt", + "sovereign_knowledge_data.txt", + "sovereign_knowledge.txt" + }; + size_t i; + for (i = 0u; i < sizeof(candidates) / sizeof(candidates[0]); ++i) { + FILE *f = fopen(candidates[i], "r"); + if (f) return f; + } } return NULL; } -static int parse_int(const char *s, int fallback) { - if (!s || !s[0]) return fallback; +static int parse_int(const char *s, int fallback) +{ char *end = NULL; - long v = strtol(s, &end, 10); - if (end == s || *end != '\0' || v <= 0 || v > INT32_MAX) return fallback; - return (int)v; + long value; + if (!s || !s[0]) return fallback; + errno = 0; + value = strtol(s, &end, 10); + if (errno != 0 || end == s || *end != '\0' || value <= 0 || value > INT32_MAX) { + return fallback; + } + return (int)value; } -static float parse_float(const char *s, float fallback) { +static float parse_float(const char *s, float fallback) +{ + char *end = NULL; + float value; if (!s || !s[0]) return fallback; + errno = 0; + value = strtof(s, &end); + if (errno != 0 || end == s || *end != '\0' || !isfinite(value) || value <= 0.0f) { + return fallback; + } + return value; +} + +static uint64_t parse_u64(const char *s, uint64_t fallback) +{ char *end = NULL; - float v = strtof(s, &end); - if (end == s || *end != '\0' || !isfinite(v) || v <= 0.0f) return fallback; - return v; + unsigned long long value; + if (!s || !s[0]) return fallback; + errno = 0; + value = strtoull(s, &end, 0); + if (errno != 0 || end == s || *end != '\0') return fallback; + return (uint64_t)value; } -static float cosine_lr(float base, float min_lr, uint32_t step, uint32_t total, uint32_t warmup) { +static float cosine_lr(float base, float min_lr, uint32_t step, + uint32_t total, uint32_t warmup) +{ if (step < warmup) { - float t = (float)step / (float)(warmup ? warmup : 1u); - return min_lr + (base - min_lr) * t; + const float p = (float)step / (float)(warmup ? warmup : 1u); + return min_lr + (base - min_lr) * p; } - uint32_t denom = total > warmup ? total - warmup : 1u; - float p = (float)(step - warmup) / (float)denom; - if (p > 1.0f) p = 1.0f; - return min_lr + (base - min_lr) * 0.5f * (1.0f + cosf(3.14159265f * p)); -} -static void clamp_tokens(uint32_t *tokens, uint32_t n, uint32_t vocab_size) { - if (!tokens || vocab_size == 0u) return; - for (uint32_t i = 0; i < n; ++i) tokens[i] %= vocab_size; + { + const uint32_t denom = total > warmup ? total - warmup : 1u; + float p = (float)(step - warmup) / (float)denom; + if (p > 1.0f) p = 1.0f; + return min_lr + (base - min_lr) * 0.5f + * (1.0f + cosf(3.14159265f * p)); + } } -int main(int argc, char **argv) { - NiyahConfig cfg = { - .magic = NIYAH_MAGIC, - .version = NIYAH_VER, - .vocab_size = 8192, - .ctx_len = 64, - .embed_dim = 128, - .n_layers = 4, - .n_heads = 8, - .n_kv_heads = 8, - .ffn_mult = 4, - .rope_theta = 10000.0f, - .rms_eps = 1e-5f, - .flags = 0 - }; - +int main(int argc, char **argv) +{ const char *data_path = argc > 1 ? argv[1] : NULL; - int epochs = argc > 2 ? parse_int(argv[2], 5) : 5; - float base_lr = argc > 3 ? parse_float(argv[3], 3e-4f) : 3e-4f; + const int epochs = argc > 2 ? parse_int(argv[2], 5) : 5; + const float base_lr = argc > 3 ? parse_float(argv[3], 3e-4f) : 3e-4f; float min_lr = argc > 4 ? parse_float(argv[4], 3e-5f) : 3e-5f; + const uint64_t seed = argc > 5 + ? parse_u64(argv[5], UINT64_C(0x4E49594148)) + : UINT64_C(0x4E49594148); + uint32_t tokenizer_vocab; + NiyahConfig cfg; + NiyahModel *model = NULL; + NiyahAdam *opt = NULL; + FILE *data = NULL; + char line[4096]; + uint32_t total_lines = 0u; + uint64_t step_budget; + uint32_t total_steps; + uint32_t warmup_steps; + float ema = 0.0f; + float best_ema = INFINITY; + int bad_windows = 0; + uint32_t global_step = 0u; + clock_t t0; + int rc = 0; + int ep; + if (min_lr > base_lr) min_lr = base_lr * 0.1f; - NiyahModel *model = niyah_alloc(&cfg); - if (!model) { fputs("[NIYAH] alloc failed\n", stderr); return 1; } + tokenizer_init(); + tokenizer_vocab = tokenizer_vocab_size(); + if (tokenizer_vocab == 0u || tokenizer_vocab > NIYAH_MAX_VOCAB) { + fputs("[NIYAH] invalid tokenizer vocabulary\n", stderr); + tokenizer_free(); + return 1; + } + + memset(&cfg, 0, sizeof(cfg)); + cfg.magic = NIYAH_MAGIC; + cfg.version = NIYAH_VER; + cfg.vocab_size = tokenizer_vocab; + cfg.ctx_len = 64u; + cfg.embed_dim = 128u; + cfg.n_layers = 4u; + cfg.n_heads = 8u; + cfg.n_kv_heads = 8u; + cfg.ffn_mult = 4u; + cfg.rope_theta = 10000.0f; + cfg.rms_eps = 1e-5f; - NiyahAdam *opt = niyah_adam_alloc(model); - if (!opt) { fputs("[NIYAH] adam alloc failed\n", stderr); niyah_free(model); return 1; } + model = niyah_alloc(&cfg); + if (!model) { + fputs("[NIYAH] alloc failed\n", stderr); + tokenizer_free(); + return 1; + } + niyah_init_weights(model, seed); + + opt = niyah_adam_alloc(model); + if (!opt) { + fputs("[NIYAH] adam alloc failed\n", stderr); + niyah_free(model); + tokenizer_free(); + return 1; + } opt->lr = base_lr; opt->beta1 = 0.9f; opt->beta2 = 0.999f; opt->eps = 1e-8f; opt->wd = 0.01f; - FILE *data = open_training_data_path(data_path); + data = open_training_data_path(data_path); if (!data) { fputs("[NIYAH] no data file found\n", stderr); - niyah_adam_free(opt); niyah_free(model); return 1; + niyah_adam_free(opt); + niyah_free(model); + tokenizer_free(); + return 1; } - tokenizer_init(); - char line[4096]; - uint32_t total_lines = 0u; while (fgets(line, sizeof(line), data)) { if (strlen(line) > 2u) ++total_lines; } rewind(data); if (total_lines == 0u) { fputs("[NIYAH] no usable lines\n", stderr); - fclose(data); tokenizer_free(); niyah_adam_free(opt); niyah_free(model); return 1; + fclose(data); + niyah_adam_free(opt); + niyah_free(model); + tokenizer_free(); + return 1; } - uint64_t step_budget = (uint64_t)total_lines * (uint64_t)epochs; + step_budget = (uint64_t)total_lines * (uint64_t)epochs; if (step_budget == 0u || step_budget > UINT32_MAX) { fputs("[NIYAH] invalid training step budget\n", stderr); - fclose(data); tokenizer_free(); niyah_adam_free(opt); niyah_free(model); return 1; + fclose(data); + niyah_adam_free(opt); + niyah_free(model); + tokenizer_free(); + return 1; } - uint32_t total_steps = (uint32_t)step_budget; - uint32_t warmup_steps = total_steps / 20u; + total_steps = (uint32_t)step_budget; + warmup_steps = total_steps / 20u; if (warmup_steps < 100u && total_steps > 100u) warmup_steps = 100u; if (warmup_steps >= total_steps && total_steps > 1u) warmup_steps = total_steps - 1u; - float ema = 0.0f; - float best_ema = INFINITY; - int bad_windows = 0; - uint32_t global_step = 0u; - clock_t t0 = clock(); - int rc = 0; + printf("training_mode=full_parameter_detached_kv\n"); + printf("vocab_size=%u\n", cfg.vocab_size); + printf("parameters=%zu\n", niyah_param_count(model)); + printf("seed=%llu\n", (unsigned long long)seed); + fflush(stdout); - for (int ep = 0; ep < epochs; ++ep) { - rewind(data); + t0 = clock(); + + for (ep = 0; ep < epochs; ++ep) { float loss_sum = 0.0f; uint32_t steps = 0u; + rewind(data); while (fgets(line, sizeof(line), data)) { uint32_t tokens[256]; uint32_t n = tokenizer_encode(line, tokens, 256u); + float loss; + if (n < 2u) continue; if (n > cfg.ctx_len + 1u) n = cfg.ctx_len + 1u; - clamp_tokens(tokens, n, cfg.vocab_size); - opt->lr = cosine_lr(base_lr, min_lr, global_step, total_steps, warmup_steps); - float loss = niyah_train_step(model, opt, tokens, n); + opt->lr = cosine_lr(base_lr, min_lr, global_step, + total_steps, warmup_steps); + loss = niyah_full_train_step(model, opt, tokens, n); if (!isfinite(loss)) { fputs("[NIYAH] non-finite loss\n", stderr); rc = 1; @@ -153,16 +228,20 @@ int main(int argc, char **argv) { ++global_step; ema = (ema <= 0.0f) ? loss : 0.995f * ema + 0.005f * loss; - if (steps % 500u == 0u) { + if (steps % 100u == 0u) { printf("ep%d step%u loss=%.4f ema=%.4f lr=%.2e\n", ep + 1, steps, (double)(loss_sum / (float)steps), (double)ema, (double)opt->lr); fflush(stdout); } - if (steps % 2000u == 0u) { - if (ema < best_ema - 1e-3f) { best_ema = ema; bad_windows = 0; } - else if (++bad_windows >= 6) goto cleanup; + if (steps % 1000u == 0u) { + if (ema < best_ema - 1e-3f) { + best_ema = ema; + bad_windows = 0; + } else if (++bad_windows >= 6) { + goto cleanup; + } } } } @@ -170,13 +249,14 @@ int main(int argc, char **argv) { cleanup: fclose(data); tokenizer_free(); - if (rc == 0) { - if (niyah_save(model, "niyah_trained.bin") != 0) { - fputs("[NIYAH] model save failed\n", stderr); - rc = 1; - } + + if (rc == 0 && niyah_save(model, "niyah_trained.bin") != 0) { + fputs("[NIYAH] model save failed\n", stderr); + rc = 1; } - printf("training_elapsed_s=%.3f\n", (double)(clock() - t0) / (double)CLOCKS_PER_SEC); + + printf("training_elapsed_s=%.3f\n", + (double)(clock() - t0) / (double)CLOCKS_PER_SEC); niyah_adam_free(opt); niyah_free(model); return rc; From 8e14b354e39bef2f4098ed22c9e512f656c22e5a Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:36:23 +0300 Subject: [PATCH 4/7] test: prove full-parameter trainer learns and updates backbone --- Core_CPP/niyah_main.c | 305 ++++++++++++++++++++++++++++-------------- 1 file changed, 206 insertions(+), 99 deletions(-) diff --git a/Core_CPP/niyah_main.c b/Core_CPP/niyah_main.c index 7136778..67fe076 100644 --- a/Core_CPP/niyah_main.c +++ b/Core_CPP/niyah_main.c @@ -1,134 +1,241 @@ /* * niyah_main.c — NIYAH engine self-check driver. * - * niyah_smoke() was merged into Niyah.Engine / NiyahKernel and no longer - * exists in this tree, so every target that linked niyah_core.c died at - * `undefined reference to niyah_smoke`. The checks belong to a driver rather - * than to the inference library, so they are inline here and use only the API - * that Core_CPP/niyah_core.c actually defines. - * * Exit code: 0 when every check passes, 1 otherwise. */ #include "niyah_core.h" +#include "niyah_train_full.h" #include #include +#include +#include -static int check(int ok, const char *label) { +static int check(int ok, const char *label) +{ if (!ok) (void)fprintf(stderr, "[niyah] FAIL %s\n", label); return ok ? 0 : 1; } -int main(void) { +static int any_float_changed(const float *before, const float *after, size_t n) +{ + size_t i; + for (i = 0u; i < n; ++i) { + if (before[i] != after[i]) return 1; + } + return 0; +} + +int main(void) +{ int failed = 0; failed += check(sizeof(NiyahConfig) == 64u, "sizeof(NiyahConfig) == 64"); failed += check(niyah_alloc(NULL) == NULL, "alloc rejects a NULL config"); failed += check(niyah_param_count(NULL) == 0u, "param_count(NULL) == 0"); - const NiyahConfig cfg = { - .magic = NIYAH_MAGIC, .version = NIYAH_VER, - .embed_dim = 32u, .n_heads = 4u, .n_kv_heads = 2u, - .n_layers = 2u, .ffn_mult = 2u, .vocab_size = 64u, - .ctx_len = 8u, .rope_theta = 10000.0f, .rms_eps = 1e-5f, - .flags = 0u - }; - - NiyahConfig bad = cfg; - bad.n_kv_heads = cfg.n_heads + 1u; - failed += check(niyah_alloc(&bad) == NULL, "alloc rejects kv_heads > heads"); - bad = cfg; - bad.embed_dim = 30u; - failed += check(niyah_alloc(&bad) == NULL, "alloc rejects embed_dim % n_heads"); - bad = cfg; - bad.ctx_len = NIYAH_MAX_CTX + 1u; - failed += check(niyah_alloc(&bad) == NULL, "alloc rejects ctx_len > NIYAH_MAX_CTX"); - bad = cfg; - bad.vocab_size = 0u; - failed += check(niyah_alloc(&bad) == NULL, "alloc rejects vocab_size 0"); - - NiyahModel *m = niyah_alloc(&cfg); - if (!m) { - (void)fputs("[niyah] FAIL alloc rejected a valid config\n", stderr); - return 1; - } + { + const NiyahConfig cfg = { + .magic = NIYAH_MAGIC, .version = NIYAH_VER, + .embed_dim = 32u, .n_heads = 4u, .n_kv_heads = 2u, + .n_layers = 2u, .ffn_mult = 2u, .vocab_size = 64u, + .ctx_len = 8u, .rope_theta = 10000.0f, .rms_eps = 1e-5f, + .flags = 0u + }; + NiyahConfig bad = cfg; + NiyahModel *m; + + bad.n_kv_heads = cfg.n_heads + 1u; + failed += check(niyah_alloc(&bad) == NULL, "alloc rejects kv_heads > heads"); + bad = cfg; + bad.embed_dim = 30u; + failed += check(niyah_alloc(&bad) == NULL, "alloc rejects embed_dim % n_heads"); + bad = cfg; + bad.ctx_len = NIYAH_MAX_CTX + 1u; + failed += check(niyah_alloc(&bad) == NULL, "alloc rejects ctx_len > NIYAH_MAX_CTX"); + bad = cfg; + bad.vocab_size = 0u; + failed += check(niyah_alloc(&bad) == NULL, "alloc rejects vocab_size 0"); + + m = niyah_alloc(&cfg); + if (!m) { + (void)fputs("[niyah] FAIL alloc rejected a valid config\n", stderr); + return 1; + } - failed += check(niyah_param_count(m) > 0u, "param_count > 0"); - failed += check(m->head_dim == cfg.embed_dim / cfg.n_heads, "head_dim"); - failed += check(m->kv_dim == cfg.n_kv_heads * m->head_dim, "kv_dim"); - failed += check(m->ffn_dim == cfg.embed_dim * cfg.ffn_mult, "ffn_dim"); - - /* Deterministic weights: exercises matvec, rmsnorm, rope and silu. */ - float *w = (float *)m->_pool; - size_t nw = niyah_param_count(m); - for (size_t i = 0; i < nw; i++) w[i] = ((float)(i % 37u) - 18.0f) * 0.005f; - for (uint32_t l = 0; l < cfg.n_layers; l++) - for (uint32_t j = 0; j < cfg.embed_dim; j++) { - m->layers[l].rms_att[j] = 1.0f; - m->layers[l].rms_ffn[j] = 1.0f; + failed += check(niyah_param_count(m) > 0u, "param_count > 0"); + failed += check(m->head_dim == cfg.embed_dim / cfg.n_heads, "head_dim"); + failed += check(m->kv_dim == cfg.n_kv_heads * m->head_dim, "kv_dim"); + failed += check(m->ffn_dim == cfg.embed_dim * cfg.ffn_mult, "ffn_dim"); + + { + float *w = (float *)m->_pool; + size_t nw = niyah_param_count(m); + size_t i; + uint32_t l; + for (i = 0u; i < nw; ++i) w[i] = ((float)(i % 37u) - 18.0f) * 0.005f; + for (l = 0u; l < cfg.n_layers; ++l) { + uint32_t j; + for (j = 0u; j < cfg.embed_dim; ++j) { + m->layers[l].rms_att[j] = 1.0f; + m->layers[l].rms_ffn[j] = 1.0f; + } + } + for (i = 0u; i < cfg.embed_dim; ++i) m->rms_final[i] = 1.0f; } - for (uint32_t j = 0; j < cfg.embed_dim; j++) m->rms_final[j] = 1.0f; - - failed += check(niyah_forward(NULL, 0u, 0u) == NULL, "forward rejects a NULL model"); - failed += check(niyah_forward(m, cfg.vocab_size, 0u) == NULL, "forward rejects token >= vocab_size"); - failed += check(niyah_forward(m, 0u, cfg.ctx_len) == NULL, "forward rejects pos >= ctx_len"); - - const float *logits = niyah_forward(m, 1u, 0u); - failed += check(logits != NULL, "forward returns logits"); - if (logits) { - int finite = 1; - for (uint32_t i = 0; i < cfg.vocab_size; i++) - if (!isfinite(logits[i])) { finite = 0; break; } - failed += check(finite, "every logit is finite"); - } - float probe[4] = { 0.5f, 2.5f, -1.0f, 1.0f }; - NiyahSampler greedy = { .temperature = 0.0f, .top_p = 1.0f, .seed = 1u }; - failed += check(niyah_sample(probe, 4u, &greedy) == 1u, "temperature 0 is argmax"); - NiyahSampler warm = { .temperature = 0.8f, .top_p = 0.9f, .seed = 42u }; - failed += check(niyah_sample(probe, 4u, &warm) < 4u, "sample stays in range"); - failed += check(niyah_sample(probe, 0u, &warm) == 0u, "vocab_size 0 is rejected"); - failed += check(niyah_sample(NULL, 4u, &warm) == 0u, "NULL logits are rejected"); - - NiyahAdam *opt = niyah_adam_alloc(m); - failed += check(opt != NULL, "adam alloc"); - if (opt) { - const uint32_t toks[4] = { 1u, 2u, 3u, 4u }; - float loss = niyah_train_step(m, opt, toks, 4u); - failed += check(isfinite(loss) && loss >= 0.0f, "train_step loss is finite"); - failed += check(opt->step == 1u, "adam step counter advances"); - failed += check(niyah_train_step(m, opt, toks, 1u) == 0.0f, "train_step rejects n < 2"); - niyah_adam_free(opt); - } + failed += check(niyah_forward(NULL, 0u, 0u) == NULL, "forward rejects a NULL model"); + failed += check(niyah_forward(m, cfg.vocab_size, 0u) == NULL, + "forward rejects token >= vocab_size"); + failed += check(niyah_forward(m, 0u, cfg.ctx_len) == NULL, + "forward rejects pos >= ctx_len"); + + { + const float *logits = niyah_forward(m, 1u, 0u); + failed += check(logits != NULL, "forward returns logits"); + if (logits) { + int finite = 1; + uint32_t i; + for (i = 0u; i < cfg.vocab_size; ++i) { + if (!isfinite(logits[i])) { + finite = 0; + break; + } + } + failed += check(finite, "every logit is finite"); + } + } - const char *path = "niyah_selfcheck.bin"; - if (check(niyah_save(m, path) == 0, "save writes the model") == 0) { - NiyahModel *loaded = NULL; - failed += check(niyah_load(&loaded, path) == 0 && loaded != NULL, "load reads it back"); - if (loaded) { - failed += check(loaded->cfg.embed_dim == cfg.embed_dim && - loaded->cfg.n_layers == cfg.n_layers && - loaded->cfg.vocab_size == cfg.vocab_size, - "round trip preserves the config"); - const float *a = (const float *)m->_pool; - const float *b = (const float *)loaded->_pool; - int same = 1; - for (size_t i = 0; i < nw; i++) - if (a[i] != b[i]) { same = 0; break; } - failed += check(same, "round trip preserves the weights"); - niyah_free(loaded); + { + float probe[4] = { 0.5f, 2.5f, -1.0f, 1.0f }; + NiyahSampler greedy = { .temperature = 0.0f, .top_p = 1.0f, .seed = 1u }; + NiyahSampler warm = { .temperature = 0.8f, .top_p = 0.9f, .seed = 42u }; + failed += check(niyah_sample(probe, 4u, &greedy) == 1u, + "temperature 0 is argmax"); + failed += check(niyah_sample(probe, 4u, &warm) < 4u, + "sample stays in range"); + failed += check(niyah_sample(probe, 0u, &warm) == 0u, + "vocab_size 0 is rejected"); + failed += check(niyah_sample(NULL, 4u, &warm) == 0u, + "NULL logits are rejected"); } - (void)remove(path); - } else { - failed += 1; + + { + NiyahAdam *opt = niyah_adam_alloc(m); + failed += check(opt != NULL, "adam alloc"); + if (opt) { + const uint32_t toks[4] = { 1u, 2u, 3u, 4u }; + float loss = niyah_train_step(m, opt, toks, 4u); + failed += check(isfinite(loss) && loss >= 0.0f, + "legacy output-head train_step loss is finite"); + failed += check(opt->step == 1u, "legacy adam step counter advances"); + failed += check(niyah_train_step(m, opt, toks, 1u) == 0.0f, + "legacy train_step rejects n < 2"); + niyah_adam_free(opt); + } + } + + { + const char *path = "niyah_selfcheck.bin"; + if (check(niyah_save(m, path) == 0, "save writes the model") == 0) { + NiyahModel *loaded = NULL; + failed += check(niyah_load(&loaded, path) == 0 && loaded != NULL, + "load reads it back"); + if (loaded) { + size_t nw = niyah_param_count(m); + const float *a = (const float *)m->_pool; + const float *b = (const float *)loaded->_pool; + int same = 1; + size_t i; + failed += check(loaded->cfg.embed_dim == cfg.embed_dim && + loaded->cfg.n_layers == cfg.n_layers && + loaded->cfg.vocab_size == cfg.vocab_size, + "round trip preserves the config"); + for (i = 0u; i < nw; ++i) { + if (a[i] != b[i]) { + same = 0; + break; + } + } + failed += check(same, "round trip preserves the weights"); + niyah_free(loaded); + } + (void)remove(path); + } else { + failed += 1; + } + } + + niyah_free(m); } - niyah_free(m); + /* + * Full-parameter training regression. The model starts from deterministic + * non-zero weights, repeatedly sees one tiny sequence, must reduce loss, + * and must change a backbone attention matrix rather than only lm_head. + */ + { + const NiyahConfig train_cfg = { + .magic = NIYAH_MAGIC, .version = NIYAH_VER, + .embed_dim = 16u, .n_heads = 4u, .n_kv_heads = 2u, + .n_layers = 1u, .ffn_mult = 2u, .vocab_size = 16u, + .ctx_len = 8u, .rope_theta = 10000.0f, .rms_eps = 1e-5f, + .flags = 0u + }; + const uint32_t seq[6] = { 1u, 2u, 1u, 2u, 1u, 2u }; + const uint32_t invalid_seq[2] = { 1u, 16u }; + NiyahModel *tm = niyah_alloc(&train_cfg); + NiyahAdam *to = NULL; + float *wq_before = NULL; + + failed += check(tm != NULL, "full trainer model alloc"); + if (tm) { + const size_t wq_count = (size_t)train_cfg.embed_dim * train_cfg.embed_dim; + float first_loss = NAN; + float last_loss = NAN; + int iteration; + + niyah_init_weights(tm, UINT64_C(0x434153504552)); + to = niyah_adam_alloc(tm); + failed += check(to != NULL, "full trainer adam alloc"); + wq_before = malloc(wq_count * sizeof(float)); + failed += check(wq_before != NULL, "full trainer snapshot alloc"); + + if (to && wq_before) { + memcpy(wq_before, tm->layers[0].wq, wq_count * sizeof(float)); + to->lr = 0.01f; + to->wd = 0.0f; + + first_loss = niyah_full_train_step(tm, to, seq, 6u); + for (iteration = 0; iteration < 79; ++iteration) { + last_loss = niyah_full_train_step(tm, to, seq, 6u); + if (!isfinite(last_loss)) break; + } + + failed += check(isfinite(first_loss), "full trainer initial loss finite"); + failed += check(isfinite(last_loss), "full trainer final loss finite"); + failed += check(isfinite(first_loss) && isfinite(last_loss) + && last_loss < first_loss, + "full trainer reduces overfit loss"); + failed += check(any_float_changed(wq_before, tm->layers[0].wq, wq_count), + "full trainer updates attention backbone"); + failed += check(to->step == 80u, + "full trainer adam step count"); + failed += check(isnan(niyah_full_train_step(tm, to, invalid_seq, 2u)), + "full trainer rejects token >= vocab_size"); + } + + free(wq_before); + niyah_adam_free(to); + niyah_free(tm); + } + } if (failed == 0) { (void)printf("NIYAH SELF-CHECK PASS simd=%s\n", niyah_simd_name()); return 0; } + (void)printf("NIYAH SELF-CHECK FAIL %d checks\n", failed); return 1; } From 4199687533418cbc1160a813fbc2566fbb411f0f Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:36:53 +0300 Subject: [PATCH 5/7] build: link full trainer into smoke coverage --- scripts/build.sh | 38 ++++++++------------------------------ 1 file changed, 8 insertions(+), 30 deletions(-) diff --git a/scripts/build.sh b/scripts/build.sh index 3916034..2f3b34c 100644 --- a/scripts/build.sh +++ b/scripts/build.sh @@ -2,14 +2,6 @@ # # scripts/build.sh - single build entry point for the Casper/NIYAH C sources. # -# Replaces the two divergent scripts that used to exist: -# Core_CPP/build_gcc.sh had -std=c11 and -I include, but only built niyah -# and looked for bench_niyah.c in a bench/ directory -# that does not exist in this repository. -# scripts/build_gcc.sh built every target, but with no -std=, no -I, and -# wrote artifacts straight into Core_CPP/, which is -# how compiled binaries kept landing in git. -# # Usage: # bash scripts/build.sh # release # bash scripts/build.sh --debug # -O0 -g3, ASan+UBSan when available @@ -45,13 +37,12 @@ while [[ $# -gt 0 ]]; do --lint) RUN_LINT=1 ;; --smoke) RUN_SMOKE=1 ;; --bench) RUN_BENCH=1 ;; - -h|--help) sed -n '3,24p' "${BASH_SOURCE[0]}"; exit 0 ;; + -h|--help) sed -n '3,12p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "[build] unknown flag: $1" >&2; exit 2 ;; esac shift done -# ---- compiler ------------------------------------------------------------- if [[ -n "$CC_OVERRIDE" ]]; then CC="$CC_OVERRIDE" elif [[ -n "${CC:-}" ]]; then @@ -66,21 +57,18 @@ else fi command -v "$CC" >/dev/null 2>&1 || { echo "[build] compiler unavailable: $CC" >&2; exit 1; } -# ---- architecture --------------------------------------------------------- if [[ -n "$FORCE_ARCH" ]]; then ARCH="$FORCE_ARCH" else case "$(uname -m)" in - x86_64) ARCH="x86_64" ;; + x86_64) ARCH="x86_64" ;; aarch64|arm64) ARCH="arm64" ;; - *) ARCH="generic" ;; + *) ARCH="generic" ;; esac fi case "$ARCH" in x86_64) - # -march=native makes the binary unusable on older CPUs. Use - # --arch generic for anything you intend to distribute. if echo "" | "$CC" -x c -mavx2 -mfma -E - >/dev/null 2>&1; then ARCH_FLAGS="-mavx2 -mfma -march=native" ARCH_NAME="x86_64 AVX2+FMA (native)" @@ -94,7 +82,6 @@ case "$ARCH" in *) echo "[build] unsupported arch: $ARCH" >&2; exit 2 ;; esac -# ---- flags --------------------------------------------------------------- STD="-std=c11" WARN="-Wall -Wextra -Werror -Wstrict-prototypes -Wmissing-prototypes" WARN="$WARN -Wcast-align -Wwrite-strings -Wshadow -pedantic" @@ -106,9 +93,6 @@ if [[ "$CONFIG" == "release" ]]; then else OPT="-O0 -g3 -DDEBUG" LDFLAGS="-lm" - # The old probe compiled AND linked an empty translation unit, so it - # always failed on "undefined reference to main" and ASan was never - # enabled. Compile only (-c) to test flag support. if echo 'int main(void){return 0;}' \ | "$CC" -fsanitize=address,undefined -x c -c - -o /dev/null >/dev/null 2>&1; then SAN="-fsanitize=address,undefined -fno-omit-frame-pointer" @@ -131,7 +115,6 @@ echo "=============================================" mkdir -p "$BUILD" -# ---- cppcheck gate ------------------------------------------------------- if [[ "$RUN_LINT" == "1" ]]; then command -v cppcheck >/dev/null 2>&1 || { echo "[build] cppcheck not found - apt install cppcheck" >&2; exit 1; } @@ -145,7 +128,8 @@ if [[ "$RUN_LINT" == "1" ]]; then -I "$INCLUDE" -I "$CORE" \ "$ROOT/tokenizer.c" \ "$CORE/niyah_core.c" "$CORE/niyah_main.c" \ - "$CORE/niyah_train.c" "$CORE/niyah_hybrid_main.c" \ + "$CORE/niyah_train.c" "$CORE/niyah_train_full.c" \ + "$CORE/niyah_hybrid_main.c" \ "$CORE/casper_cli.c" "$CORE/casper_rag.c" \ "$CORE/rule_parser.c" "$CORE/proof_generator.c" \ "$CORE/constraint_solver.c" "$CORE/hybrid_reasoner.c" \ @@ -153,7 +137,6 @@ if [[ "$RUN_LINT" == "1" ]]; then echo " cppcheck: clean" fi -# ---- build --------------------------------------------------------------- build_target() { local out="$BUILD/$1"; shift printf '%-16s' "$(basename "$out")" @@ -166,10 +149,11 @@ build_target() { echo "-- compile ----------------------------------" build_target niyah \ - "$CORE/niyah_core.c" "$CORE/niyah_main.c" + "$CORE/niyah_core.c" "$CORE/niyah_train_full.c" "$CORE/niyah_main.c" build_target trainer \ - "$CORE/niyah_train.c" "$CORE/niyah_core.c" "$ROOT/tokenizer.c" + "$CORE/niyah_train.c" "$CORE/niyah_train_full.c" \ + "$CORE/niyah_core.c" "$ROOT/tokenizer.c" build_target niyah_hybrid \ "$CORE/niyah_hybrid_main.c" "$CORE/niyah_core.c" \ @@ -177,9 +161,6 @@ build_target niyah_hybrid \ "$CORE/rule_parser.c" "$CORE/proof_generator.c" \ "$CORE/khz_q_svd.c" "$CORE/casper_rag.c" "$ROOT/tokenizer.c" -# casper_cli.c includes casper_rag.h, rule_parser.h and proof_generator.h -# only, and calls no tokenizer function, so tokenizer.c is deliberately not -# linked here. build_target casper \ "$CORE/casper_cli.c" "$CORE/casper_rag.c" "$CORE/rule_parser.c" \ "$CORE/proof_generator.c" "$CORE/khz_q_svd.c" @@ -188,14 +169,12 @@ if [[ -f "$CORE/bench_niyah.c" ]]; then build_target bench_niyah "$CORE/bench_niyah.c" "$CORE/niyah_core.c" fi -# Tokenizer round-trip test: fails the build if any character decodes to '?'. printf '%-16s' "tokenizer_test" # shellcheck disable=SC2086 "$CC" $CFLAGS -DTOKENIZER_TEST "$ROOT/tokenizer.c" -o "$BUILD/tokenizer_test" $LDFLAGS printf ' ok\n' "$BUILD/tokenizer_test" -# ---- optional runs ------------------------------------------------------- if [[ "$RUN_SMOKE" == "1" ]]; then echo "-- smoke ------------------------------------" "$BUILD/niyah" @@ -207,7 +186,6 @@ if [[ "$RUN_BENCH" == "1" && -x "$BUILD/bench_niyah" ]]; then "$BUILD/bench_niyah" fi -# ---- manifest ------------------------------------------------------------ echo "-- artifacts --------------------------------" for art in "$BUILD/niyah" "$BUILD/trainer" "$BUILD/niyah_hybrid" "$BUILD/casper"; do [[ -s "$art" ]] || { echo "[build] required artifact missing: $art" >&2; exit 1; } From 0c0a2a21c62f3bca4e4c401277cb16d5cfc46f1a Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:37:14 +0300 Subject: [PATCH 6/7] docs: describe actual trainer gradient boundary --- README.md | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 90afa13..836afbe 100644 --- a/README.md +++ b/README.md @@ -57,7 +57,8 @@ The build uses C11 plus warnings-as-errors. `--arch generic` avoids host-specifi | Path | Implemented role | |---|---| -| `Core_CPP/niyah_core.c` | model allocation, forward computation, sampling, Adam step, save/load | +| `Core_CPP/niyah_core.c` | model allocation, inference, sampling, persistence, legacy output-head adaptation | +| `Core_CPP/niyah_train_full.c` | deterministic initialization and full-parameter truncated-BPTT training | | `Core_CPP/niyah_train.c` | training executable | | `Core_CPP/hybrid_reasoner.c` | terms, unification, clause solving | | `Core_CPP/constraint_solver.c` | rational constraints and propagation | @@ -109,7 +110,15 @@ Current supported entry points include: ./build/trainer Data_Training/sovereign_knowledge.txt 3 0.001 0.0001 ``` -The trainer writes `niyah_trained.bin` on a successful run. +The trainer derives `vocab_size` from the live tokenizer, applies deterministic non-zero initialization, and updates token embeddings, all attention and FFN projections, RMSNorm scales, and the LM head. A successful run writes `niyah_trained.bin`. + +The training algorithm uses a deliberate detached-KV boundary: each position backpropagates through its current Q/K/V path, but future losses do not propagate into earlier cached K/V states. This is full-parameter truncated backpropagation, not exact full-sequence BPTT. The C self-check includes an overfit regression that requires loss reduction and a change in an attention backbone matrix. + +An optional fifth argument supplies the deterministic initialization seed: + +```bash +./build/trainer Data_Training/sovereign_knowledge.txt 3 0.001 0.0001 0x434153504552 +``` ## Proof Verification @@ -127,4 +136,4 @@ Constraint values use integer numerator/denominator representation. Where availa ## CI -GitHub Actions builds and smokes the C runtime with GCC and Clang, checks Node.js source syntax, and builds the WPF UI on Windows. CI is the repository-level evidence for buildability; documentation claims are not treated as implementation evidence. +GitHub Actions builds and smokes the C runtime with GCC and Clang, checks Node.js source syntax, and builds the WPF UI on Windows. The C smoke path runs the trainer overfit/backbone-update regression. CI is the repository-level evidence for buildability; documentation claims are not treated as implementation evidence. From f96e3c3a611fa1bef1429fa75ea0d110b9cd4747 Mon Sep 17 00:00:00 2001 From: SULIMAN ALSHAMMARI <239743497+Grar00t@users.noreply.github.com> Date: Mon, 7 Sep 2026 12:37:24 +0300 Subject: [PATCH 7/7] ci: exercise full trainer in release and sanitizer smoke --- .github/workflows/ci.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6d5c2b1..3b484e0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,6 +22,12 @@ jobs: run: | bash scripts/build.sh --arch generic --smoke ./build/casper --self-check + - name: Debug sanitizer smoke + env: + CC: ${{ matrix.compiler }} + run: | + bash scripts/build.sh --debug --arch generic --smoke + ./build/casper --self-check node-runtime: runs-on: ubuntu-latest