The hardware and bandwidth for this mirror is donated by dogado GmbH, the Webhosting and Full Service-Cloud Provider. Check out our Wordpress Tutorial.
If you wish to report a bug, or if you are interested in having us mirror your free-software or open-source project, please feel free to contact us at mirror[@]dogado.de.

Autograd Engine

ggmlR includes a PyTorch-style dynamic autograd engine built on top of ggml tensors. Gradients are computed via reverse-mode AD on a tape recorded inside with_grad_tape({}).

Tensor layout: all autograd tensors are column-major — shape is [features, batch], matching ggml’s native C layout.

library(ggmlR)

1. Core primitives

ag_tensor and ag_param

# ag_tensor — non-trainable input (e.g. data)
x <- ag_tensor(matrix(1:6 / 6, nrow = 2L))   # [2, 3]

# ag_param — trainable parameter (gradient accumulated)
W <- ag_param(matrix(rnorm(4), nrow = 2L))    # [2, 2]
b <- ag_param(matrix(0.0, 2L, 1L))

Forward pass and gradient tape

with_grad_tape({
  h    <- ag_relu(ag_add(ag_matmul(W, x), b))   # [2, 3]
  loss <- ag_mean(ag_mul(h, h))                  # scalar MSE-like
})

grads <- backward(loss)   # returns named list of gradients

cat("dL/dW:\n"); print(grads[["W"]])
cat("dL/db:\n"); print(grads[["b"]])

backward() returns gradients keyed by the parameter’s variable name.


2. Built-in operations

Function Description
ag_matmul(A, B) Matrix multiply
ag_add(A, B) Element-wise add (broadcast supported)
ag_mul(A, B) Element-wise multiply
ag_relu(x) ReLU activation
ag_sigmoid(x) Sigmoid activation
ag_tanh(x) Tanh activation
ag_softmax(x) Softmax (column-wise)
ag_sum(x) Sum all elements → scalar
ag_mean(x) Mean all elements → scalar
ag_transpose(x) Transpose [m,n] → [n,m]
ag_reshape(x, dims) Reshape tensor
ag_softmax_cross_entropy_loss(logits, y) Fused softmax + cross-entropy

3. ag_sequential — module API

ag_sequential stacks layers into a callable module with .forward() and .parameters().

data(iris)
set.seed(42)

x_all <- t(scale(as.matrix(iris[, 1:4])))        # [4, 150]
y_oh  <- model.matrix(~ Species - 1, iris)
y_all <- t(y_oh)                                  # [3, 150]

idx  <- sample(150L)
x_tr <- x_all[, idx[1:120]];  x_vl <- x_all[, idx[121:150]]
y_tr <- y_all[, idx[1:120]];  y_vl <- y_all[, idx[121:150]]

model <- ag_sequential(
  ag_linear(4L,  64L, activation = "relu"),
  ag_batch_norm(64L),
  ag_dropout(0.3),
  ag_linear(64L, 32L, activation = "relu"),
  ag_linear(32L,  3L)
)

params <- model$parameters()
cat("Parameter tensors:", length(params), "\n")

ag_layer_norm() is the other normalization available. The two differ in what they average over: batch norm normalizes each feature across the batch, so its result depends on the other examples in the batch and it needs running statistics for inference; layer norm normalizes each example across its own features, so it behaves identically in training and evaluation and works with a batch of one. That independence from batch size is why transformers use it.

model_ln <- ag_sequential(
  ag_linear(4L,  64L, activation = "relu"),
  ag_layer_norm(64L),
  ag_linear(64L, 32L, activation = "relu"),
  ag_linear(32L,  3L)
)
cat("LayerNorm model parameters:", length(model_ln$parameters()), "\n")

4. Optimizers

# Adam (default lr = 1e-3)
opt <- optimizer_adam(params, lr = 1e-3)

# SGD with momentum
opt_sgd <- optimizer_sgd(params, lr = 0.05, momentum = 0.9)

Training loop

BS <- 32L
n  <- ncol(x_tr)

ag_train(model)   # set training mode (enables dropout, batch norm train)
set.seed(1)

for (ep in seq_len(150L)) {
  perm <- sample(n)
  for (b in seq_len(ceiling(n / BS))) {
    idx <- perm[seq((b-1L)*BS + 1L, min(b*BS, n))]
    xb  <- ag_tensor(x_tr[, idx, drop = FALSE])
    yb  <- y_tr[, idx, drop = FALSE]

    with_grad_tape({
      loss <- ag_softmax_cross_entropy_loss(model$forward(xb), yb)
    })
    grads <- backward(loss)
    opt$step(grads)
    opt$zero_grad()
  }

  if (ep %% 50L == 0L)
    cat(sprintf("epoch %3d  loss %.4f\n", ep, loss$data[1]))
}

5. LR schedulers

opt2 <- optimizer_adam(params, lr = 1e-3)

# Cosine annealing: lr goes from lr_max to lr_min over T_max epochs
sch_cos <- lr_scheduler_cosine(opt2, T_max = 150L, lr_min = 1e-5)

# Step decay: multiply lr by gamma every step_size epochs
sch_step <- lr_scheduler_step(opt2, step_size = 30L, gamma = 0.5)

# SGDR warm restarts: each cycle T_mult times longer than the last.
# T_mult only takes effect with restart = TRUE.
sch_sgdr <- lr_scheduler_cosine(opt2, T_max = 20L, lr_min = 1e-5,
                                restart = TRUE, T_mult = 2L)

# 1cycle: lr rises then falls, momentum (or Adam beta1) moves inversely
sch_1cyc <- lr_scheduler_onecycle(opt2, max_lr = 1e-2, total_steps = 150L)

# Cyclic: triangular by default, also "triangular2" and "exp_range"
sch_cyc <- lr_scheduler_cyclic(opt2, base_lr = 1e-4, max_lr = 1e-2,
                               step_size_up = 25L)

# Linear warmup, then cosine decay
sch_warm <- lr_scheduler_warmup_cosine(opt2, warmup_steps = 10L,
                                       total_steps = 150L)

# Call after each epoch:
# sch_cos$step()

6. Gradient clipping

with_grad_tape({
  loss <- ag_softmax_cross_entropy_loss(model$forward(ag_tensor(x_tr)), y_tr)
})
grads <- backward(loss)

# Clip global gradient norm to max_norm -- rescales every gradient by one
# factor, so their relative directions are preserved.
clip_grad_norm(params, grads, max_norm = 5.0)

# Element-wise alternative: clamp each value into [-clip_value, clip_value].
# Blunter than norm clipping, but it bounds every element individually.
clip_grad_value(params, grads, clip_value = 1.0)

# Catch NaN/Inf gradients where they appear, rather than after the weights
# have already been poisoned. action = "warn" (default), "stop" or "silent";
# max_abs additionally flags gradients that are merely oversized.
check_grad_anomaly(params, grads, action = "warn")

opt$step(grads)
opt$zero_grad()

7. Dataloader

ag_dataloader shuffles and batches column-major data matrices:

dl <- ag_dataloader(x_tr, y_tr, batch_size = BS, shuffle = TRUE)

ag_train(model)
for (ep in seq_len(100L)) {
  for (batch in dl$epoch()) {
    with_grad_tape({
      loss <- ag_softmax_cross_entropy_loss(model$forward(batch$x), batch$y$data)
    })
    grads <- backward(loss)
    opt$step(grads);  opt$zero_grad()
  }
}

8. Eval mode and inference

ag_eval(model)   # disables dropout, switches batch norm to inference stats

# Forward in chunks to avoid memory pressure
predict_cm <- function(mod, x_cm, chunk = 64L) {
  n   <- ncol(x_cm)
  out <- NULL
  for (s in seq(1L, n, by = chunk)) {
    e  <- min(s + chunk - 1L, n)
    lg <- mod$forward(ag_tensor(x_cm[, s:e, drop = FALSE]))$data
    ev <- exp(lg - apply(lg, 2, max))
    sm <- ev / colSums(ev)
    out <- if (is.null(out)) sm else cbind(out, sm)
  }
  out
}

probs <- predict_cm(model, x_vl)          # [3, 30]
preds <- apply(probs, 2, which.max)
truth <- apply(y_vl, 1, which.max)
cat(sprintf("Val accuracy: %.4f\n", mean(preds == truth)))

9. Raw ag_param — full manual control

For complete flexibility, build the network from scratch:

set.seed(7)
W1 <- ag_param(matrix(rnorm(64*4) * sqrt(2/4),  64, 4))
b1 <- ag_param(matrix(0.0, 64, 1))
W2 <- ag_param(matrix(rnorm(3*64) * sqrt(2/64),  3, 64))
b2 <- ag_param(matrix(0.0,  3, 1))

forward <- function(x)
  ag_add(ag_matmul(W2, ag_relu(ag_add(ag_matmul(W1, x), b1))), b2)

opt_raw <- optimizer_adam(list(W1=W1, b1=b1, W2=W2, b2=b2), lr = 1e-3)

for (ep in seq_len(200L)) {
  perm <- sample(n)
  for (b in seq_len(ceiling(n / BS))) {
    idx <- perm[seq((b-1L)*BS+1L, min(b*BS, n))]
    xb  <- ag_tensor(x_tr[, idx, drop = FALSE])
    yb  <- y_tr[, idx, drop = FALSE]
    with_grad_tape({ loss_r <- ag_softmax_cross_entropy_loss(forward(xb), yb) })
    gr <- backward(loss_r)
    opt_raw$step(gr);  opt_raw$zero_grad()
  }
}

10. Mixed precision

# f16 on GPU, f32 on CPU
device <- tryCatch({ ag_device("gpu"); "gpu" }, error = function(e) "cpu")
ag_dtype(if (device == "gpu") "f16" else "f32")

# All subsequent ag_param / ag_tensor use the selected dtype

See also vignette("gpu-vulkan", package = "ggmlR") for device management.


11. Data-parallel training

For multi-GPU or faster single-GPU training see dp_train():

# Full example: inst/examples/dp_train_demo.R

See also vignette("data-parallel-training", package = "ggmlR").

These binaries (installable software) and packages are in development.
They may not be fully stable and should be used with caution. We make no claims about them.
Health stats visible at Monitor.