BRAIDGROUP
RESEARCH & DEV
67. ML Library Docs

Model Training API

The Braid LLM API provides a complete C and C++ interface for creating, training, and deploying transformer language models. All tensor operations are backed by the runtime tensor library with SIMD acceleration on CPU.

LLMConfig

typedef struct {
    int64_t d_model;          // hidden dimension (e.g., 512)
    int64_t d_ff;             // feed-forward dimension (e.g., 1024)
    int64_t num_layers;       // number of transformer layers (e.g., 6)
    int64_t num_heads;        // number of query heads (e.g., 8)
    int64_t num_kv_heads;     // number of key/value heads for GQA (e.g., 4)
    int64_t head_dim;         // dimension per head (e.g., 64)
    int64_t vocab_size;       // vocabulary size (e.g., 50257)
    int64_t max_seq_len;      // maximum sequence length (e.g., 2048)
    double learning_rate;     // initial learning rate (e.g., 3e-4)
    double weight_decay;      // weight decay (e.g., 0.1)
    int use_ternary;          // enable ternary quantization
} LLMConfig;

The default config macro provides a 13M-parameter baseline:

LLMConfig cfg = LLM_CONFIG_DEFAULT;
// Equivalent to: d_model=512, d_ff=1024, layers=6,
// heads=8, kv_heads=4, head_dim=64, vocab=64,
// max_seq=2048, lr=3e-4, wd=0.1, ternary=0

C API

Model

LLMModel* llm_model_create(const LLMConfig* config);
void      llm_model_free(LLMModel* model);
LLMConfig llm_model_config(const LLMModel* model);
int       llm_model_save(const LLMModel* model, const char* path);
LLMModel* llm_model_load(const char* path);

Forward & Generation

// Forward pass: returns logits tensor
LLMTensor* llm_forward(LLMModel* model,
    const int64_t* tokens, int64_t seq_len);

// Autoregressive generation with temperature & top-k sampling
int64_t llm_generate(LLMModel* model,
    const int64_t* prompt, int64_t prompt_len,
    int64_t* output, int64_t max_tokens,
    double temperature);

Trainer

LLMTrainer* llm_trainer_create(LLMModel* model, const LLMConfig* config);
void        llm_trainer_free(LLMTrainer* trainer);
double      llm_train_step(LLMTrainer* trainer,
    const int64_t* input_ids, const int64_t* labels,
    int64_t batch_size, int64_t seq_len);

Tokenizer

LLMTokenizer* llm_tokenizer_create(int64_t vocab_size);
void          llm_tokenizer_free(LLMTokenizer* tokenizer);
int64_t       llm_tokenizer_encode(LLMTokenizer* tok,
    const char* text, int64_t* output, int64_t max_tokens);
char*         llm_tokenizer_decode(LLMTokenizer* tok,
    const int64_t* tokens, int64_t num_tokens);

Checkpoint Format

Model checkpoints use the binary format with the "BRAIDMODL" magic header. llm_model_save writes all model weights, configuration, and optimizer state. llm_model_load reconstructs the model from disk — this enables resuming training across restarts.

C++ RAII API

The header llm.hpp provides move-only RAII wrappers in the braid namespace:

braid::Model

braid::Model model(cfg);
LLMConfig c = model.config();
model.save("checkpoint.bin");
auto loaded = braid::Model::load("checkpoint.bin");
auto logits = model.forward({1, 5, 23, 42});  // vector<int64_t>
auto tokens = model.generate({1}, 16, 1.0);   // temperature=1.0

braid::Trainer

braid::Trainer trainer(model, cfg);
double loss = trainer.step(input_ids, labels);
auto losses = trainer.epoch(input_ids, labels, 100);

braid::Tokenizer

braid::Tokenizer tokenizer(vocab_size);
auto ids = tokenizer.encode("Hello world");
auto text = tokenizer.decode(ids);

Full Training Example (C++)

#include "llm.hpp"
#include <iostream>
#include <vector>

int main() {
    // 1. Configuration
    LLMConfig cfg;
    cfg.d_model = 512;  cfg.d_ff = 1024;
    cfg.num_layers = 6; cfg.num_heads = 8;
    cfg.num_kv_heads = 4; cfg.head_dim = 64;
    cfg.vocab_size = 64; cfg.max_seq_len = 2048;
    cfg.learning_rate = 3e-4; cfg.weight_decay = 0.1;
    cfg.use_ternary = 0;

    // 2. Model creation
    braid::Model model(cfg);

    // 3. Resume from checkpoint if available
    int start_epoch = 0;
    // ... (check for checkpoint_epoch_N.bin)
    if (start_epoch > 0) {
        char path[64];
        snprintf(path, 64, "checkpoint_epoch_%d.bin", start_epoch);
        model = braid::Model::load(path);
    }

    // 4. Trainer
    braid::Trainer trainer(model, cfg);

    // 5. Training loop
    int epochs = 10, steps_per_epoch = 20;
    int global_step = start_epoch * steps_per_epoch;

    for (int ep = start_epoch; ep < epochs; ep++) {
        double epoch_loss = 0.0;
        for (int step = 0; step < steps_per_epoch; step++) {
            auto input_ids = make_batch(seq_len, cfg.vocab_size, global_step);
            auto labels = make_labels(input_ids, cfg.vocab_size);
            double loss = trainer.step(input_ids, labels);
            epoch_loss += loss;
            global_step++;
        }
        printf("epoch %d avg loss: %.6f\n", ep + 1, epoch_loss / steps_per_epoch);

        // 6. Checkpoint
        char save_path[64];
        snprintf(save_path, 64, "checkpoint_epoch_%d.bin", ep + 1);
        model.save(save_path);
    }

    // 7. Generation
    auto output = model.generate({1}, 16, 1.0);
    for (auto t : output) printf("%lld ", (long long)t);
    printf("\n");
    return 0;
}

Braid Script Training

Training can also be orchestrated from Braid scripts using FFI bindings:

native fn llm_model_create(d_model, d_ff, num_layers, vocab_size) -> int
native fn llm_train_step(model, input_ids, labels, batch, seq) -> float
native fn llm_generate(model, prompt, prompt_len, output, max_tokens, temp) -> int

fn main() {
    let model = llm_model_create(512, 1024, 6, 64)
    let epoch = 0
    while epoch < 3 {
        let step = 0
        while step < 10 {
            let loss = llm_train_step(model, 0, 0, 1, 8)
            print(loss)
            step = step + 1
        }
        epoch = epoch + 1
    }
    llm_model_free(model)
}