67. ML Library Docs
Model Training API
The Braid LLM API provides a complete C and C++ interface for creating, training, and deploying transformer language models. All tensor operations are backed by the runtime tensor library with SIMD acceleration on CPU.
LLMConfig
typedef struct {
int64_t d_model; // hidden dimension (e.g., 512)
int64_t d_ff; // feed-forward dimension (e.g., 1024)
int64_t num_layers; // number of transformer layers (e.g., 6)
int64_t num_heads; // number of query heads (e.g., 8)
int64_t num_kv_heads; // number of key/value heads for GQA (e.g., 4)
int64_t head_dim; // dimension per head (e.g., 64)
int64_t vocab_size; // vocabulary size (e.g., 50257)
int64_t max_seq_len; // maximum sequence length (e.g., 2048)
double learning_rate; // initial learning rate (e.g., 3e-4)
double weight_decay; // weight decay (e.g., 0.1)
int use_ternary; // enable ternary quantization
} LLMConfig;The default config macro provides a 13M-parameter baseline:
LLMConfig cfg = LLM_CONFIG_DEFAULT;
// Equivalent to: d_model=512, d_ff=1024, layers=6,
// heads=8, kv_heads=4, head_dim=64, vocab=64,
// max_seq=2048, lr=3e-4, wd=0.1, ternary=0C API
Model
LLMModel* llm_model_create(const LLMConfig* config);
void llm_model_free(LLMModel* model);
LLMConfig llm_model_config(const LLMModel* model);
int llm_model_save(const LLMModel* model, const char* path);
LLMModel* llm_model_load(const char* path);Forward & Generation
// Forward pass: returns logits tensor
LLMTensor* llm_forward(LLMModel* model,
const int64_t* tokens, int64_t seq_len);
// Autoregressive generation with temperature & top-k sampling
int64_t llm_generate(LLMModel* model,
const int64_t* prompt, int64_t prompt_len,
int64_t* output, int64_t max_tokens,
double temperature);Trainer
LLMTrainer* llm_trainer_create(LLMModel* model, const LLMConfig* config);
void llm_trainer_free(LLMTrainer* trainer);
double llm_train_step(LLMTrainer* trainer,
const int64_t* input_ids, const int64_t* labels,
int64_t batch_size, int64_t seq_len);Tokenizer
LLMTokenizer* llm_tokenizer_create(int64_t vocab_size);
void llm_tokenizer_free(LLMTokenizer* tokenizer);
int64_t llm_tokenizer_encode(LLMTokenizer* tok,
const char* text, int64_t* output, int64_t max_tokens);
char* llm_tokenizer_decode(LLMTokenizer* tok,
const int64_t* tokens, int64_t num_tokens);Checkpoint Format
Model checkpoints use the binary format with the "BRAIDMODL" magic header. llm_model_save writes all model weights, configuration, and optimizer state. llm_model_load reconstructs the model from disk — this enables resuming training across restarts.
C++ RAII API
The header llm.hpp provides move-only RAII wrappers in the braid namespace:
braid::Model
braid::Model model(cfg);
LLMConfig c = model.config();
model.save("checkpoint.bin");
auto loaded = braid::Model::load("checkpoint.bin");
auto logits = model.forward({1, 5, 23, 42}); // vector<int64_t>
auto tokens = model.generate({1}, 16, 1.0); // temperature=1.0braid::Trainer
braid::Trainer trainer(model, cfg);
double loss = trainer.step(input_ids, labels);
auto losses = trainer.epoch(input_ids, labels, 100);braid::Tokenizer
braid::Tokenizer tokenizer(vocab_size);
auto ids = tokenizer.encode("Hello world");
auto text = tokenizer.decode(ids);Full Training Example (C++)
#include "llm.hpp"
#include <iostream>
#include <vector>
int main() {
// 1. Configuration
LLMConfig cfg;
cfg.d_model = 512; cfg.d_ff = 1024;
cfg.num_layers = 6; cfg.num_heads = 8;
cfg.num_kv_heads = 4; cfg.head_dim = 64;
cfg.vocab_size = 64; cfg.max_seq_len = 2048;
cfg.learning_rate = 3e-4; cfg.weight_decay = 0.1;
cfg.use_ternary = 0;
// 2. Model creation
braid::Model model(cfg);
// 3. Resume from checkpoint if available
int start_epoch = 0;
// ... (check for checkpoint_epoch_N.bin)
if (start_epoch > 0) {
char path[64];
snprintf(path, 64, "checkpoint_epoch_%d.bin", start_epoch);
model = braid::Model::load(path);
}
// 4. Trainer
braid::Trainer trainer(model, cfg);
// 5. Training loop
int epochs = 10, steps_per_epoch = 20;
int global_step = start_epoch * steps_per_epoch;
for (int ep = start_epoch; ep < epochs; ep++) {
double epoch_loss = 0.0;
for (int step = 0; step < steps_per_epoch; step++) {
auto input_ids = make_batch(seq_len, cfg.vocab_size, global_step);
auto labels = make_labels(input_ids, cfg.vocab_size);
double loss = trainer.step(input_ids, labels);
epoch_loss += loss;
global_step++;
}
printf("epoch %d avg loss: %.6f\n", ep + 1, epoch_loss / steps_per_epoch);
// 6. Checkpoint
char save_path[64];
snprintf(save_path, 64, "checkpoint_epoch_%d.bin", ep + 1);
model.save(save_path);
}
// 7. Generation
auto output = model.generate({1}, 16, 1.0);
for (auto t : output) printf("%lld ", (long long)t);
printf("\n");
return 0;
}Braid Script Training
Training can also be orchestrated from Braid scripts using FFI bindings:
native fn llm_model_create(d_model, d_ff, num_layers, vocab_size) -> int
native fn llm_train_step(model, input_ids, labels, batch, seq) -> float
native fn llm_generate(model, prompt, prompt_len, output, max_tokens, temp) -> int
fn main() {
let model = llm_model_create(512, 1024, 6, 64)
let epoch = 0
while epoch < 3 {
let step = 0
while step < 10 {
let loss = llm_train_step(model, 0, 0, 1, 8)
print(loss)
step = step + 1
}
epoch = epoch + 1
}
llm_model_free(model)
}