51 lines
1.3 KiB
C++
51 lines
1.3 KiB
C++
#pragma once
|
|
#include "model.hpp"
|
|
#include "checkpoint.hpp"
|
|
#include <vector>
|
|
|
|
namespace xt {
|
|
|
|
float backward(Params& p, const ForwardCache& fc,
|
|
const int* x, const int* y, int n_pairs);
|
|
|
|
float softmax_eval(const Params& p, const ForwardCache& fc,
|
|
const int* x, const int* y, int n_pairs, Tensor& per_token_loss);
|
|
|
|
struct TrainConfig {
|
|
int steps = 1000;
|
|
int batch_size = 8;
|
|
int block = 64;
|
|
float lr = 3e-4f;
|
|
float beta1 = 0.9f;
|
|
float beta2 = 0.999f;
|
|
float eps = 1e-8f;
|
|
float weight_decay = 0.01f;
|
|
float clip = 1.0f;
|
|
int warmup = 100;
|
|
float lr_min_frac = 0.1f;
|
|
uint64_t seed = 1337;
|
|
int threads = 0;
|
|
int log_every = 50;
|
|
int ckpt_every = 0;
|
|
int val_every = 0;
|
|
int val_tokens = 20000;
|
|
};
|
|
|
|
struct Dataset {
|
|
std::vector<int> ids;
|
|
size_t n() const { return ids.size(); }
|
|
};
|
|
|
|
struct TrainStats {
|
|
float loss = 0;
|
|
float val_loss = 0;
|
|
int step = 0;
|
|
double tokens_seen = 0;
|
|
};
|
|
|
|
TrainStats train_model(Params& p, const Tokenizer& tok, const std::string& corpus_path,
|
|
const TrainConfig& tc, const std::string& out_path,
|
|
int resume_step = 0);
|
|
|
|
}
|