136 lines
5.9 KiB
C++
136 lines
5.9 KiB
C++
#include "tokenizer.hpp"
|
|
|
|
#include <algorithm>
|
|
#include <cmath>
|
|
#include <cstring>
|
|
|
|
void proccesing_promt(const std::string& text, Model& mdl, uint32_t layer_idx)
|
|
{
|
|
std::vector<uint32_t> tokens;
|
|
size_t i = 0;
|
|
while (i < text.length()) {
|
|
std::string longest_match = "";
|
|
uint32_t match_id = 0;
|
|
bool found = false;
|
|
for (size_t len = 1; i + len <= text.length(); ++len) {
|
|
std::string sub = text.substr(i, len);
|
|
if (mdl.vocab.count(sub)) {
|
|
if (sub.length() > longest_match.length()) {
|
|
longest_match = sub;
|
|
match_id = mdl.vocab[sub];
|
|
found = true;
|
|
}
|
|
}
|
|
}
|
|
if (found) {
|
|
tokens.push_back(match_id);
|
|
i += longest_match.length();
|
|
} else {
|
|
++i;
|
|
}
|
|
}
|
|
const size_t emb_dim = mdl.config.embedding_dim;
|
|
const size_t n_heads = mdl.config.num_heads;
|
|
const size_t head_dim = mdl.config.head_dim;
|
|
const float* w_qkv_ptr = mdl.w_qkv.data();
|
|
if (mdl.w_emb.empty()) mdl.load_edm();
|
|
if (mdl.w_qkv.empty()) mdl.load_qkv();
|
|
if (!mdl.kv_cache) {
|
|
mdl.allocate_kv_cache();
|
|
}
|
|
for (unsigned int token_pos = 0; token_pos < tokens.size(); ++token_pos) {
|
|
const float* emb_ptr = mdl.w_emb.data()
|
|
+ static_cast<size_t>(emb_dim) * tokens[token_pos];
|
|
std::vector<float> emb(emb_ptr, emb_ptr + emb_dim);
|
|
std::vector<std::vector<float>> Q(n_heads, std::vector<float>(head_dim, 0.0f));
|
|
std::vector<std::vector<float>> K(n_heads, std::vector<float>(head_dim, 0.0f));
|
|
std::vector<std::vector<float>> V(n_heads, std::vector<float>(head_dim, 0.0f));
|
|
for (size_t i_emb = 0; i_emb < emb_dim; ++i_emb) {
|
|
for (size_t idx_qkv = 0; idx_qkv < 3; ++idx_qkv) {
|
|
for (size_t head = 0; head < n_heads; ++head) {
|
|
size_t flat_idx = i_emb * 3 * n_heads
|
|
+ idx_qkv * n_heads
|
|
+ head;
|
|
for (size_t d = 0; d < head_dim; ++d) {
|
|
float val = emb[i_emb] * w_qkv_ptr[flat_idx];
|
|
if (idx_qkv == 0) Q[head][d] += val;
|
|
else if (idx_qkv == 1) K[head][d] += val;
|
|
else V[head][d] += val;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
for (size_t head = 0; head < n_heads; ++head) {
|
|
for (size_t d = 0; d < head_dim; ++d) {
|
|
float* k_ptr = mdl.get_kv(token_pos, layer_idx, K_cache, head, d);
|
|
*k_ptr = K[head][d];
|
|
float* v_ptr = mdl.get_kv(token_pos, layer_idx, V_cache, head, d);
|
|
*v_ptr = V[head][d];
|
|
}
|
|
}
|
|
std::vector<std::vector<float>> attention_output(n_heads, std::vector<float>(head_dim, 0.0f));
|
|
size_t num_prev_tokens = token_pos + 1;
|
|
for (size_t head = 0; head < n_heads; ++head) {
|
|
std::vector<float> scores(num_prev_tokens, 0.0f);
|
|
for (size_t t = 0; t < num_prev_tokens; ++t) {
|
|
float score = 0.0f;
|
|
for (size_t d = 0; d < head_dim; ++d) {
|
|
float* k_ptr = mdl.get_kv(t, layer_idx, K_cache, head, d);
|
|
score += Q[head][d] * (*k_ptr);
|
|
}
|
|
scores[t] = score / std::sqrt(static_cast<float>(head_dim));
|
|
}
|
|
float max_score = *std::max_element(scores.begin(), scores.end());
|
|
float sum_exp = 0.0f;
|
|
std::vector<float> exp_scores(num_prev_tokens, 0.0f);
|
|
for (size_t t = 0; t < num_prev_tokens; ++t) {
|
|
exp_scores[t] = std::exp(scores[t] - max_score);
|
|
sum_exp += exp_scores[t];
|
|
}
|
|
for (size_t t = 0; t < num_prev_tokens; ++t) {
|
|
exp_scores[t] /= sum_exp;
|
|
}
|
|
for (size_t d = 0; d < head_dim; ++d) {
|
|
float weighted_sum = 0.0f;
|
|
for (size_t t = 0; t < num_prev_tokens; ++t) {
|
|
float* v_ptr = mdl.get_kv(t, layer_idx, V_cache, head, d);
|
|
weighted_sum += exp_scores[t] * (*v_ptr);
|
|
}
|
|
attention_output[head][d] = weighted_sum;
|
|
}
|
|
}
|
|
std::cout << "--- Token " << token_pos << " (id " << tokens[token_pos]
|
|
<< "), position in cache: " << mdl.ctx_len << " ---\n";
|
|
std::cout << "Attention output:\n";
|
|
for (size_t head = 0; head < n_heads; ++head) {
|
|
std::cout << " head " << head << ": ";
|
|
for (size_t d = 0; d < head_dim; ++d) {
|
|
std::cout << attention_output[head][d] << ' ';
|
|
}
|
|
std::cout << '\n';
|
|
}
|
|
mdl.ctx_len++;
|
|
}
|
|
}
|
|
|
|
Model create_test_model() {
|
|
Model model;
|
|
model.config.magic = 0x31524642; // "BFR1"
|
|
model.config.vocab_size = TEST_VOCAB_SIZE;
|
|
model.config.embedding_dim = TEST_EMBEDDING_DIM;
|
|
model.config.num_heads = TEST_NUM_HEADS;
|
|
model.config.head_dim = TEST_HEAD_DIM;
|
|
model.config.num_layers = TEST_NUM_LAYERS;
|
|
model.config.max_ctx = TEST_MAX_CTX;
|
|
model.vocab = TEST_VOCAB;
|
|
size_t emb_elements = TEST_VOCAB_SIZE * TEST_EMBEDDING_DIM;
|
|
size_t qkv_elements = TEST_NUM_LAYERS * TEST_EMBEDDING_DIM * 3 * TEST_NUM_HEADS;
|
|
size_t o_elements = TEST_NUM_LAYERS * (TEST_NUM_HEADS * TEST_HEAD_DIM) * TEST_EMBEDDING_DIM;
|
|
const float* emb_ptr = reinterpret_cast<const float*>(TEST_W_EMB);
|
|
model.w_emb = std::vector<float>(emb_ptr, emb_ptr + emb_elements);
|
|
const float* qkv_ptr = reinterpret_cast<const float*>(TEST_W_QKV);
|
|
model.w_qkv = std::vector<float>(qkv_ptr, qkv_ptr + qkv_elements);
|
|
const float* o_ptr = reinterpret_cast<const float*>(TEST_W_O);
|
|
model.w_o = std::vector<float>(o_ptr, o_ptr + o_elements);
|
|
return model;
|
|
} |