#include "tokenizer.h" #include #include #include void proccesing_promt(const std::string& text, Model& mdl, uint32_t layer_idx) { std::vector tokens; size_t i = 0; while (i < text.length()) { std::string longest_match = ""; uint32_t match_id = 0; bool found = false; for (size_t len = 1; i + len <= text.length(); ++len) { std::string sub = text.substr(i, len); if (mdl.vocab.count(sub)) { if (sub.length() > longest_match.length()) { longest_match = sub; match_id = mdl.vocab[sub]; found = true; } } } if (found) { tokens.push_back(match_id); i += longest_match.length(); } else { ++i; } } const size_t emb_dim = mdl.config.embedding_dim; const size_t n_heads = mdl.config.num_heads; const size_t head_dim = mdl.config.head_dim; const float* w_qkv_ptr = mdl.w_qkv.data(); if (mdl.w_emb.empty()) mdl.load_edm(); if (mdl.w_qkv.empty()) mdl.load_qkv(); if (!mdl.kv_cache) { mdl.allocate_kv_cache(); } for (unsigned int token_pos = 0; token_pos < tokens.size(); ++token_pos) { const float* emb_ptr = mdl.w_emb.data() + static_cast(emb_dim) * tokens[token_pos]; std::vector emb(emb_ptr, emb_ptr + emb_dim); std::vector> Q(n_heads, std::vector(head_dim, 0.0f)); std::vector> K(n_heads, std::vector(head_dim, 0.0f)); std::vector> V(n_heads, std::vector(head_dim, 0.0f)); for (size_t i_emb = 0; i_emb < emb_dim; ++i_emb) { for (size_t idx_qkv = 0; idx_qkv < 3; ++idx_qkv) { for (size_t head = 0; head < n_heads; ++head) { size_t flat_idx = i_emb * 3 * n_heads + idx_qkv * n_heads + head; for (size_t d = 0; d < head_dim; ++d) { float val = emb[i_emb] * w_qkv_ptr[flat_idx]; if (idx_qkv == 0) Q[head][d] += val; else if (idx_qkv == 1) K[head][d] += val; else V[head][d] += val; } } } } for (size_t head = 0; head < n_heads; ++head) { for (size_t d = 0; d < head_dim; ++d) { float* k_ptr = mdl.get_kv(token_pos, layer_idx, K_cache, head, d); *k_ptr = K[head][d]; float* v_ptr = mdl.get_kv(token_pos, layer_idx, V_cache, head, d); *v_ptr = V[head][d]; } } std::vector> attention_output(n_heads, std::vector(head_dim, 0.0f)); size_t num_prev_tokens = token_pos + 1; for (size_t head = 0; head < n_heads; ++head) { std::vector scores(num_prev_tokens, 0.0f); for (size_t t = 0; t < num_prev_tokens; ++t) { float score = 0.0f; for (size_t d = 0; d < head_dim; ++d) { float* k_ptr = mdl.get_kv(t, layer_idx, K_cache, head, d); score += Q[head][d] * (*k_ptr); } scores[t] = score / std::sqrt(static_cast(head_dim)); } float max_score = *std::max_element(scores.begin(), scores.end()); float sum_exp = 0.0f; std::vector exp_scores(num_prev_tokens, 0.0f); for (size_t t = 0; t < num_prev_tokens; ++t) { exp_scores[t] = std::exp(scores[t] - max_score); sum_exp += exp_scores[t]; } for (size_t t = 0; t < num_prev_tokens; ++t) { exp_scores[t] /= sum_exp; } for (size_t d = 0; d < head_dim; ++d) { float weighted_sum = 0.0f; for (size_t t = 0; t < num_prev_tokens; ++t) { float* v_ptr = mdl.get_kv(t, layer_idx, V_cache, head, d); weighted_sum += exp_scores[t] * (*v_ptr); } attention_output[head][d] = weighted_sum; } } std::cout << "--- Token " << token_pos << " (id " << tokens[token_pos] << "), position in cache: " << mdl.ctx_len << " ---\n"; std::cout << "Attention output:\n"; for (size_t head = 0; head < n_heads; ++head) { std::cout << " head " << head << ": "; for (size_t d = 0; d < head_dim; ++d) { std::cout << attention_output[head][d] << ' '; } std::cout << '\n'; } mdl.ctx_len++; } } Model create_test_model() { Model model; model.config.magic = 0x31524642; // "BFR1" model.config.vocab_size = TEST_VOCAB_SIZE; model.config.embedding_dim = TEST_EMBEDDING_DIM; model.config.num_heads = TEST_NUM_HEADS; model.config.head_dim = TEST_HEAD_DIM; model.config.num_layers = TEST_NUM_LAYERS; model.config.max_ctx = TEST_MAX_CTX; model.vocab = TEST_VOCAB; size_t emb_elements = TEST_VOCAB_SIZE * TEST_EMBEDDING_DIM; size_t qkv_elements = TEST_NUM_LAYERS * TEST_EMBEDDING_DIM * 3 * TEST_NUM_HEADS; size_t o_elements = TEST_NUM_LAYERS * (TEST_NUM_HEADS * TEST_HEAD_DIM) * TEST_EMBEDDING_DIM; const float* emb_ptr = reinterpret_cast(TEST_W_EMB); model.w_emb = std::vector(emb_ptr, emb_ptr + emb_elements); const float* qkv_ptr = reinterpret_cast(TEST_W_QKV); model.w_qkv = std::vector(qkv_ptr, qkv_ptr + qkv_elements); const float* o_ptr = reinterpret_cast(TEST_W_O); model.w_o = std::vector(o_ptr, o_ptr + o_elements); return model; }