KBaba7/llama.cpp
0
1#include "arg.h"2#include "common.h"3#include "llama.h"4 5#include <string>6#include <vector>7 8// #define GRIT_DEBUG9 10static std::vector<std::vector<float>> encode(llama_context * ctx, const std::vector<std::string> & sentences, const std::string & instruction) {11 std::vector<std::vector<float>> result;12 13 const llama_model * model = llama_get_model(ctx);14 const llama_vocab * vocab = llama_model_get_vocab(model);15 16 llama_batch batch = llama_batch_init(llama_n_batch(ctx), 0, 1);17 18 for (uint64_t i = 0; i < sentences.size(); i++) {19 common_batch_clear(batch);20 21 const std::string input_string = instruction + sentences[i];22 23 std::vector<llama_token> inputs = common_tokenize(vocab, input_string, true, false);24 25 const int32_t n_toks = inputs.size();26 27 // GritLM seems to have EOS = ""28 // https://github.com/ContextualAI/gritlm/blob/92025b16534712b31b3c4aaaf069350e222bd5f8/gritlm/gritlm.py#L1829 // inputs.push_back(llama_vocab_eos(vocab));30 31 // we want to ignore instruction tokens for mean pooling32 const int32_t n_inst = common_tokenize(vocab, instruction, true, false).size();33 34#ifdef GRIT_DEBUG35 // debug tokens - should be matching as referenced in the GritLM sample36 std::for_each(inputs.begin(), inputs.end(), [&ctx](llama_token t) {37 std::printf("[%u:%s]", t, llama_token_to_piece(ctx, t).c_str());38 });39 std::printf("\n");40#endif41 42 // add input to batch (this increments n_tokens)43 for (int32_t j = 0; j < n_toks; j++) {44 common_batch_add(batch, inputs[j], j, { 0 }, j >= n_inst);45 }46 47 // clear previous kv_cache values (irrelevant for embeddings)48 llama_kv_cache_clear(ctx);49 llama_set_embeddings(ctx, true);50 llama_set_causal_attn(ctx, false);51 52 // run model53 llama_decode(ctx, batch);54 55 // get embedding dimensions56 uint64_t n_embd = llama_model_n_embd(model);57 58 // allocate embedding output59 std::vector<float> emb_unorm(n_embd, 0.0f);60 61 // sum up all token embeddings62 for (int32_t k = n_inst; k < n_toks; k++) {63 float * emb = llama_get_embeddings_ith(ctx, k);64 for (uint64_t j = 0; j < n_embd; j++) {65 emb_unorm[j] += emb[j];66 }67 }68 69 // divide by number of tokens (mean pooling)70 {71 const uint64_t n_sent = n_toks - n_inst;72 73 for (uint64_t j = 0; j < n_embd; j++) {74 emb_unorm[j] /= n_sent;75 }76 }77 78 std::vector<float> emb_norm(emb_unorm.size());79 common_embd_normalize(emb_unorm.data(), emb_norm.data(), n_embd, 2);80 result.push_back(emb_norm);81 82#ifdef GRIT_DEBUG83 // print out emb_norm84 std::printf("embedding %ld: ", i);85 for (uint64_t j = 0; j < n_embd; j++) {86 std::printf("%.5f ", emb_norm[j]);87 }88 std::printf("\n\n");89#endif90 }91 92 llama_batch_free(batch);93 94 return result;95}96 97static std::string generate(llama_context * ctx, llama_sampler * smpl, const std::string & prompt, bool stream) {98 std::string result;99 100 const llama_model * model = llama_get_model(ctx);101 const llama_vocab * vocab = llama_model_get_vocab(model);102 103 llama_token eos_token = llama_vocab_eos(vocab);104 105 llama_kv_cache_clear(ctx);106 llama_set_embeddings(ctx, false);107 llama_set_causal_attn(ctx, true);108 109 llama_batch bat = llama_batch_init(llama_n_batch(ctx), 0, 1);110 111 std::vector<llama_token> inputs = common_tokenize(vocab, prompt, false, true);112 int32_t i_current_token = 0;113 114 while (true) {115 common_batch_clear(bat);116 {117 const int32_t n_inputs = inputs.size();118 119 for (int32_t i = 0; i < n_inputs; i++) {120 common_batch_add(bat, inputs[i], i_current_token++, { 0 }, i == n_inputs - 1);121 }122 }123 inputs.clear();124 125 llama_decode(ctx, bat);126 127 llama_token token = llama_sampler_sample(smpl, ctx, bat.n_tokens - 1);128 129 if (token == eos_token) {130 break;131 }132 133 std::string piece = common_token_to_piece(ctx, token);134 if (stream) {135 std::printf("%s", piece.c_str());136 std::fflush(stdout);137 }138 139 inputs.push_back(token);140 141 result += piece;142 }143 144 if (stream) {145 std::printf("\n");146 }147 148 llama_batch_free(bat);149 150 return result;151}152 153static std::string gritlm_instruction(const std::string & instruction) {154 return !instruction.empty() ? "<|user|>\n" + instruction + "\n<|embed|>\n" : "<|embed|>\n";155}156 157int main(int argc, char * argv[]) {158 common_params params;159 160 if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_COMMON)) {161 return 1;162 }163 164 common_init();165 166 llama_model_params mparams = common_model_params_to_llama(params);167 llama_context_params cparams = common_context_params_to_llama(params);168 169 llama_backend_init();170 171 llama_model * model = llama_model_load_from_file(params.model.c_str(), mparams);172 173 // create generation context174 llama_context * ctx = llama_init_from_model(model, cparams);175 176 auto sparams = llama_sampler_chain_default_params();177 178 sparams.no_perf = false;179 180 llama_sampler * smpl = llama_sampler_chain_init(sparams);181 182 llama_sampler_chain_add(smpl, llama_sampler_init_greedy());183 184 // ### Embedding/Representation ###185 // samples taken from: https://github.com/ContextualAI/gritlm#basic186 {187 const std::string instruction = "Given a scientific paper title, retrieve the paper's abstract";188 189 const std::vector<std::string> queries = {190 "Bitcoin: A Peer-to-Peer Electronic Cash System",191 "Generative Representational Instruction Tuning",192 };193 194 const std::vector<std::string> documents = {195 "A purely peer-to-peer version of electronic cash would allow online payments to be sent directly from one party to another without going through a financial institution. Digital signatures provide part of the solution, but the main benefits are lost if a trusted third party is still required to prevent double-spending. We propose a solution to the double-spending problem using a peer-to-peer network. The network timestamps transactions by hashing them into an ongoing chain of hash-based proof-of-work, forming a record that cannot be changed without redoing the proof-of-work. The longest chain not only serves as proof of the sequence of events witnessed, but proof that it came from the largest pool of CPU power. As long as a majority of CPU power is controlled by nodes that are not cooperating to attack the network, they'll generate the longest chain and outpace attackers. The network itself requires minimal structure. Messages are broadcast on a best effort basis, and nodes can leave and rejoin the network at will, accepting the longest proof-of-work chain as proof of what happened while they were gone.",196 "All text-based language problems can be reduced to either generation or embedding. Current models only perform well at one or the other. We introduce generative representational instruction tuning (GRIT) whereby a large language model is trained to handle both generative and embedding tasks by distinguishing between them through instructions. Compared to other open models, our resulting GritLM 7B sets a new state of the art on the Massive Text Embedding Benchmark (MTEB) and outperforms all models up to its size on a range of generative tasks. By scaling up further, GritLM 8X7B outperforms all open generative language models that we tried while still being among the best embedding models. Notably, we find that GRIT matches training on only generative or embedding data, thus we can unify both at no performance loss. Among other benefits, the unification via GRIT speeds up Retrieval-Augmented Generation (RAG) by > 60% for long documents, by no longer requiring separate retrieval and generation models. Models, code, etc. are freely available at https://github.com/ContextualAI/gritlm.",197 };198 199 // No need to add instruction for retrieval documents200 const std::vector<std::vector<float>> d_rep = encode(ctx, documents, gritlm_instruction(""));201 const std::vector<std::vector<float>> q_rep = encode(ctx, queries, gritlm_instruction(instruction));202 203 const int n_embd = llama_model_n_embd(model);204 205 const float cosine_sim_q0_d0 = common_embd_similarity_cos(q_rep[0].data(), d_rep[0].data(), n_embd);206 const float cosine_sim_q0_d1 = common_embd_similarity_cos(q_rep[0].data(), d_rep[1].data(), n_embd);207 const float cosine_sim_q1_d0 = common_embd_similarity_cos(q_rep[1].data(), d_rep[0].data(), n_embd);208 const float cosine_sim_q1_d1 = common_embd_similarity_cos(q_rep[1].data(), d_rep[1].data(), n_embd);209 210 std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[0].c_str(), documents[0].c_str(), cosine_sim_q0_d0);211 std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[0].c_str(), documents[1].c_str(), cosine_sim_q0_d1);212 std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[1].c_str(), documents[0].c_str(), cosine_sim_q1_d0);213 std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[1].c_str(), documents[1].c_str(), cosine_sim_q1_d1);214 }215 216 // ### Generation ###217 // GritLM models are not finetuned with system prompts, as you can just include system-like instructions together with your user instruction218 {219 const std::string prompt = "<|user|>\nPlease write me a poem about my recent hike of Mt. Fuji at midnight in the style of Shakespeare.\n<|assistant|>\n";220 std::string response = generate(ctx, smpl, prompt, true);221 }222 223 llama_sampler_free(smpl);224 llama_free(ctx);225 llama_model_free(model);226 llama_backend_free();227 228 return 0;229}230 