Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
gritlm.cpp230 linesDownload Raw Back to gritlm
1#include "arg.h"2#include "common.h"3#include "llama.h"4 5#include <string>6#include <vector>7 8// #define GRIT_DEBUG9 10static std::vector<std::vector<float>> encode(llama_context * ctx, const std::vector<std::string> & sentences, const std::string & instruction) {11    std::vector<std::vector<float>> result;12 13    const llama_model * model = llama_get_model(ctx);14    const llama_vocab * vocab = llama_model_get_vocab(model);15 16    llama_batch batch = llama_batch_init(llama_n_batch(ctx), 0, 1);17 18    for (uint64_t i = 0; i < sentences.size(); i++) {19        common_batch_clear(batch);20 21        const std::string input_string = instruction + sentences[i];22 23        std::vector<llama_token> inputs = common_tokenize(vocab, input_string, true, false);24 25        const int32_t n_toks = inputs.size();26 27        // GritLM seems to have EOS = ""28        // https://github.com/ContextualAI/gritlm/blob/92025b16534712b31b3c4aaaf069350e222bd5f8/gritlm/gritlm.py#L1829        // inputs.push_back(llama_vocab_eos(vocab));30 31        // we want to ignore instruction tokens for mean pooling32        const int32_t n_inst = common_tokenize(vocab, instruction, true, false).size();33 34#ifdef GRIT_DEBUG35        // debug tokens - should be matching as referenced in the GritLM sample36        std::for_each(inputs.begin(), inputs.end(), [&ctx](llama_token t) {37            std::printf("[%u:%s]", t, llama_token_to_piece(ctx, t).c_str());38        });39        std::printf("\n");40#endif41 42        // add input to batch (this increments n_tokens)43        for (int32_t j = 0; j < n_toks; j++) {44            common_batch_add(batch, inputs[j], j, { 0 }, j >= n_inst);45        }46 47        // clear previous kv_cache values (irrelevant for embeddings)48        llama_kv_cache_clear(ctx);49        llama_set_embeddings(ctx, true);50        llama_set_causal_attn(ctx, false);51 52        // run model53        llama_decode(ctx, batch);54 55        // get embedding dimensions56        uint64_t n_embd = llama_model_n_embd(model);57 58        // allocate embedding output59        std::vector<float> emb_unorm(n_embd, 0.0f);60 61        // sum up all token embeddings62        for (int32_t k = n_inst; k < n_toks; k++) {63            float * emb = llama_get_embeddings_ith(ctx, k);64            for (uint64_t j = 0; j < n_embd; j++) {65                emb_unorm[j] += emb[j];66            }67        }68 69        // divide by number of tokens (mean pooling)70        {71            const uint64_t n_sent = n_toks - n_inst;72 73            for (uint64_t j = 0; j < n_embd; j++) {74                emb_unorm[j] /= n_sent;75            }76        }77 78        std::vector<float> emb_norm(emb_unorm.size());79        common_embd_normalize(emb_unorm.data(), emb_norm.data(), n_embd, 2);80        result.push_back(emb_norm);81 82#ifdef GRIT_DEBUG83        // print out emb_norm84        std::printf("embedding %ld: ", i);85        for (uint64_t j = 0; j < n_embd; j++) {86            std::printf("%.5f ", emb_norm[j]);87        }88        std::printf("\n\n");89#endif90    }91 92    llama_batch_free(batch);93 94    return result;95}96 97static std::string generate(llama_context * ctx, llama_sampler * smpl, const std::string & prompt, bool stream) {98    std::string result;99 100    const llama_model * model = llama_get_model(ctx);101    const llama_vocab * vocab = llama_model_get_vocab(model);102 103    llama_token eos_token = llama_vocab_eos(vocab);104 105    llama_kv_cache_clear(ctx);106    llama_set_embeddings(ctx, false);107    llama_set_causal_attn(ctx, true);108 109    llama_batch bat = llama_batch_init(llama_n_batch(ctx), 0, 1);110 111    std::vector<llama_token> inputs = common_tokenize(vocab, prompt, false, true);112    int32_t i_current_token = 0;113 114    while (true) {115        common_batch_clear(bat);116        {117            const int32_t n_inputs = inputs.size();118 119            for (int32_t i = 0; i < n_inputs; i++) {120                common_batch_add(bat, inputs[i], i_current_token++, { 0 }, i == n_inputs - 1);121            }122        }123        inputs.clear();124 125        llama_decode(ctx, bat);126 127        llama_token token = llama_sampler_sample(smpl, ctx, bat.n_tokens - 1);128 129        if (token == eos_token) {130            break;131        }132 133        std::string piece = common_token_to_piece(ctx, token);134        if (stream) {135            std::printf("%s", piece.c_str());136            std::fflush(stdout);137        }138 139        inputs.push_back(token);140 141        result += piece;142    }143 144    if (stream) {145        std::printf("\n");146    }147 148    llama_batch_free(bat);149 150    return result;151}152 153static std::string gritlm_instruction(const std::string & instruction) {154    return !instruction.empty() ? "<|user|>\n" + instruction + "\n<|embed|>\n" : "<|embed|>\n";155}156 157int main(int argc, char * argv[]) {158    common_params params;159 160    if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_COMMON)) {161        return 1;162    }163 164    common_init();165 166    llama_model_params mparams = common_model_params_to_llama(params);167    llama_context_params cparams = common_context_params_to_llama(params);168 169    llama_backend_init();170 171    llama_model * model = llama_model_load_from_file(params.model.c_str(), mparams);172 173    // create generation context174    llama_context * ctx = llama_init_from_model(model, cparams);175 176    auto sparams = llama_sampler_chain_default_params();177 178    sparams.no_perf = false;179 180    llama_sampler * smpl = llama_sampler_chain_init(sparams);181 182    llama_sampler_chain_add(smpl, llama_sampler_init_greedy());183 184    // ### Embedding/Representation ###185    // samples taken from: https://github.com/ContextualAI/gritlm#basic186    {187        const std::string instruction = "Given a scientific paper title, retrieve the paper's abstract";188 189        const std::vector<std::string> queries = {190            "Bitcoin: A Peer-to-Peer Electronic Cash System",191            "Generative Representational Instruction Tuning",192        };193 194        const std::vector<std::string> documents = {195            "A purely peer-to-peer version of electronic cash would allow online payments to be sent directly from one party to another without going through a financial institution. Digital signatures provide part of the solution, but the main benefits are lost if a trusted third party is still required to prevent double-spending. We propose a solution to the double-spending problem using a peer-to-peer network. The network timestamps transactions by hashing them into an ongoing chain of hash-based proof-of-work, forming a record that cannot be changed without redoing the proof-of-work. The longest chain not only serves as proof of the sequence of events witnessed, but proof that it came from the largest pool of CPU power. As long as a majority of CPU power is controlled by nodes that are not cooperating to attack the network, they'll generate the longest chain and outpace attackers. The network itself requires minimal structure. Messages are broadcast on a best effort basis, and nodes can leave and rejoin the network at will, accepting the longest proof-of-work chain as proof of what happened while they were gone.",196            "All text-based language problems can be reduced to either generation or embedding. Current models only perform well at one or the other. We introduce generative representational instruction tuning (GRIT) whereby a large language model is trained to handle both generative and embedding tasks by distinguishing between them through instructions. Compared to other open models, our resulting GritLM 7B sets a new state of the art on the Massive Text Embedding Benchmark (MTEB) and outperforms all models up to its size on a range of generative tasks. By scaling up further, GritLM 8X7B outperforms all open generative language models that we tried while still being among the best embedding models. Notably, we find that GRIT matches training on only generative or embedding data, thus we can unify both at no performance loss. Among other benefits, the unification via GRIT speeds up Retrieval-Augmented Generation (RAG) by > 60% for long documents, by no longer requiring separate retrieval and generation models. Models, code, etc. are freely available at https://github.com/ContextualAI/gritlm.",197        };198 199        // No need to add instruction for retrieval documents200        const std::vector<std::vector<float>> d_rep = encode(ctx, documents, gritlm_instruction(""));201        const std::vector<std::vector<float>> q_rep = encode(ctx, queries,   gritlm_instruction(instruction));202 203        const int n_embd = llama_model_n_embd(model);204 205        const float cosine_sim_q0_d0 = common_embd_similarity_cos(q_rep[0].data(), d_rep[0].data(), n_embd);206        const float cosine_sim_q0_d1 = common_embd_similarity_cos(q_rep[0].data(), d_rep[1].data(), n_embd);207        const float cosine_sim_q1_d0 = common_embd_similarity_cos(q_rep[1].data(), d_rep[0].data(), n_embd);208        const float cosine_sim_q1_d1 = common_embd_similarity_cos(q_rep[1].data(), d_rep[1].data(), n_embd);209 210        std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[0].c_str(), documents[0].c_str(), cosine_sim_q0_d0);211        std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[0].c_str(), documents[1].c_str(), cosine_sim_q0_d1);212        std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[1].c_str(), documents[0].c_str(), cosine_sim_q1_d0);213        std::printf("Cosine similarity between \"%.50s\" and \"%.50s\" is: %.3f\n", queries[1].c_str(), documents[1].c_str(), cosine_sim_q1_d1);214    }215 216    // ### Generation ###217    // GritLM models are not finetuned with system prompts, as you can just include system-like instructions together with your user instruction218    {219        const std::string prompt = "<|user|>\nPlease write me a poem about my recent hike of Mt. Fuji at midnight in the style of Shakespeare.\n<|assistant|>\n";220        std::string response = generate(ctx, smpl, prompt, true);221    }222 223    llama_sampler_free(smpl);224    llama_free(ctx);225    llama_model_free(model);226    llama_backend_free();227 228    return 0;229}230