finish porting speculative decoding in server

2026-02-04 05:23:58 +00:00 · 2025-07-25 04:00:02 +00:00
parent 99c1ef3c01
commit 642b70a64b
3 changed files with 27 additions and 49 deletions
--- a/common/sampling.cpp
+++ b/common/sampling.cpp
@@ -507,15 +507,12 @@ void llama_sampling_accept(
    }
 }

-std::vector<llama_token> llama_sampling_sample_and_accept_n(struct llama_sampling_context * gsmpl, struct llama_context * ctx, const std::vector<int> & idxs, const std::vector<llama_token> & draft) {
-    GGML_ASSERT(idxs.size() == draft.size() + 1 && "idxs.size() must be draft.size() + 1");
-
+std::vector<llama_token> llama_sampling_sample_and_accept_n(struct llama_sampling_context * gsmpl, struct llama_context * ctx, const std::vector<llama_token> & draft) {
    std::vector<llama_token> result;
-    result.reserve(idxs.size());

    size_t i = 0;
    for (; i < draft.size(); i++) {
-        const llama_token id = llama_sampling_sample(gsmpl, ctx, nullptr, idxs[i]);
+        const llama_token id = llama_sampling_sample(gsmpl, ctx, nullptr, i);

        llama_sampling_accept(gsmpl, ctx, id, true);

@@ -527,7 +524,7 @@ std::vector<llama_token> llama_sampling_sample_and_accept_n(struct llama_samplin
    }

    if (i == draft.size()) {
-        const llama_token id = llama_sampling_sample(gsmpl, ctx, nullptr, idxs[i]);
+        const llama_token id = llama_sampling_sample(gsmpl, ctx, nullptr, i);

        llama_sampling_accept(gsmpl, ctx, id, true);