2024-07-23 13:10:17 +03:00
|
|
|
#pragma once
|
|
|
|
|
|
2025-01-12 11:32:42 +02:00
|
|
|
#include "llama.h"
|
|
|
|
|
|
|
|
|
|
#include <vector>
|
2024-07-23 13:10:17 +03:00
|
|
|
|
2024-09-07 15:16:19 +03:00
|
|
|
struct llama_vocab;
|
|
|
|
|
struct llama_grammar;
|
2024-07-23 13:10:17 +03:00
|
|
|
|
2024-09-07 15:16:19 +03:00
|
|
|
// sampler chain
|
2024-07-23 13:10:17 +03:00
|
|
|
|
2024-09-07 15:16:19 +03:00
|
|
|
struct llama_sampler_chain {
|
|
|
|
|
llama_sampler_chain_params params;
|
|
|
|
|
|
2026-01-04 21:22:16 +01:00
|
|
|
// has .backend_init() been called?
|
|
|
|
|
bool is_init = false;
|
|
|
|
|
|
|
|
|
|
struct info {
|
|
|
|
|
bool is_backend;
|
|
|
|
|
|
|
|
|
|
llama_sampler * ptr;
|
|
|
|
|
};
|
|
|
|
|
|
|
|
|
|
std::vector<info> samplers;
|
2024-09-07 15:16:19 +03:00
|
|
|
|
2025-12-30 06:27:49 -08:00
|
|
|
// pre-allocated buffer for llama_sampler_sample to avoid repeated allocations
|
|
|
|
|
std::vector<llama_token_data> cur;
|
|
|
|
|
|
2024-09-07 15:16:19 +03:00
|
|
|
// timing
|
|
|
|
|
|
|
|
|
|
mutable int64_t t_sample_us;
|
|
|
|
|
|
|
|
|
|
mutable int32_t n_sample;
|
2024-07-23 13:10:17 +03:00
|
|
|
};
|
|
|
|
|
|
2024-10-25 10:07:34 -06:00
|
|
|
struct llama_sampler * llama_sampler_init_dry_testing(
|
2026-01-04 21:22:16 +01:00
|
|
|
int32_t context_size,
|
|
|
|
|
float dry_multiplier,
|
|
|
|
|
float dry_base,
|
|
|
|
|
int32_t dry_allowed_length,
|
|
|
|
|
int32_t dry_penalty_last_n,
|
|
|
|
|
const std::vector<std::vector<llama_token>> & seq_breakers);
|