2023-03-10 20:40:58 +02:00
// Various helper functions and utilities
#pragma once
2023-03-22 07:32:36 +02:00
#include "llama.h"
2023-03-10 20:40:58 +02:00
#include <string>
#include <vector>
2024-09-15 20:46:12 +03:00
#include <sstream>
2023-03-10 20:40:58 +02:00
2023-08-28 17:59:39 +02:00
#ifdef _WIN32
#define DIRECTORY_SEPARATOR '\\'
#else
#define DIRECTORY_SEPARATOR '/'
#endif // _WIN32
2023-09-15 14:02:01 -04:00
#define die(msg) do { fputs("error: " msg "\n", stderr); exit(1); } while (0)
#define die_fmt(fmt, ...) do { fprintf(stderr, "error: " fmt "\n", __VA_ARGS__); exit(1); } while (0)
2023-09-07 13:22:29 -04:00
2023-11-02 02:50:16 -04:00
#define print_build_info() do { \
2024-05-22 20:04:20 +03:00
fprintf(stderr, "%s: build = %d (%s)\n", __func__, LLAMA_BUILD_NUMBER, LLAMA_COMMIT); \
2023-11-02 02:50:16 -04:00
fprintf(stderr, "%s: built with %s for %s\n", __func__, LLAMA_COMPILER, LLAMA_BUILD_TARGET); \
2023-09-15 16:59:49 -04:00
} while(0)
2024-04-30 00:52:50 +01:00
#define DEFAULT_MODEL_PATH "models/7B/ggml-model-f16.gguf"
2024-10-10 22:57:42 +02:00
struct common_lora_adapter_info {
2024-08-06 17:33:39 +02:00
std :: string path ;
float scale ;
};
2024-10-10 22:57:42 +02:00
struct common_lora_adapter_container : common_lora_adapter_info {
2024-08-06 17:33:39 +02:00
struct llama_lora_adapter * adapter ;
};
2024-11-25 09:58:41 +02:00
using llama_tokens = std :: vector < llama_token > ;
2023-11-02 02:50:16 -04:00
// build info
extern int LLAMA_BUILD_NUMBER ;
2024-05-22 20:04:20 +03:00
extern char const * LLAMA_COMMIT ;
extern char const * LLAMA_COMPILER ;
extern char const * LLAMA_BUILD_TARGET ;
2023-11-02 02:50:16 -04:00
2024-10-10 22:57:42 +02:00
struct common_control_vector_load_info ;
2024-03-15 13:43:02 -07:00
2024-05-22 20:04:20 +03:00
//
// CPU utils
//
2024-09-09 23:36:09 +02:00
struct cpu_params {
int n_threads = - 1 ;
bool cpumask [ GGML_MAX_N_THREADS ] = { false }; // CPU affinity mask.
bool mask_valid = false ; // Default: any CPU
enum ggml_sched_priority priority = GGML_SCHED_PRIO_NORMAL ; // Scheduling prio : (0 - normal, 1 - medium, 2 - high, 3 - realtime)
bool strict_cpu = false ; // Use strict CPU placement
uint32_t poll = 50 ; // Polling (busywait) level (0 - no polling, 100 - mostly polling)
};
2024-05-22 20:04:20 +03:00
int32_t cpu_get_num_physical_cores ();
int32_t cpu_get_num_math ();
2024-03-15 13:43:02 -07:00
2023-03-10 20:40:58 +02:00
//
2024-09-09 23:36:09 +02:00
// Common params
2023-03-10 20:40:58 +02:00
//
2024-09-07 20:43:51 +02:00
enum llama_example {
LLAMA_EXAMPLE_COMMON ,
LLAMA_EXAMPLE_SPECULATIVE ,
LLAMA_EXAMPLE_MAIN ,
LLAMA_EXAMPLE_INFILL ,
LLAMA_EXAMPLE_EMBEDDING ,
LLAMA_EXAMPLE_PERPLEXITY ,
LLAMA_EXAMPLE_RETRIEVAL ,
LLAMA_EXAMPLE_PASSKEY ,
LLAMA_EXAMPLE_IMATRIX ,
LLAMA_EXAMPLE_BENCH ,
LLAMA_EXAMPLE_SERVER ,
LLAMA_EXAMPLE_CVECTOR_GENERATOR ,
LLAMA_EXAMPLE_EXPORT_LORA ,
LLAMA_EXAMPLE_LLAVA ,
2024-09-09 23:36:09 +02:00
LLAMA_EXAMPLE_LOOKUP ,
LLAMA_EXAMPLE_PARALLEL ,
2024-09-07 20:43:51 +02:00
LLAMA_EXAMPLE_COUNT ,
};
2024-10-10 22:57:42 +02:00
enum common_sampler_type {
COMMON_SAMPLER_TYPE_NONE = 0 ,
2024-10-25 10:07:34 -06:00
COMMON_SAMPLER_TYPE_DRY = 1 ,
COMMON_SAMPLER_TYPE_TOP_K = 2 ,
COMMON_SAMPLER_TYPE_TOP_P = 3 ,
COMMON_SAMPLER_TYPE_MIN_P = 4 ,
2024-10-29 10:42:05 +02:00
//COMMON_SAMPLER_TYPE_TFS_Z = 5,
2024-10-25 10:07:34 -06:00
COMMON_SAMPLER_TYPE_TYPICAL_P = 6 ,
COMMON_SAMPLER_TYPE_TEMPERATURE = 7 ,
COMMON_SAMPLER_TYPE_XTC = 8 ,
COMMON_SAMPLER_TYPE_INFILL = 9 ,
2024-09-09 23:36:09 +02:00
};
2024-06-25 13:59:54 +02:00
// dimensionality reduction methods, used by cvector-generator
enum dimre_method {
DIMRE_METHOD_PCA ,
DIMRE_METHOD_MEAN ,
};
2024-11-25 09:58:41 +02:00
// sampling parameters
struct common_params_sampling {
2024-09-09 23:36:09 +02:00
uint32_t seed = LLAMA_DEFAULT_SEED ; // the seed used to initialize llama_sampler
2024-10-25 10:07:34 -06:00
int32_t n_prev = 64 ; // number of previous tokens to remember
int32_t n_probs = 0 ; // if greater than 0, output the probabilities of top n_probs tokens.
int32_t min_keep = 0 ; // 0 = disabled, otherwise samplers should return at least min_keep tokens
int32_t top_k = 40 ; // <= 0 to use vocab size
float top_p = 0.95f ; // 1.0 = disabled
float min_p = 0.05f ; // 0.0 = disabled
float xtc_probability = 0.00f ; // 0.0 = disabled
float xtc_threshold = 0.10f ; // > 0.5 disables XTC
float typ_p = 1.00f ; // typical_p, 1.0 = disabled
float temp = 0.80f ; // <= 0.0 to sample greedily, 0.0 to not output probabilities
float dynatemp_range = 0.00f ; // 0.0 = disabled
float dynatemp_exponent = 1.00f ; // controls how entropy maps to temperature in dynamic temperature sampler
int32_t penalty_last_n = 64 ; // last n tokens to penalize (0 = disable penalty, -1 = context size)
float penalty_repeat = 1.00f ; // 1.0 = disabled
float penalty_freq = 0.00f ; // 0.0 = disabled
float penalty_present = 0.00f ; // 0.0 = disabled
float dry_multiplier = 0.0f ; // 0.0 = disabled; DRY repetition penalty for tokens extending repetition:
float dry_base = 1.75f ; // 0.0 = disabled; multiplier * base ^ (length of sequence before token - allowed length)
int32_t dry_allowed_length = 2 ; // tokens extending repetitions beyond this receive penalty
int32_t dry_penalty_last_n = - 1 ; // how many tokens to scan for repetitions (0 = disable penalty, -1 = context size)
int32_t mirostat = 0 ; // 0 = disabled, 1 = mirostat, 2 = mirostat 2.0
float mirostat_tau = 5.00f ; // target entropy
float mirostat_eta = 0.10f ; // learning rate
bool penalize_nl = false ; // consider newlines as a repeatable token
bool ignore_eos = false ;
bool no_perf = false ; // disable performance metrics
std :: vector < std :: string > dry_sequence_breakers = { " \n " , ":" , " \" " , "*" }; // default sequence breakers for DRY
2024-09-09 23:36:09 +02:00
2024-10-15 15:54:55 +05:00
2024-10-10 22:57:42 +02:00
std :: vector < enum common_sampler_type > samplers = {
2024-10-25 10:07:34 -06:00
COMMON_SAMPLER_TYPE_DRY ,
2024-10-10 22:57:42 +02:00
COMMON_SAMPLER_TYPE_TOP_K ,
COMMON_SAMPLER_TYPE_TYPICAL_P ,
COMMON_SAMPLER_TYPE_TOP_P ,
COMMON_SAMPLER_TYPE_MIN_P ,
2024-10-15 15:54:55 +05:00
COMMON_SAMPLER_TYPE_XTC ,
2024-10-15 16:35:33 +03:00
COMMON_SAMPLER_TYPE_TEMPERATURE ,
2024-09-09 23:36:09 +02:00
};
std :: string grammar ; // optional BNF-like grammar to constrain sampling
std :: vector < llama_logit_bias > logit_bias ; // logit biases to apply
// print the parameters into a string
std :: string print () const ;
2024-08-29 19:20:53 -04:00
};
2024-11-25 09:58:41 +02:00
struct common_params_speculative {
2024-11-25 19:30:06 +01:00
std :: vector < ggml_backend_dev_t > devices ; // devices to use for offloading
2024-11-25 09:58:41 +02:00
int32_t n_ctx = 0 ; // draft context size
int32_t n_max = 16 ; // maximum number of tokens to draft during speculative decoding
int32_t n_min = 5 ; // minimum number of draft tokens to use for speculative decoding
int32_t n_gpu_layers = - 1 ; // number of layers to store in VRAM for the draft model (-1 - use default)
float p_split = 0.1f ; // speculative decoding split probability
float p_min = 0.9f ; // minimum speculative decoding probability (greedy)
struct cpu_params cpuparams ;
struct cpu_params cpuparams_batch ;
std :: string model = "" ; // draft model for speculative decoding // NOLINT
};
2024-10-10 22:57:42 +02:00
struct common_params {
2024-06-06 16:30:58 +03:00
int32_t n_predict = - 1 ; // new tokens to predict
2024-11-02 15:18:56 +02:00
int32_t n_ctx = 4096 ; // context size
2024-06-06 16:30:58 +03:00
int32_t n_batch = 2048 ; // logical batch size for prompt processing (must be >=32 to use BLAS)
int32_t n_ubatch = 512 ; // physical batch size for prompt processing (must be >=32 to use BLAS)
int32_t n_keep = 0 ; // number of tokens to keep from initial prompt
int32_t n_chunks = - 1 ; // max number of chunks to process (-1 = unlimited)
int32_t n_parallel = 1 ; // number of parallel sequences to decode
int32_t n_sequences = 1 ; // number of sequences to decode
int32_t grp_attn_n = 1 ; // group-attention factor
int32_t grp_attn_w = 512 ; // group-attention width
int32_t n_print = - 1 ; // print token count every n tokens (-1 = disabled)
float rope_freq_base = 0.0f ; // RoPE base frequency
float rope_freq_scale = 0.0f ; // RoPE frequency scaling factor
2024-01-31 17:30:17 +02:00
float yarn_ext_factor = - 1.0f ; // YaRN extrapolation mix factor
2024-06-06 16:30:58 +03:00
float yarn_attn_factor = 1.0f ; // YaRN magnitude scaling factor
2024-01-31 17:30:17 +02:00
float yarn_beta_fast = 32.0f ; // YaRN low correction dim
2024-06-06 16:30:58 +03:00
float yarn_beta_slow = 1.0f ; // YaRN high correction dim
int32_t yarn_orig_ctx = 0 ; // YaRN original context length
2024-11-11 08:38:43 +02:00
float defrag_thold = 0.1f ; // KV cache defragmentation threshold
2024-03-03 04:40:27 -06:00
2024-11-25 19:30:06 +01:00
// offload params
std :: vector < ggml_backend_dev_t > devices ; // devices to use for offloading
int32_t n_gpu_layers = - 1 ; // number of layers to store in VRAM (-1 - use default)
int32_t main_gpu = 0 ; // the GPU that is used for scratch and small tensors
float tensor_split [ 128 ] = { 0 }; // how split tensors should be distributed across GPUs
enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER ; // how to split the model across GPUs
2024-08-29 19:20:53 -04:00
struct cpu_params cpuparams ;
struct cpu_params cpuparams_batch ;
2024-04-11 14:51:07 +02:00
ggml_backend_sched_eval_callback cb_eval = nullptr ;
void * cb_eval_user_data = nullptr ;
2024-03-03 04:40:27 -06:00
ggml_numa_strategy numa = GGML_NUMA_STRATEGY_DISABLED ;
2024-04-24 08:10:07 -05:00
enum llama_rope_scaling_type rope_scaling_type = LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED ;
enum llama_pooling_type pooling_type = LLAMA_POOLING_TYPE_UNSPECIFIED ; // pooling type for embeddings
2024-07-05 02:05:56 -05:00
enum llama_attention_type attention_type = LLAMA_ATTENTION_TYPE_UNSPECIFIED ; // attention type for embeddings
2023-03-17 21:46:46 +02:00
2024-11-25 09:58:41 +02:00
struct common_params_sampling sampling ;
struct common_params_speculative speculative ;
2023-07-12 00:18:43 +08:00
2024-09-09 23:36:09 +02:00
std :: string model = "" ; // model path // NOLINT
std :: string model_alias = "unknown" ; // model alias // NOLINT
std :: string model_url = "" ; // model url to download // NOLINT
std :: string hf_token = "" ; // HF token // NOLINT
std :: string hf_repo = "" ; // HF repo // NOLINT
std :: string hf_file = "" ; // HF file // NOLINT
std :: string prompt = "" ; // NOLINT
std :: string prompt_file = "" ; // store the external prompt file name // NOLINT
std :: string path_prompt_cache = "" ; // path to file for saving/loading prompt eval state // NOLINT
std :: string input_prefix = "" ; // string to prefix user inputs with // NOLINT
std :: string input_suffix = "" ; // string to suffix user inputs with // NOLINT
std :: string lookup_cache_static = "" ; // path of static ngram cache file for lookup decoding // NOLINT
std :: string lookup_cache_dynamic = "" ; // path of dynamic ngram cache file for lookup decoding // NOLINT
std :: string logits_file = "" ; // file for saving *all* logits // NOLINT
std :: string rpc_servers = "" ; // comma separated list of RPC servers // NOLINT
2023-03-21 17:32:14 +02:00
2024-06-06 16:30:58 +03:00
std :: vector < std :: string > in_files ; // all input files
2024-06-04 21:23:39 +03:00
std :: vector < std :: string > antiprompt ; // strings upon which more user input is prompted (a.k.a. reverse prompts)
2023-12-05 10:19:18 -07:00
std :: vector < llama_model_kv_override > kv_overrides ;
2024-08-06 17:33:39 +02:00
bool lora_init_without_apply = false ; // only load lora to memory, but do not apply it to ctx (user can manually apply lora later using llama_lora_adapter_apply)
2024-10-10 22:57:42 +02:00
std :: vector < common_lora_adapter_info > lora_adapters ; // lora adapter path with user defined scale
2023-04-17 17:28:55 +02:00
2024-10-10 22:57:42 +02:00
std :: vector < common_control_vector_load_info > control_vectors ; // control vector with user defined scale
2024-03-15 13:43:02 -07:00
2024-06-06 16:30:58 +03:00
int32_t verbosity = 0 ;
2024-03-15 13:43:02 -07:00
int32_t control_vector_layer_start = - 1 ; // layer range for control vector
int32_t control_vector_layer_end = - 1 ; // layer range for control vector
2024-06-06 16:30:58 +03:00
int32_t ppl_stride = 0 ; // stride for perplexity calculations. If left at 0, the pre-existing approach will be used.
int32_t ppl_output_type = 0 ; // = 0 -> ppl output is as usual, = 1 -> ppl output is num_tokens, ppl, one per line
// (which is more convenient to use for plotting)
//
bool hellaswag = false ; // compute HellaSwag score over random tasks from datafile supplied in prompt
size_t hellaswag_tasks = 400 ; // number of tasks to use when computing the HellaSwag score
2023-07-28 20:25:36 +02:00
2024-06-06 16:30:58 +03:00
bool winogrande = false ; // compute Winogrande score over random tasks from datafile supplied in prompt
size_t winogrande_tasks = 0 ; // number of tasks to use when computing the Winogrande score. If 0, all tasks will be computed
2024-01-18 13:46:27 +02:00
2024-06-06 16:30:58 +03:00
bool multiple_choice = false ; // compute TruthfulQA score over random tasks from datafile supplied in prompt
size_t multiple_choice_tasks = 0 ; // number of tasks to use when computing the TruthfulQA score. If 0, all tasks will be computed
2024-01-21 14:42:44 +02:00
2024-06-06 16:30:58 +03:00
bool kl_divergence = false ; // compute KL divergence
2024-01-22 16:10:14 +02:00
2024-06-04 21:23:39 +03:00
bool usage = false ; // print usage
2023-03-21 17:32:14 +02:00
bool use_color = false ; // use color to distinguish generations and inputs
2024-05-27 00:10:17 +10:00
bool special = false ; // enable special token output
2024-06-04 21:23:39 +03:00
bool interactive = false ; // interactive mode
bool interactive_first = false ; // wait for user input immediately
2024-05-09 02:32:32 +12:00
bool conversation = false ; // conversation mode (does not print special tokens and suffix/prefix)
2023-05-10 11:37:14 -04:00
bool prompt_cache_all = false ; // save user input and generations to prompt cache
2023-06-07 04:10:17 +02:00
bool prompt_cache_ro = false ; // open the prompt cache read-only and do not update it
2023-03-24 08:05:13 -07:00
2024-06-04 21:23:39 +03:00
bool escape = true ; // escape "\n", "\r", "\t", "\'", "\"", and "\\"
2023-05-08 19:45:48 -07:00
bool multiline_input = false ; // reverse the usage of `\`
2023-08-04 08:20:12 -07:00
bool simple_io = false ; // improves compatibility with subprocesses and limited consoles
2024-03-22 13:08:28 +02:00
bool cont_batching = true ; // insert new sequences for decoding on-the-fly
2024-04-30 12:16:08 +03:00
bool flash_attn = false ; // flash attention
2024-09-13 09:53:38 +03:00
bool no_perf = false ; // disable performance metrics
2024-09-16 01:20:01 -05:00
bool ctx_shift = true ; // context shift on inifinite text generation
2023-03-24 08:05:13 -07:00
2023-07-25 07:19:11 -05:00
bool input_prefix_bos = false ; // prefix BOS to user inputs, preceding input_prefix
2023-09-28 19:04:36 +03:00
bool logits_all = false ; // return logits for all tokens in the batch
2023-04-08 12:24:37 -07:00
bool use_mmap = true ; // use mmap for faster loads
2023-03-24 08:19:05 -07:00
bool use_mlock = false ; // use mlock to keep model in memory
2023-03-25 17:16:50 +02:00
bool verbose_prompt = false ; // print prompt tokens before generation
2024-01-14 00:09:08 +08:00
bool display_prompt = true ; // print prompt before generation
2023-11-23 19:07:56 +02:00
bool dump_kv_cache = false ; // dump the KV cache contents for debugging purposes
2023-12-07 13:03:17 +02:00
bool no_kv_offload = false ; // disable KV offloading
2024-04-11 14:51:07 +02:00
bool warmup = true ; // warmup run
2024-04-26 18:39:58 +02:00
bool check_tensors = false ; // validate tensor data
2023-12-07 13:03:17 +02:00
std :: string cache_type_k = "f16" ; // KV cache data type for the K
std :: string cache_type_v = "f16" ; // KV cache data type for the V
2023-10-12 18:23:18 +03:00
// multimodal models (see examples/llava)
2024-09-09 23:36:09 +02:00
std :: string mmproj = "" ; // path to multimodal projector // NOLINT
2024-04-29 07:34:24 -07:00
std :: vector < std :: string > image ; // path to image file(s)
2024-06-04 21:23:39 +03:00
2024-06-24 13:30:24 +08:00
// embedding
bool embedding = false ; // get only sentence embedding
2024-10-22 09:40:02 +02:00
int32_t embd_normalize = 2 ; // normalisation for embeddings (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm)
2024-06-24 13:30:24 +08:00
std :: string embd_out = "" ; // empty = default, "array" = [[],[]...], "json" = openai style, "json+" = same "json" + cosine similarity matrix
2024-10-22 09:40:02 +02:00
std :: string embd_sep = " \n " ; // separator of embeddings
2024-09-28 17:42:03 +03:00
bool reranking = false ; // enable reranking support on server
2024-06-24 13:30:24 +08:00
2024-06-04 21:23:39 +03:00
// server params
2024-06-06 16:30:58 +03:00
int32_t port = 8080 ; // server listens on this network port
int32_t timeout_read = 600 ; // http read timeout in seconds
int32_t timeout_write = timeout_read ; // http write timeout in seconds
2024-10-13 18:52:48 +03:00
int32_t n_threads_http = - 1 ; // number of threads to process HTTP requests (TODO: support threadpool)
int32_t n_cache_reuse = 0 ; // min chunk size to reuse from the cache via KV shifting
2024-06-04 21:23:39 +03:00
std :: string hostname = "127.0.0.1" ;
2024-09-09 23:36:09 +02:00
std :: string public_path = "" ; // NOLINT
std :: string chat_template = "" ; // NOLINT
2024-06-30 20:27:13 +02:00
bool enable_chat_template = true ;
2024-06-04 21:23:39 +03:00
std :: vector < std :: string > api_keys ;
2024-09-09 23:36:09 +02:00
std :: string ssl_file_key = "" ; // NOLINT
std :: string ssl_file_cert = "" ; // NOLINT
2024-06-04 21:23:39 +03:00
2024-10-08 13:27:04 +02:00
// "advanced" endpoints are disabled by default for better security
bool webui = true ;
bool endpoint_slots = false ;
bool endpoint_props = false ; // only control POST requests, not GET
2024-06-04 21:23:39 +03:00
bool endpoint_metrics = false ;
bool log_json = false ;
std :: string slot_save_path ;
2024-06-08 07:50:31 +00:00
float slot_prompt_similarity = 0.5f ;
2024-06-04 21:23:39 +03:00
// batched-bench params
bool is_pp_shared = false ;
std :: vector < int32_t > n_pp ;
std :: vector < int32_t > n_tg ;
std :: vector < int32_t > n_pl ;
// retrieval params
std :: vector < std :: string > context_files ; // context files to embed
int32_t chunk_size = 64 ; // chunk size for context embedding
std :: string chunk_separator = " \n " ; // chunk separator for context embedding
// passkey params
int32_t n_junk = 250 ; // number of times to repeat the junk text
int32_t i_pos = - 1 ; // position of the passkey in the junk text
2024-06-06 16:30:58 +03:00
// imatrix params
std :: string out_file = "imatrix.dat" ; // save the resulting imatrix to this file
int32_t n_out_freq = 10 ; // output the imatrix every n_out_freq iterations
int32_t n_save_freq = 0 ; // save the imatrix every n_save_freq iterations
int32_t i_chunk = 0 ; // start processing from this chunk
bool process_output = false ; // collect data for the output tensor
bool compute_ppl = true ; // whether to compute perplexity
2024-06-15 18:53:40 +02:00
// cvector-generator params
2024-06-25 13:59:54 +02:00
int n_pca_batch = 100 ;
2024-06-15 18:53:40 +02:00
int n_pca_iterations = 1000 ;
2024-06-25 13:59:54 +02:00
dimre_method cvector_dimre_method = DIMRE_METHOD_PCA ;
std :: string cvector_outfile = "control_vector.gguf" ;
std :: string cvector_positive_file = "examples/cvector-generator/positive.txt" ;
std :: string cvector_negative_file = "examples/cvector-generator/negative.txt" ;
2024-06-28 12:53:43 +02:00
bool spm_infill = false ; // suffix/prefix/middle pattern for infill
2024-07-23 23:48:37 +02:00
std :: string lora_outfile = "ggml-lora-merged-f16.gguf" ;
2024-09-06 18:59:58 +03:00
// batched-bench params
bool batched_bench_output_jsonl = false ;
2023-03-10 20:40:58 +02:00
};
2024-09-15 20:46:12 +03:00
// call once at the start of a program if it uses libcommon
// initializes the logging system and prints info about the build
2024-10-10 22:57:42 +02:00
void common_init ();
2024-09-15 20:46:12 +03:00
2024-10-10 22:57:42 +02:00
std :: string common_params_get_system_info ( const common_params & params );
2024-04-08 20:43:30 +08:00
2024-10-12 08:21:51 +03:00
bool parse_cpu_range ( const std :: string & range , bool ( & boolmask )[ GGML_MAX_N_THREADS ]);
bool parse_cpu_mask ( const std :: string & mask , bool ( & boolmask )[ GGML_MAX_N_THREADS ]);
void postprocess_cpu_params ( cpu_params & cpuparams , const cpu_params * role_model = nullptr );
2024-08-29 19:20:53 -04:00
bool set_process_priority ( enum ggml_sched_priority prio );
2023-12-05 15:05:51 +05:00
//
2024-02-11 13:43:31 +00:00
// String utils
2023-12-05 15:05:51 +05:00
//
2024-10-12 08:21:51 +03:00
#ifdef __GNUC__
#ifdef __MINGW32__
#define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(gnu_printf, __VA_ARGS__)))
#else
#define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(printf, __VA_ARGS__)))
#endif
#else
#define LLAMA_COMMON_ATTRIBUTE_FORMAT(...)
#endif
LLAMA_COMMON_ATTRIBUTE_FORMAT ( 1 , 2 )
std :: string string_format ( const char * fmt , ...);
2024-04-29 16:58:41 +03:00
std :: string string_strip ( const std :: string & str );
2024-05-22 20:04:20 +03:00
std :: string string_get_sortable_timestamp ();
2024-06-04 21:23:39 +03:00
2024-08-09 18:23:52 +03:00
void string_replace_all ( std :: string & s , const std :: string & search , const std :: string & replace );
2024-06-04 21:23:39 +03:00
template < class T >
static std :: vector < T > string_split ( const std :: string & str , char delim ) {
2024-10-25 17:57:54 +02:00
static_assert ( ! std :: is_same < T , std :: string >:: value , "Please use the specialized version for std::string" );
2024-06-04 21:23:39 +03:00
std :: vector < T > values ;
std :: istringstream str_stream ( str );
std :: string token ;
while ( std :: getline ( str_stream , token , delim )) {
T value ;
std :: istringstream token_stream ( token );
token_stream >> value ;
values . push_back ( value );
}
return values ;
}
2024-05-22 20:04:20 +03:00
2024-10-25 17:57:54 +02:00
template <>
std :: vector < std :: string > string_split < std :: string > ( const std :: string & input , char separator )
{
std :: vector < std :: string > parts ;
size_t begin_pos = 0 ;
size_t separator_pos = input . find ( separator );
while ( separator_pos != std :: string :: npos ) {
std :: string part = input . substr ( begin_pos , separator_pos - begin_pos );
parts . emplace_back ( part );
begin_pos = separator_pos + 1 ;
separator_pos = input . find ( separator , begin_pos );
}
parts . emplace_back ( input . substr ( begin_pos , separator_pos - begin_pos ));
return parts ;
}
2024-05-22 20:04:20 +03:00
bool string_parse_kv_override ( const char * data , std :: vector < llama_model_kv_override > & overrides );
void string_process_escapes ( std :: string & input );
2024-09-15 20:46:12 +03:00
std :: string string_from ( bool value );
std :: string string_from ( const std :: vector < int > & values );
std :: string string_from ( const struct llama_context * ctx , const std :: vector < llama_token > & tokens );
std :: string string_from ( const struct llama_context * ctx , const struct llama_batch & batch );
2024-05-22 20:04:20 +03:00
//
// Filesystem utils
//
bool fs_validate_filename ( const std :: string & filename );
bool fs_create_directory_with_parents ( const std :: string & path );
std :: string fs_get_cache_directory ();
2024-06-08 20:21:08 +01:00
std :: string fs_get_cache_file ( const std :: string & filename );
2023-12-05 15:05:51 +05:00
2023-05-02 22:39:51 +02:00
//
// Model utils
//
2024-10-10 22:57:42 +02:00
struct common_init_result {
2024-08-06 17:33:39 +02:00
struct llama_model * model = nullptr ;
2024-08-06 00:14:10 +08:00
struct llama_context * context = nullptr ;
2024-10-10 22:57:42 +02:00
std :: vector < common_lora_adapter_container > lora_adapters ;
2024-08-06 00:14:10 +08:00
};
2024-10-10 22:57:42 +02:00
struct common_init_result common_init_from_params ( common_params & params );
2023-10-18 16:21:57 +03:00
2024-11-25 19:30:06 +01:00
struct llama_model_params common_model_params_to_llama ( common_params & params );
2024-10-10 22:57:42 +02:00
struct llama_context_params common_context_params_to_llama ( const common_params & params );
2024-08-29 19:20:53 -04:00
struct ggml_threadpool_params ggml_threadpool_params_from_cpu_params ( const cpu_params & params );
2023-08-21 23:07:43 +03:00
2024-10-10 22:57:42 +02:00
struct llama_model * common_load_model_from_url ( const char * model_url , const char * path_model , const char * hf_token , const struct llama_model_params & params );
struct llama_model * common_load_model_from_hf ( const char * repo , const char * file , const char * path_model , const char * hf_token , const struct llama_model_params & params );
2024-03-17 19:12:37 +01:00
2024-08-06 17:33:39 +02:00
// clear LoRA adapters from context, then apply new list of adapters
2024-10-10 22:57:42 +02:00
void common_lora_adapters_apply ( struct llama_context * ctx , std :: vector < common_lora_adapter_container > & lora_adapters );
2024-08-06 17:33:39 +02:00
2024-11-25 09:58:41 +02:00
//
2023-10-18 16:21:57 +03:00
// Batch utils
2024-11-25 09:58:41 +02:00
//
2023-10-18 16:21:57 +03:00
2024-10-10 22:57:42 +02:00
void common_batch_clear ( struct llama_batch & batch );
2023-10-18 16:21:57 +03:00
2024-10-10 22:57:42 +02:00
void common_batch_add (
2023-10-18 16:21:57 +03:00
struct llama_batch & batch ,
llama_token id ,
llama_pos pos ,
const std :: vector < llama_seq_id > & seq_ids ,
bool logits );
2024-11-25 09:58:41 +02:00
//
// Token utils
//
// longest common prefix
size_t common_lcp ( const llama_tokens & a , const llama_tokens & b );
// longet common subsequence
size_t common_lcs ( const llama_tokens & a , const llama_tokens & b );
2023-08-21 23:07:43 +03:00
//
// Vocab utils
//
2023-08-27 14:19:19 +03:00
// tokenizes a string into a vector of tokens
// should work similar to Python's `tokenizer.encode`
2024-10-10 22:57:42 +02:00
std :: vector < llama_token > common_tokenize (
2023-09-28 21:42:38 +02:00
const struct llama_context * ctx ,
const std :: string & text ,
2024-04-09 13:44:08 -04:00
bool add_special ,
bool parse_special = false );
2023-09-28 21:42:38 +02:00
2024-10-10 22:57:42 +02:00
std :: vector < llama_token > common_tokenize (
2023-09-28 21:42:38 +02:00
const struct llama_model * model ,
2023-08-21 23:07:43 +03:00
const std :: string & text ,
2024-04-09 13:44:08 -04:00
bool add_special ,
bool parse_special = false );
2023-08-21 23:07:43 +03:00
2024-04-24 05:15:29 -05:00
// tokenizes a token into a piece, optionally renders special/control tokens
2023-08-27 14:19:19 +03:00
// should work similar to Python's `tokenizer.id_to_piece`
2024-10-10 22:57:42 +02:00
std :: string common_token_to_piece (
2023-08-21 23:07:43 +03:00
const struct llama_context * ctx ,
2024-04-24 05:15:29 -05:00
llama_token token ,
bool special = true );
2023-08-27 14:19:19 +03:00
// detokenizes a vector of tokens into a string
// should work similar to Python's `tokenizer.decode`
2024-07-05 19:01:35 +02:00
// optionally renders special/control tokens
2024-10-10 22:57:42 +02:00
std :: string common_detokenize (
2023-08-27 14:19:19 +03:00
llama_context * ctx ,
2024-07-05 19:01:35 +02:00
const std :: vector < llama_token > & tokens ,
bool special = true );
2023-08-28 17:59:39 +02:00
2024-06-04 21:23:39 +03:00
//
// Chat template utils
//
2024-06-25 13:56:49 +02:00
// same with llama_chat_message, but uses std::string
2024-10-10 22:57:42 +02:00
struct common_chat_msg {
2024-06-25 13:56:49 +02:00
std :: string role ;
std :: string content ;
};
2024-06-04 21:23:39 +03:00
// Check if the template supplied via "--chat-template" is supported or not. Returns true if it's valid
2024-10-10 22:57:42 +02:00
bool common_chat_verify_template ( const std :: string & tmpl );
2024-06-04 21:23:39 +03:00
2024-06-25 13:56:49 +02:00
// CPP wrapper for llama_chat_apply_template
2024-06-27 18:14:19 +02:00
// If the built-in template is not supported, we default to chatml
// If the custom "tmpl" is not supported, we throw an error
2024-10-10 22:57:42 +02:00
std :: string common_chat_apply_template ( const struct llama_model * model ,
2024-06-25 13:56:49 +02:00
const std :: string & tmpl ,
2024-10-10 22:57:42 +02:00
const std :: vector < common_chat_msg > & chat ,
2024-06-25 13:56:49 +02:00
bool add_ass );
// Format single message, while taking into account the position of that message in chat history
2024-10-10 22:57:42 +02:00
std :: string common_chat_format_single ( const struct llama_model * model ,
2024-06-25 13:56:49 +02:00
const std :: string & tmpl ,
2024-10-10 22:57:42 +02:00
const std :: vector < common_chat_msg > & past_msg ,
const common_chat_msg & new_msg ,
2024-06-25 13:56:49 +02:00
bool add_ass );
// Returns an example of formatted chat
2024-10-10 22:57:42 +02:00
std :: string common_chat_format_example ( const struct llama_model * model ,
2024-06-25 13:56:49 +02:00
const std :: string & tmpl );
2023-11-23 19:07:56 +02:00
//
// KV cache utils
//
// Dump the KV cache view with the number of sequences per cell.
2024-10-10 22:57:42 +02:00
void common_kv_cache_dump_view ( const llama_kv_cache_view & view , int row_size = 80 );
2023-11-23 19:07:56 +02:00
// Dump the KV cache view showing individual sequences in each cell (long output).
2024-10-10 22:57:42 +02:00
void common_kv_cache_dump_view_seqs ( const llama_kv_cache_view & view , int row_size = 40 );
2024-03-09 21:27:58 +09:00
//
// Embedding utils
//
2024-10-10 22:57:42 +02:00
void common_embd_normalize ( const float * inp , float * out , int n , int embd_norm = 2 );
2024-03-09 21:27:58 +09:00
2024-10-10 22:57:42 +02:00
float common_embd_similarity_cos ( const float * embd1 , const float * embd2 , int n );
2024-03-15 13:43:02 -07:00
//
// Control vector utils
//
2024-10-10 22:57:42 +02:00
struct common_control_vector_data {
2024-03-15 13:43:02 -07:00
int n_embd ;
// stores data for layers [1, n_layer] where n_layer = data.size() / n_embd
std :: vector < float > data ;
};
2024-10-10 22:57:42 +02:00
struct common_control_vector_load_info {
2024-03-15 13:43:02 -07:00
float strength ;
std :: string fname ;
};
// Load control vectors, scale each by strength, and add them together.
// On error, returns {-1, empty}
2024-10-10 22:57:42 +02:00
common_control_vector_data common_control_vector_load ( const std :: vector < common_control_vector_load_info > & load_infos );
2024-03-23 18:07:00 +01:00
//
// Split utils
//
2024-05-22 20:04:20 +03:00
2024-03-23 18:07:00 +01:00
static const char * const LLM_KV_SPLIT_NO = "split.no" ;
static const char * const LLM_KV_SPLIT_COUNT = "split.count" ;
static const char * const LLM_KV_SPLIT_TENSORS_COUNT = "split.tensors.count" ;