2023-03-25 20:26:40 +02:00
#include "common.h"
2023-08-28 17:59:39 +02:00
#include "llama.h"
2023-03-24 08:19:05 -07:00
2023-03-22 07:32:36 +02:00
#include <algorithm>
2023-08-28 17:59:39 +02:00
#include <cassert>
#include <cmath>
#include <cstring>
#include <ctime>
#include <fstream>
#include <iterator>
#include <iostream>
2023-06-06 21:33:23 +02:00
#include <regex>
2023-08-28 17:59:39 +02:00
#include <sstream>
#include <string>
2023-11-23 19:07:56 +02:00
#include <unordered_map>
2023-08-28 17:59:39 +02:00
#include <unordered_set>
#include <vector>
2023-08-28 21:51:47 +02:00
#include <cinttypes>
2023-04-30 14:41:35 -04:00
#if defined(__APPLE__) && defined(__MACH__)
#include <sys/types.h>
#include <sys/sysctl.h>
#endif
2023-03-10 20:40:58 +02:00
2023-05-08 19:45:48 -07:00
#if defined(_WIN32)
#define WIN32_LEAN_AND_MEAN
2023-09-01 09:34:50 -04:00
#ifndef NOMINMAX
# define NOMINMAX
#endif
2023-08-28 17:59:39 +02:00
#include <codecvt>
#include <locale>
2023-05-08 19:45:48 -07:00
#include <windows.h>
2023-04-08 17:49:39 +02:00
#include <fcntl.h>
#include <io.h>
2023-05-08 19:45:48 -07:00
#else
#include <sys/ioctl.h>
2023-08-28 17:59:39 +02:00
#include <sys/stat.h>
2023-05-08 19:45:48 -07:00
#include <unistd.h>
2023-03-28 17:09:55 +03:00
#endif
2023-03-12 17:15:00 -03:00
2023-06-16 21:23:53 +03:00
#if defined(_MSC_VER)
#pragma warning(disable: 4244 4267) // possible loss of data
#endif
2023-04-30 14:41:35 -04:00
int32_t get_num_physical_cores () {
2023-03-17 17:47:35 +00:00
#ifdef __linux__
2023-05-14 22:25:42 -04:00
// enumerate the set of thread siblings, num entries is num cores
std :: unordered_set < std :: string > siblings ;
for ( uint32_t cpu = 0 ; cpu < UINT32_MAX ; ++ cpu ) {
std :: ifstream thread_siblings ( "/sys/devices/system/cpu"
+ std :: to_string ( cpu ) + "/topology/thread_siblings" );
if ( ! thread_siblings . is_open ()) {
break ; // no more cpus
2023-04-30 14:41:35 -04:00
}
2023-05-14 22:25:42 -04:00
std :: string line ;
if ( std :: getline ( thread_siblings , line )) {
siblings . insert ( line );
}
}
2023-09-07 13:22:29 -04:00
if ( ! siblings . empty ()) {
2023-05-14 22:25:42 -04:00
return static_cast < int32_t > ( siblings . size ());
2023-03-17 17:47:35 +00:00
}
2023-04-30 14:41:35 -04:00
#elif defined(__APPLE__) && defined(__MACH__)
int32_t num_physical_cores ;
size_t len = sizeof ( num_physical_cores );
int result = sysctlbyname ( "hw.perflevel0.physicalcpu" , & num_physical_cores , & len , NULL , 0 );
if ( result == 0 ) {
return num_physical_cores ;
}
result = sysctlbyname ( "hw.physicalcpu" , & num_physical_cores , & len , NULL , 0 );
if ( result == 0 ) {
return num_physical_cores ;
}
#elif defined(_WIN32)
//TODO: Implement
#endif
unsigned int n_threads = std :: thread :: hardware_concurrency ();
return n_threads > 0 ? ( n_threads <= 4 ? n_threads : n_threads / 2 ) : 4 ;
}
2023-03-17 17:47:35 +00:00
2023-09-28 20:40:11 +02:00
void process_escapes ( std :: string & input ) {
2023-05-04 05:08:25 -07:00
std :: size_t input_len = input . length ();
std :: size_t output_idx = 0 ;
2023-05-02 18:46:20 -07:00
2023-05-04 05:08:25 -07:00
for ( std :: size_t input_idx = 0 ; input_idx < input_len ; ++ input_idx ) {
if ( input [ input_idx ] == '\\' && input_idx + 1 < input_len ) {
switch ( input [ ++ input_idx ]) {
case 'n' : input [ output_idx ++ ] = '\n' ; break ;
case 'r' : input [ output_idx ++ ] = '\r' ; break ;
case 't' : input [ output_idx ++ ] = '\t' ; break ;
case '\'' : input [ output_idx ++ ] = '\'' ; break ;
case '\"' : input [ output_idx ++ ] = '\"' ; break ;
case '\\' : input [ output_idx ++ ] = '\\' ; break ;
2023-11-05 10:06:06 -07:00
case 'x' :
// Handle \x12, etc
if ( input_idx + 2 < input_len ) {
const char x [ 3 ] = { input [ input_idx + 1 ], input [ input_idx + 2 ], 0 };
char * err_p = nullptr ;
const long val = std :: strtol ( x , & err_p , 16 );
if ( err_p == x + 2 ) {
input_idx += 2 ;
input [ output_idx ++ ] = char ( val );
break ;
}
}
2023-11-05 18:45:16 +01:00
// fall through
2023-05-04 05:08:25 -07:00
default : input [ output_idx ++ ] = '\\' ;
input [ output_idx ++ ] = input [ input_idx ]; break ;
2023-05-02 18:46:20 -07:00
}
2023-05-04 05:08:25 -07:00
} else {
input [ output_idx ++ ] = input [ input_idx ];
2023-05-02 18:46:20 -07:00
}
}
2023-05-04 05:08:25 -07:00
input . resize ( output_idx );
2023-05-02 18:46:20 -07:00
}
2023-04-30 14:41:35 -04:00
bool gpt_params_parse ( int argc , char ** argv , gpt_params & params ) {
2023-11-01 14:42:01 -03:00
bool result = true ;
try {
if ( ! gpt_params_parse_ex ( argc , argv , params )) {
gpt_print_usage ( argc , argv , gpt_params ());
exit ( 0 );
}
}
2023-11-01 21:15:55 +02:00
catch ( const std :: invalid_argument & ex ) {
fprintf ( stderr , "%s \n " , ex . what ());
2023-11-01 14:42:01 -03:00
gpt_print_usage ( argc , argv , gpt_params ());
exit ( 1 );
}
return result ;
}
bool gpt_params_parse_ex ( int argc , char ** argv , gpt_params & params ) {
2023-03-23 19:54:28 +02:00
bool invalid_param = false ;
std :: string arg ;
2023-05-12 16:34:55 +02:00
const std :: string arg_prefix = "--" ;
2023-10-20 21:07:23 +03:00
llama_sampling_params & sparams = params . sparams ;
2023-04-01 23:41:12 -03:00
2023-03-10 20:40:58 +02:00
for ( int i = 1 ; i < argc ; i ++ ) {
2023-03-23 19:54:28 +02:00
arg = argv [ i ];
2023-05-12 16:34:55 +02:00
if ( arg . compare ( 0 , arg_prefix . size (), arg_prefix ) == 0 ) {
std :: replace ( arg . begin (), arg . end (), '_' , '-' );
}
2023-04-14 21:58:43 +02:00
if ( arg == "-s" || arg == "--seed" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-06-29 21:15:15 +08:00
params . seed = std :: stoul ( argv [ i ]);
2023-04-14 21:58:43 +02:00
} else if ( arg == "-t" || arg == "--threads" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_threads = std :: stoi ( argv [ i ]);
2023-07-23 21:33:02 +08:00
if ( params . n_threads <= 0 ) {
params . n_threads = std :: thread :: hardware_concurrency ();
}
2023-09-28 21:42:38 +02:00
} else if ( arg == "-tb" || arg == "--threads-batch" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_threads_batch = std :: stoi ( argv [ i ]);
if ( params . n_threads_batch <= 0 ) {
params . n_threads_batch = std :: thread :: hardware_concurrency ();
}
2023-04-14 21:58:43 +02:00
} else if ( arg == "-p" || arg == "--prompt" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-05-04 05:08:25 -07:00
params . prompt = argv [ i ];
2023-08-28 17:59:39 +02:00
} else if ( arg == "-e" || arg == "--escape" ) {
params . escape = true ;
2023-05-10 11:37:14 -04:00
} else if ( arg == "--prompt-cache" ) {
2023-04-28 11:59:37 -04:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-05-10 11:37:14 -04:00
params . path_prompt_cache = argv [ i ];
} else if ( arg == "--prompt-cache-all" ) {
params . prompt_cache_all = true ;
2023-06-07 04:10:17 +02:00
} else if ( arg == "--prompt-cache-ro" ) {
params . prompt_cache_ro = true ;
2023-04-14 21:58:43 +02:00
} else if ( arg == "-f" || arg == "--file" ) {
2023-03-23 19:54:28 +02:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
std :: ifstream file ( argv [ i ]);
2023-03-31 20:03:48 +02:00
if ( ! file ) {
fprintf ( stderr , "error: failed to open file '%s' \n " , argv [ i ]);
invalid_param = true ;
break ;
}
2023-10-06 14:16:38 +01:00
// store the external file name in params
params . prompt_file = argv [ i ];
2023-03-19 18:37:02 +02:00
std :: copy ( std :: istreambuf_iterator < char > ( file ), std :: istreambuf_iterator < char > (), back_inserter ( params . prompt ));
2023-10-07 15:31:41 -06:00
if ( ! params . prompt . empty () && params . prompt . back () == '\n' ) {
2023-03-19 19:04:44 +02:00
params . prompt . pop_back ();
}
2023-05-12 16:34:55 +02:00
} else if ( arg == "-n" || arg == "--n-predict" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . n_predict = std :: stoi ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--top-k" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . top_k = std :: stoi ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "-c" || arg == "--ctx-size" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . n_ctx = std :: stoi ( argv [ i ]);
2023-07-15 06:34:16 -04:00
} else if ( arg == "--rope-freq-base" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . rope_freq_base = std :: stof ( argv [ i ]);
} else if ( arg == "--rope-freq-scale" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . rope_freq_scale = std :: stof ( argv [ i ]);
2023-11-01 18:04:33 -04:00
} else if ( arg == "--rope-scaling" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
std :: string value ( argv [ i ]);
/**/ if ( value == "none" ) { params . rope_scaling_type = LLAMA_ROPE_SCALING_NONE ; }
else if ( value == "linear" ) { params . rope_scaling_type = LLAMA_ROPE_SCALING_LINEAR ; }
else if ( value == "yarn" ) { params . rope_scaling_type = LLAMA_ROPE_SCALING_YARN ; }
else { invalid_param = true ; break ; }
2023-08-07 19:07:19 +02:00
} else if ( arg == "--rope-scale" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . rope_freq_scale = 1.0f / std :: stof ( argv [ i ]);
2023-11-01 18:04:33 -04:00
} else if ( arg == "--yarn-orig-ctx" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . yarn_orig_ctx = std :: stoi ( argv [ i ]);
} else if ( arg == "--yarn-ext-factor" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . yarn_ext_factor = std :: stof ( argv [ i ]);
} else if ( arg == "--yarn-attn-factor" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . yarn_attn_factor = std :: stof ( argv [ i ]);
} else if ( arg == "--yarn-beta-fast" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . yarn_beta_fast = std :: stof ( argv [ i ]);
} else if ( arg == "--yarn-beta-slow" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . yarn_beta_slow = std :: stof ( argv [ i ]);
2023-12-05 15:05:51 +05:00
} else if ( arg == "--samplers" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
sparams . samplers_sequence = parse_samplers_input ( argv [ i ]);
} else if ( arg == "--sampling-seq" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
sparams . samplers_sequence = argv [ i ];
2023-05-12 16:34:55 +02:00
} else if ( arg == "--top-p" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . top_p = std :: stof ( argv [ i ]);
2023-10-31 14:44:49 -05:00
} else if ( arg == "--min-p" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
sparams . min_p = std :: stof ( argv [ i ]);
2023-03-10 20:40:58 +02:00
} else if ( arg == "--temp" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . temp = std :: stof ( argv [ i ]);
2023-10-28 14:23:11 +03:00
sparams . temp = std :: max ( sparams . temp , 0.0f );
2023-04-29 08:34:41 +03:00
} else if ( arg == "--tfs" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . tfs_z = std :: stof ( argv [ i ]);
2023-04-29 08:34:41 +03:00
} else if ( arg == "--typical" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . typical_p = std :: stof ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--repeat-last-n" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-10-20 21:07:23 +03:00
sparams . penalty_last_n = std :: stoi ( argv [ i ]);
sparams . n_prev = std :: max ( sparams . n_prev , sparams . penalty_last_n );
2023-05-12 16:34:55 +02:00
} else if ( arg == "--repeat-penalty" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-10-20 21:07:23 +03:00
sparams . penalty_repeat = std :: stof ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--frequency-penalty" ) {
2023-04-29 08:34:41 +03:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-20 21:07:23 +03:00
sparams . penalty_freq = std :: stof ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--presence-penalty" ) {
2023-04-29 08:34:41 +03:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-20 21:07:23 +03:00
sparams . penalty_present = std :: stof ( argv [ i ]);
2023-04-29 08:34:41 +03:00
} else if ( arg == "--mirostat" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . mirostat = std :: stoi ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--mirostat-lr" ) {
2023-04-29 08:34:41 +03:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . mirostat_eta = std :: stof ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "--mirostat-ent" ) {
2023-04-29 08:34:41 +03:00
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . mirostat_tau = std :: stof ( argv [ i ]);
2023-07-12 00:18:43 +08:00
} else if ( arg == "--cfg-negative-prompt" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . cfg_negative_prompt = argv [ i ];
2023-08-17 07:29:44 -06:00
} else if ( arg == "--cfg-negative-prompt-file" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
std :: ifstream file ( argv [ i ]);
if ( ! file ) {
fprintf ( stderr , "error: failed to open file '%s' \n " , argv [ i ]);
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
std :: copy ( std :: istreambuf_iterator < char > ( file ), std :: istreambuf_iterator < char > (), back_inserter ( sparams . cfg_negative_prompt ));
if ( ! sparams . cfg_negative_prompt . empty () && sparams . cfg_negative_prompt . back () == '\n' ) {
sparams . cfg_negative_prompt . pop_back ();
2023-08-17 07:29:44 -06:00
}
2023-07-12 00:18:43 +08:00
} else if ( arg == "--cfg-scale" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-11 13:35:46 -06:00
sparams . cfg_scale = std :: stof ( argv [ i ]);
2023-05-12 16:34:55 +02:00
} else if ( arg == "-b" || arg == "--batch-size" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . n_batch = std :: stoi ( argv [ i ]);
2023-03-25 21:36:22 +02:00
} else if ( arg == "--keep" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-25 21:36:22 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . n_keep = std :: stoi ( argv [ i ]);
2023-09-03 15:12:08 +03:00
} else if ( arg == "--draft" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_draft = std :: stoi ( argv [ i ]);
2023-07-18 14:24:43 +03:00
} else if ( arg == "--chunks" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_chunks = std :: stoi ( argv [ i ]);
2023-09-28 19:04:36 +03:00
} else if ( arg == "-np" || arg == "--parallel" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_parallel = std :: stoi ( argv [ i ]);
} else if ( arg == "-ns" || arg == "--sequences" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . n_sequences = std :: stoi ( argv [ i ]);
2023-11-03 09:41:17 +02:00
} else if ( arg == "--p-accept" || arg == "-pa" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . p_accept = std :: stof ( argv [ i ]);
} else if ( arg == "--p-split" || arg == "-ps" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . p_split = std :: stof ( argv [ i ]);
2023-03-10 20:40:58 +02:00
} else if ( arg == "-m" || arg == "--model" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . model = argv [ i ];
2023-09-03 15:12:08 +03:00
} else if ( arg == "-md" || arg == "--model-draft" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . model_draft = argv [ i ];
2023-05-28 20:14:24 +03:00
} else if ( arg == "-a" || arg == "--alias" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . model_alias = argv [ i ];
2023-04-17 17:28:55 +02:00
} else if ( arg == "--lora" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-06 01:16:39 +08:00
params . lora_adapter . push_back ( std :: make_tuple ( argv [ i ], 1.0f ));
2023-09-28 20:40:11 +02:00
params . use_mmap = false ;
} else if ( arg == "--lora-scaled" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
const char * lora_adapter = argv [ i ];
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-06 01:16:39 +08:00
params . lora_adapter . push_back ( std :: make_tuple ( lora_adapter , std :: stof ( argv [ i ])));
2023-07-13 21:58:25 +08:00
params . use_mmap = false ;
2023-04-17 17:28:55 +02:00
} else if ( arg == "--lora-base" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . lora_base = argv [ i ];
2023-10-12 18:23:18 +03:00
} else if ( arg == "--mmproj" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . mmproj = argv [ i ];
} else if ( arg == "--image" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . image = argv [ i ];
2023-03-12 22:13:28 +01:00
} else if ( arg == "-i" || arg == "--interactive" ) {
params . interactive = true ;
2023-03-24 08:05:13 -07:00
} else if ( arg == "--embedding" ) {
params . embedding = true ;
2023-03-22 18:16:35 +01:00
} else if ( arg == "--interactive-first" ) {
2023-04-24 17:45:32 +02:00
params . interactive_first = true ;
2023-03-19 18:37:02 +02:00
} else if ( arg == "-ins" || arg == "--instruct" ) {
2023-04-14 21:58:43 +02:00
params . instruct = true ;
2023-11-21 00:26:59 +10:30
} else if ( arg == "-cml" || arg == "--chatml" ) {
params . chatml = true ;
2023-10-02 09:42:02 +02:00
} else if ( arg == "--infill" ) {
params . infill = true ;
2023-11-23 19:07:56 +02:00
} else if ( arg == "-dkvc" || arg == "--dump-kv-cache" ) {
params . dump_kv_cache = true ;
2023-12-07 13:03:17 +02:00
} else if ( arg == "-nkvo" || arg == "--no-kv-offload" ) {
params . no_kv_offload = true ;
} else if ( arg == "-ctk" || arg == "--cache-type-k" ) {
params . cache_type_k = argv [ ++ i ];
} else if ( arg == "-ctv" || arg == "--cache-type-v" ) {
params . cache_type_v = argv [ ++ i ];
2023-05-08 19:45:48 -07:00
} else if ( arg == "--multiline-input" ) {
params . multiline_input = true ;
2023-08-04 08:20:12 -07:00
} else if ( arg == "--simple-io" ) {
params . simple_io = true ;
2023-09-28 19:04:36 +03:00
} else if ( arg == "-cb" || arg == "--cont-batching" ) {
params . cont_batching = true ;
2023-03-12 22:13:28 +01:00
} else if ( arg == "--color" ) {
params . use_color = true ;
2023-03-24 08:19:05 -07:00
} else if ( arg == "--mlock" ) {
params . use_mlock = true ;
2023-05-13 15:38:36 +02:00
} else if ( arg == "--gpu-layers" || arg == "-ngl" || arg == "--n-gpu-layers" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-05-28 11:48:57 -06:00
#ifdef LLAMA_SUPPORTS_GPU_OFFLOAD
2023-05-13 15:38:36 +02:00
params . n_gpu_layers = std :: stoi ( argv [ i ]);
2023-05-28 11:48:57 -06:00
#else
fprintf ( stderr , "warning: not compiled with GPU offload support, --n-gpu-layers option will be ignored \n " );
fprintf ( stderr , "warning: see main README.md for information on enabling GPU BLAS support \n " );
2023-09-13 08:50:46 +02:00
#endif
} else if ( arg == "--gpu-layers-draft" || arg == "-ngld" || arg == "--n-gpu-layers-draft" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
#ifdef LLAMA_SUPPORTS_GPU_OFFLOAD
params . n_gpu_layers_draft = std :: stoi ( argv [ i ]);
#else
fprintf ( stderr , "warning: not compiled with GPU offload support, --n-gpu-layers-draft option will be ignored \n " );
fprintf ( stderr , "warning: see main README.md for information on enabling GPU BLAS support \n " );
2023-05-28 11:48:57 -06:00
#endif
2023-06-06 21:33:23 +02:00
} else if ( arg == "--main-gpu" || arg == "-mg" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
#ifdef GGML_USE_CUBLAS
params . main_gpu = std :: stoi ( argv [ i ]);
#else
2023-07-31 15:44:35 +02:00
fprintf ( stderr , "warning: llama.cpp was compiled without cuBLAS. It is not possible to set a main GPU. \n " );
2023-06-06 21:33:23 +02:00
#endif
} else if ( arg == "--tensor-split" || arg == "-ts" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
#ifdef GGML_USE_CUBLAS
std :: string arg_next = argv [ i ];
// split string by , and /
const std :: regex regex { R "([,/]+)" };
std :: sregex_token_iterator it { arg_next . begin (), arg_next . end (), regex , - 1 };
std :: vector < std :: string > split_arg { it , {}};
GGML_ASSERT ( split_arg . size () <= LLAMA_MAX_DEVICES );
for ( size_t i = 0 ; i < LLAMA_MAX_DEVICES ; ++ i ) {
if ( i < split_arg . size ()) {
params . tensor_split [ i ] = std :: stof ( split_arg [ i ]);
} else {
params . tensor_split [ i ] = 0.0f ;
}
}
#else
2023-07-31 15:44:35 +02:00
fprintf ( stderr , "warning: llama.cpp was compiled without cuBLAS. It is not possible to set a tensor split. \n " );
#endif // GGML_USE_CUBLAS
2023-08-22 22:47:05 +02:00
} else if ( arg == "--no-mul-mat-q" || arg == "-nommq" ) {
2023-07-31 15:44:35 +02:00
#ifdef GGML_USE_CUBLAS
2023-08-22 22:47:05 +02:00
params . mul_mat_q = false ;
2023-07-31 15:44:35 +02:00
#else
2023-08-22 22:47:05 +02:00
fprintf ( stderr , "warning: llama.cpp was compiled without cuBLAS. Disabling mul_mat_q kernels has no effect. \n " );
2023-06-06 21:33:23 +02:00
#endif // GGML_USE_CUBLAS
2023-04-08 12:24:37 -07:00
} else if ( arg == "--no-mmap" ) {
params . use_mmap = false ;
2023-06-26 13:57:59 -04:00
} else if ( arg == "--numa" ) {
params . numa = true ;
2023-03-25 21:36:22 +02:00
} else if ( arg == "--verbose-prompt" ) {
2023-03-25 17:16:50 +02:00
params . verbose_prompt = true ;
2023-03-12 22:13:28 +01:00
} else if ( arg == "-r" || arg == "--reverse-prompt" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-23 19:54:28 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . antiprompt . push_back ( argv [ i ]);
2023-08-28 17:59:39 +02:00
} else if ( arg == "-ld" || arg == "--logdir" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . logdir = argv [ i ];
if ( params . logdir . back () != DIRECTORY_SEPARATOR ) {
params . logdir += DIRECTORY_SEPARATOR ;
}
2023-09-28 19:04:36 +03:00
} else if ( arg == "--perplexity" || arg == "--all-logits" ) {
params . logits_all = true ;
2023-08-23 12:56:42 +03:00
} else if ( arg == "--ppl-stride" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . ppl_stride = std :: stoi ( argv [ i ]);
} else if ( arg == "--ppl-output-type" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . ppl_output_type = std :: stoi ( argv [ i ]);
2023-07-28 20:25:36 +02:00
} else if ( arg == "--hellaswag" ) {
params . hellaswag = true ;
} else if ( arg == "--hellaswag-tasks" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . hellaswag_tasks = std :: stoi ( argv [ i ]);
2023-03-19 19:22:48 +01:00
} else if ( arg == "--ignore-eos" ) {
2023-08-21 23:07:43 +03:00
params . ignore_eos = true ;
2023-04-29 08:34:41 +03:00
} else if ( arg == "--no-penalize-nl" ) {
2023-10-11 13:35:46 -06:00
sparams . penalize_nl = false ;
2023-04-29 08:34:41 +03:00
} else if ( arg == "-l" || arg == "--logit-bias" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
std :: stringstream ss ( argv [ i ]);
llama_token key ;
char sign ;
std :: string value_str ;
try {
if ( ss >> key && ss >> sign && std :: getline ( ss , value_str ) && ( sign == '+' || sign == '-' )) {
2023-10-11 13:35:46 -06:00
sparams . logit_bias [ key ] = std :: stof ( value_str ) * (( sign == '-' ) ? - 1.0f : 1.0f );
2023-04-29 08:34:41 +03:00
} else {
throw std :: exception ();
}
2023-06-16 21:23:53 +03:00
} catch ( const std :: exception & ) {
2023-04-29 08:34:41 +03:00
invalid_param = true ;
break ;
}
2023-03-10 20:40:58 +02:00
} else if ( arg == "-h" || arg == "--help" ) {
2023-11-01 14:42:01 -03:00
return false ;
2023-12-13 20:50:14 +08:00
} else if ( arg == "--version" ) {
fprintf ( stderr , "version: %d (%s) \n " , LLAMA_BUILD_NUMBER , LLAMA_COMMIT );
fprintf ( stderr , "built with %s for %s \n " , LLAMA_COMPILER , LLAMA_BUILD_TARGET );
exit ( 0 );
2023-03-19 19:36:19 +01:00
} else if ( arg == "--random-prompt" ) {
params . random_prompt = true ;
2023-07-25 07:19:11 -05:00
} else if ( arg == "--in-prefix-bos" ) {
params . input_prefix_bos = true ;
2023-03-25 14:03:19 +02:00
} else if ( arg == "--in-prefix" ) {
2023-04-14 21:58:43 +02:00
if ( ++ i >= argc ) {
2023-03-25 14:42:09 +02:00
invalid_param = true ;
break ;
}
2023-04-14 21:58:43 +02:00
params . input_prefix = argv [ i ];
2023-05-04 23:41:12 +08:00
} else if ( arg == "--in-suffix" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
params . input_suffix = argv [ i ];
2023-07-23 23:58:10 -04:00
} else if ( arg == "--grammar" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
2023-10-20 21:07:23 +03:00
sparams . grammar = argv [ i ];
2023-07-23 23:58:10 -04:00
} else if ( arg == "--grammar-file" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
std :: ifstream file ( argv [ i ]);
if ( ! file ) {
fprintf ( stderr , "error: failed to open file '%s' \n " , argv [ i ]);
invalid_param = true ;
break ;
}
std :: copy (
std :: istreambuf_iterator < char > ( file ),
std :: istreambuf_iterator < char > (),
2023-10-20 21:07:23 +03:00
std :: back_inserter ( sparams . grammar )
2023-07-23 23:58:10 -04:00
);
2023-12-05 10:19:18 -07:00
} else if ( arg == "--override-kv" ) {
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
char * sep = strchr ( argv [ i ], '=' );
if ( sep == nullptr || sep - argv [ i ] >= 128 ) {
fprintf ( stderr , "error: Malformed KV override: %s \n " , argv [ i ]);
invalid_param = true ;
break ;
}
struct llama_model_kv_override kvo ;
std :: strncpy ( kvo . key , argv [ i ], sep - argv [ i ]);
kvo . key [ sep - argv [ i ]] = 0 ;
sep ++ ;
if ( strncmp ( sep , "int:" , 4 ) == 0 ) {
sep += 4 ;
kvo . tag = LLAMA_KV_OVERRIDE_INT ;
kvo . int_value = std :: atol ( sep );
} else if ( strncmp ( sep , "float:" , 6 ) == 0 ) {
sep += 6 ;
kvo . tag = LLAMA_KV_OVERRIDE_FLOAT ;
kvo . float_value = std :: atof ( sep );
} else if ( strncmp ( sep , "bool:" , 5 ) == 0 ) {
sep += 5 ;
kvo . tag = LLAMA_KV_OVERRIDE_BOOL ;
if ( std :: strcmp ( sep , "true" ) == 0 ) {
kvo . bool_value = true ;
} else if ( std :: strcmp ( sep , "false" ) == 0 ) {
kvo . bool_value = false ;
} else {
fprintf ( stderr , "error: Invalid boolean value for KV override: %s \n " , argv [ i ]);
invalid_param = true ;
break ;
}
} else {
fprintf ( stderr , "error: Invalid type for KV override: %s \n " , argv [ i ]);
invalid_param = true ;
break ;
}
params . kv_overrides . push_back ( kvo );
2023-08-30 08:29:32 +02:00
#ifndef LOG_DISABLE_LOGS
// Parse args for logging parameters
} else if ( log_param_single_parse ( argv [ i ] ) ) {
// Do nothing, log_param_single_parse automatically does it's thing
// and returns if a match was found and parsed.
} else if ( log_param_pair_parse ( /*check_but_dont_parse*/ true , argv [ i ] ) ) {
// We have a matching known parameter requiring an argument,
// now we need to check if there is anything after this argv
// and flag invalid_param or parse it.
if ( ++ i >= argc ) {
invalid_param = true ;
break ;
}
if ( ! log_param_pair_parse ( /*check_but_dont_parse*/ false , argv [ i - 1 ], argv [ i ]) ) {
invalid_param = true ;
break ;
}
// End of Parse args for logging parameters
#endif // LOG_DISABLE_LOGS
2023-03-10 20:40:58 +02:00
} else {
2023-11-01 14:42:01 -03:00
throw std :: invalid_argument ( "error: unknown argument: " + arg );
2023-03-10 20:40:58 +02:00
}
}
2023-03-23 19:54:28 +02:00
if ( invalid_param ) {
2023-11-01 14:42:01 -03:00
throw std :: invalid_argument ( "error: invalid parameter for argument: " + arg );
2023-03-23 19:54:28 +02:00
}
2023-05-10 11:37:14 -04:00
if ( params . prompt_cache_all &&
( params . interactive || params . interactive_first ||
2023-05-19 10:24:59 -07:00
params . instruct )) {
2023-11-01 14:42:01 -03:00
throw std :: invalid_argument ( "error: --prompt-cache-all not supported in interactive mode yet \n " );
2023-05-10 11:37:14 -04:00
}
2023-06-15 19:06:46 +02:00
2023-08-28 17:59:39 +02:00
if ( params . escape ) {
2023-05-04 05:08:25 -07:00
process_escapes ( params . prompt );
2023-07-09 03:56:18 -05:00
process_escapes ( params . input_prefix );
process_escapes ( params . input_suffix );
2023-10-22 20:09:51 +02:00
process_escapes ( sparams . cfg_negative_prompt );
2023-10-05 18:17:29 +02:00
for ( auto & antiprompt : params . antiprompt ) {
process_escapes ( antiprompt );
}
2023-05-04 05:08:25 -07:00
}
2023-03-10 20:40:58 +02:00
2023-12-05 10:19:18 -07:00
if ( ! params . kv_overrides . empty ()) {
params . kv_overrides . emplace_back ( llama_model_kv_override ());
params . kv_overrides . back (). key [ 0 ] = 0 ;
}
2023-03-10 20:40:58 +02:00
return true ;
}
2023-04-14 21:58:43 +02:00
void gpt_print_usage ( int /*argc*/ , char ** argv , const gpt_params & params ) {
2023-10-20 21:07:23 +03:00
const llama_sampling_params & sparams = params . sparams ;
2023-10-11 13:35:46 -06:00
2023-11-01 14:42:01 -03:00
printf ( " \n " );
2023-09-05 15:10:27 -04:00
printf ( "usage: %s [options] \n " , argv [ 0 ]);
printf ( " \n " );
printf ( "options: \n " );
printf ( " -h, --help show this help message and exit \n " );
2023-12-13 20:50:14 +08:00
printf ( " --version show version and build info \n " );
2023-09-05 15:10:27 -04:00
printf ( " -i, --interactive run in interactive mode \n " );
printf ( " --interactive-first run in interactive mode and wait for input right away \n " );
printf ( " -ins, --instruct run in instruction mode (use with Alpaca models) \n " );
2023-11-21 00:26:59 +10:30
printf ( " -cml, --chatml run in chatml mode (use with ChatML-compatible models) \n " );
2023-09-05 15:10:27 -04:00
printf ( " --multiline-input allows you to write or paste multiple lines without ending each in ' \\ ' \n " );
printf ( " -r PROMPT, --reverse-prompt PROMPT \n " );
printf ( " halt generation at PROMPT, return control in interactive mode \n " );
printf ( " (can be specified more than once for multiple prompts). \n " );
printf ( " --color colorise output to distinguish prompt and user input from generations \n " );
printf ( " -s SEED, --seed SEED RNG seed (default: -1, use random seed for < 0) \n " );
2023-09-28 21:42:38 +02:00
printf ( " -t N, --threads N number of threads to use during generation (default: %d) \n " , params . n_threads );
printf ( " -tb N, --threads-batch N \n " );
printf ( " number of threads to use during batch and prompt processing (default: same as --threads) \n " );
2023-09-05 15:10:27 -04:00
printf ( " -p PROMPT, --prompt PROMPT \n " );
printf ( " prompt to start generation with (default: empty) \n " );
printf ( " -e, --escape process prompt escapes sequences ( \\ n, \\ r, \\ t, \\ ', \\\" , \\\\ ) \n " );
printf ( " --prompt-cache FNAME file to cache prompt state for faster startup (default: none) \n " );
printf ( " --prompt-cache-all if specified, saves user input and generations to cache as well. \n " );
printf ( " not supported with --interactive or other interactive options \n " );
printf ( " --prompt-cache-ro if specified, uses the prompt cache but does not update it. \n " );
printf ( " --random-prompt start with a randomized prompt. \n " );
printf ( " --in-prefix-bos prefix BOS to user inputs, preceding the `--in-prefix` string \n " );
printf ( " --in-prefix STRING string to prefix user inputs with (default: empty) \n " );
printf ( " --in-suffix STRING string to suffix after user inputs with (default: empty) \n " );
printf ( " -f FNAME, --file FNAME \n " );
printf ( " prompt file to start generation. \n " );
printf ( " -n N, --n-predict N number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled) \n " , params . n_predict );
2023-09-28 21:42:38 +02:00
printf ( " -c N, --ctx-size N size of the prompt context (default: %d, 0 = loaded from model) \n " , params . n_ctx );
2023-09-05 15:10:27 -04:00
printf ( " -b N, --batch-size N batch size for prompt processing (default: %d) \n " , params . n_batch );
2023-12-05 15:05:51 +05:00
printf ( " --samplers samplers that will be used for generation in the order, separated by \' ; \' , for example: \" top_k;tfs;typical;top_p;min_p;temp \"\n " );
printf ( " --sampling-seq simplified sequence for samplers that will be used (default: %s) \n " , sparams . samplers_sequence . c_str ());
2023-10-11 13:35:46 -06:00
printf ( " --top-k N top-k sampling (default: %d, 0 = disabled) \n " , sparams . top_k );
printf ( " --top-p N top-p sampling (default: %.1f, 1.0 = disabled) \n " , ( double ) sparams . top_p );
2023-10-31 14:44:49 -05:00
printf ( " --min-p N min-p sampling (default: %.1f, 0.0 = disabled) \n " , ( double ) sparams . min_p );
2023-10-11 13:35:46 -06:00
printf ( " --tfs N tail free sampling, parameter z (default: %.1f, 1.0 = disabled) \n " , ( double ) sparams . tfs_z );
printf ( " --typical N locally typical sampling, parameter p (default: %.1f, 1.0 = disabled) \n " , ( double ) sparams . typical_p );
2023-10-20 21:07:23 +03:00
printf ( " --repeat-last-n N last n tokens to consider for penalize (default: %d, 0 = disabled, -1 = ctx_size) \n " , sparams . penalty_last_n );
printf ( " --repeat-penalty N penalize repeat sequence of tokens (default: %.1f, 1.0 = disabled) \n " , ( double ) sparams . penalty_repeat );
printf ( " --presence-penalty N repeat alpha presence penalty (default: %.1f, 0.0 = disabled) \n " , ( double ) sparams . penalty_present );
printf ( " --frequency-penalty N repeat alpha frequency penalty (default: %.1f, 0.0 = disabled) \n " , ( double ) sparams . penalty_freq );
2023-09-05 15:10:27 -04:00
printf ( " --mirostat N use Mirostat sampling. \n " );
printf ( " Top K, Nucleus, Tail Free and Locally Typical samplers are ignored if used. \n " );
2023-10-11 13:35:46 -06:00
printf ( " (default: %d, 0 = disabled, 1 = Mirostat, 2 = Mirostat 2.0) \n " , sparams . mirostat );
printf ( " --mirostat-lr N Mirostat learning rate, parameter eta (default: %.1f) \n " , ( double ) sparams . mirostat_eta );
printf ( " --mirostat-ent N Mirostat target entropy, parameter tau (default: %.1f) \n " , ( double ) sparams . mirostat_tau );
2023-09-05 15:10:27 -04:00
printf ( " -l TOKEN_ID(+/-)BIAS, --logit-bias TOKEN_ID(+/-)BIAS \n " );
printf ( " modifies the likelihood of token appearing in the completion, \n " );
printf ( " i.e. `--logit-bias 15043+1` to increase likelihood of token ' Hello', \n " );
printf ( " or `--logit-bias 15043-1` to decrease likelihood of token ' Hello' \n " );
printf ( " --grammar GRAMMAR BNF-like grammar to constrain generations (see samples in grammars/ dir) \n " );
printf ( " --grammar-file FNAME file to read grammar from \n " );
printf ( " --cfg-negative-prompt PROMPT \n " );
printf ( " negative prompt to use for guidance. (default: empty) \n " );
printf ( " --cfg-negative-prompt-file FNAME \n " );
printf ( " negative prompt file to use for guidance. (default: empty) \n " );
2023-10-11 13:35:46 -06:00
printf ( " --cfg-scale N strength of guidance (default: %f, 1.0 = disable) \n " , sparams . cfg_scale );
2023-11-01 18:04:33 -04:00
printf ( " --rope-scaling {none,linear,yarn} \n " );
printf ( " RoPE frequency scaling method, defaults to linear unless specified by the model \n " );
printf ( " --rope-scale N RoPE context scaling factor, expands context by a factor of N \n " );
2023-09-20 12:12:47 -04:00
printf ( " --rope-freq-base N RoPE base frequency, used by NTK-aware scaling (default: loaded from model) \n " );
2023-11-01 18:04:33 -04:00
printf ( " --rope-freq-scale N RoPE frequency scaling factor, expands context by a factor of 1/N \n " );
printf ( " --yarn-orig-ctx N YaRN: original context size of model (default: 0 = model training context size) \n " );
printf ( " --yarn-ext-factor N YaRN: extrapolation mix factor (default: 1.0, 0.0 = full interpolation) \n " );
printf ( " --yarn-attn-factor N YaRN: scale sqrt(t) or attention magnitude (default: 1.0) \n " );
printf ( " --yarn-beta-slow N YaRN: high correction dim or alpha (default: %.1f) \n " , params . yarn_beta_slow );
printf ( " --yarn-beta-fast N YaRN: low correction dim or beta (default: %.1f) \n " , params . yarn_beta_fast );
2023-09-05 15:10:27 -04:00
printf ( " --ignore-eos ignore end of stream token and continue generating (implies --logit-bias 2-inf) \n " );
printf ( " --no-penalize-nl do not penalize newline token \n " );
2023-10-11 13:35:46 -06:00
printf ( " --temp N temperature (default: %.1f) \n " , ( double ) sparams . temp );
2023-09-28 19:04:36 +03:00
printf ( " --logits-all return logits for all tokens in the batch (default: disabled) \n " );
2023-09-05 15:10:27 -04:00
printf ( " --hellaswag compute HellaSwag score over random tasks from datafile supplied with -f \n " );
printf ( " --hellaswag-tasks N number of tasks to use when computing the HellaSwag score (default: %zu) \n " , params . hellaswag_tasks );
printf ( " --keep N number of tokens to keep from the initial prompt (default: %d, -1 = all) \n " , params . n_keep );
printf ( " --draft N number of tokens to draft for speculative decoding (default: %d) \n " , params . n_draft );
printf ( " --chunks N max number of chunks to process (default: %d, -1 = all) \n " , params . n_chunks );
2023-09-28 19:04:36 +03:00
printf ( " -np N, --parallel N number of parallel sequences to decode (default: %d) \n " , params . n_parallel );
printf ( " -ns N, --sequences N number of sequences to decode (default: %d) \n " , params . n_sequences );
2023-11-03 09:41:17 +02:00
printf ( " -pa N, --p-accept N speculative decoding accept probability (default: %.1f) \n " , ( double ) params . p_accept );
printf ( " -ps N, --p-split N speculative decoding split probability (default: %.1f) \n " , ( double ) params . p_split );
2023-09-28 19:04:36 +03:00
printf ( " -cb, --cont-batching enable continuous batching (a.k.a dynamic batching) (default: disabled) \n " );
2023-10-12 18:23:18 +03:00
printf ( " --mmproj MMPROJ_FILE path to a multimodal projector file for LLaVA. see examples/llava/README.md \n " );
printf ( " --image IMAGE_FILE path to an image file. use with multimodal models \n " );
2023-04-08 12:24:37 -07:00
if ( llama_mlock_supported ()) {
2023-09-05 15:10:27 -04:00
printf ( " --mlock force system to keep model in RAM rather than swapping or compressing \n " );
2023-03-24 08:19:05 -07:00
}
2023-04-08 12:24:37 -07:00
if ( llama_mmap_supported ()) {
2023-09-05 15:10:27 -04:00
printf ( " --no-mmap do not memory-map model (slower load but may reduce pageouts if not using mlock) \n " );
2023-04-08 12:24:37 -07:00
}
2023-09-05 15:10:27 -04:00
printf ( " --numa attempt optimizations that help on some NUMA systems \n " );
printf ( " if run without this previously, it is recommended to drop the system page cache before using this \n " );
printf ( " see https://github.com/ggerganov/llama.cpp/issues/1437 \n " );
2023-05-28 11:48:57 -06:00
#ifdef LLAMA_SUPPORTS_GPU_OFFLOAD
2023-09-05 15:10:27 -04:00
printf ( " -ngl N, --n-gpu-layers N \n " );
printf ( " number of layers to store in VRAM \n " );
2023-09-13 08:50:46 +02:00
printf ( " -ngld N, --n-gpu-layers-draft N \n " );
printf ( " number of layers to store in VRAM for the draft model \n " );
2023-09-05 15:10:27 -04:00
printf ( " -ts SPLIT --tensor-split SPLIT \n " );
printf ( " how to split tensors across multiple GPUs, comma-separated list of proportions, e.g. 3,1 \n " );
printf ( " -mg i, --main-gpu i the GPU to use for scratch and small tensors \n " );
2023-08-25 12:09:42 +03:00
#ifdef GGML_USE_CUBLAS
2023-09-05 15:10:27 -04:00
printf ( " -nommq, --no-mul-mat-q \n " );
printf ( " use " GGML_CUBLAS_NAME " instead of custom mul_mat_q " GGML_CUDA_NAME " kernels. \n " );
printf ( " Not recommended since this is both slower and uses more VRAM. \n " );
2023-08-25 12:09:42 +03:00
#endif // GGML_USE_CUBLAS
2023-05-28 11:48:57 -06:00
#endif
2023-09-05 15:10:27 -04:00
printf ( " --verbose-prompt print prompt before generation \n " );
2023-11-23 19:07:56 +02:00
printf ( " -dkvc, --dump-kv-cache \n " );
printf ( " verbose print of the KV cache \n " );
2023-12-07 13:03:17 +02:00
printf ( " -nkvo, --no-kv-offload \n " );
printf ( " disable KV offload \n " );
printf ( " -ctk TYPE, --cache-type-k TYPE \n " );
printf ( " KV cache data type for K (default: %s) \n " , params . cache_type_k . c_str ());
printf ( " -ctv TYPE, --cache-type-v TYPE \n " );
printf ( " KV cache data type for V (default: %s) \n " , params . cache_type_v . c_str ());
2023-10-28 12:16:33 +02:00
printf ( " --simple-io use basic IO for better compatibility in subprocesses and limited consoles \n " );
2023-09-05 15:10:27 -04:00
printf ( " --lora FNAME apply LoRA adapter (implies --no-mmap) \n " );
2023-09-28 20:40:11 +02:00
printf ( " --lora-scaled FNAME S apply LoRA adapter with user defined scaling S (implies --no-mmap) \n " );
2023-09-05 15:10:27 -04:00
printf ( " --lora-base FNAME optional model to use as a base for the layers modified by the LoRA adapter \n " );
printf ( " -m FNAME, --model FNAME \n " );
printf ( " model path (default: %s) \n " , params . model . c_str ());
printf ( " -md FNAME, --model-draft FNAME \n " );
2023-12-21 12:55:34 -05:00
printf ( " draft model for speculative decoding \n " );
2023-09-05 15:10:27 -04:00
printf ( " -ld LOGDIR, --logdir LOGDIR \n " );
printf ( " path under which to save YAML logs (no logging if unset) \n " );
2023-12-05 10:19:18 -07:00
printf ( " --override-kv KEY=TYPE:VALUE \n " );
printf ( " advanced option to override model metadata by key. may be specified multiple times. \n " );
printf ( " types: int, float, bool. example: --override-kv tokenizer.ggml.add_bos_token=bool:false \n " );
2023-09-05 15:10:27 -04:00
printf ( " \n " );
2023-11-01 14:42:01 -03:00
#ifndef LOG_DISABLE_LOGS
log_print_usage ();
#endif // LOG_DISABLE_LOGS
2023-03-10 20:40:58 +02:00
}
2023-09-28 21:42:38 +02:00
std :: string get_system_info ( const gpt_params & params ) {
std :: ostringstream os ;
os << "system_info: n_threads = " << params . n_threads ;
if ( params . n_threads_batch != - 1 ) {
os << " (n_threads_batch = " << params . n_threads_batch << ")" ;
}
os << " / " << std :: thread :: hardware_concurrency () << " | " << llama_print_system_info ();
return os . str ();
}
2023-03-10 20:40:58 +02:00
std :: string gpt_random_prompt ( std :: mt19937 & rng ) {
const int r = rng () % 10 ;
switch ( r ) {
case 0 : return "So" ;
case 1 : return "Once upon a time" ;
case 2 : return "When" ;
case 3 : return "The" ;
case 4 : return "After" ;
case 5 : return "If" ;
case 6 : return "import" ;
case 7 : return "He" ;
case 8 : return "She" ;
case 9 : return "They" ;
}
2023-09-28 17:41:44 -04:00
GGML_UNREACHABLE ();
2023-03-10 20:40:58 +02:00
}
2023-12-05 15:05:51 +05:00
//
// String parsing
//
std :: string parse_samplers_input ( std :: string input ) {
std :: string output = "" ;
// since samplers names are written multiple ways
// make it ready for both system names and input names
std :: unordered_map < std :: string , char > samplers_symbols {
{ "top_k" , 'k' },
{ "top-k" , 'k' },
{ "top_p" , 'p' },
{ "top-p" , 'p' },
{ "nucleus" , 'p' },
{ "typical_p" , 'y' },
{ "typical-p" , 'y' },
{ "typical" , 'y' },
{ "min_p" , 'm' },
{ "min-p" , 'm' },
{ "tfs_z" , 'f' },
{ "tfs-z" , 'f' },
{ "tfs" , 'f' },
{ "temp" , 't' },
{ "temperature" , 't' }
};
// expected format example: "temp;top_k;tfs_z;typical_p;top_p;min_p"
size_t separator = input . find ( ';' );
while ( separator != input . npos ) {
std :: string name = input . substr ( 0 , separator );
input = input . substr ( separator + 1 );
separator = input . find ( ';' );
if ( samplers_symbols . find ( name ) != samplers_symbols . end ()) {
output += samplers_symbols [ name ];
}
}
if ( samplers_symbols . find ( input ) != samplers_symbols . end ()) {
output += samplers_symbols [ input ];
}
return output ;
}
2023-08-21 23:07:43 +03:00
//
// Model utils
//
2023-03-28 17:09:55 +03:00
2023-09-28 21:42:38 +02:00
struct llama_model_params llama_model_params_from_gpt_params ( const gpt_params & params ) {
auto mparams = llama_model_default_params ();
2023-05-02 22:39:51 +02:00
2023-09-04 22:26:24 +03:00
if ( params . n_gpu_layers != - 1 ) {
2023-09-28 21:42:38 +02:00
mparams . n_gpu_layers = params . n_gpu_layers ;
2023-09-04 22:26:24 +03:00
}
2023-09-28 21:42:38 +02:00
mparams . main_gpu = params . main_gpu ;
mparams . tensor_split = params . tensor_split ;
mparams . use_mmap = params . use_mmap ;
mparams . use_mlock = params . use_mlock ;
2023-12-05 10:19:18 -07:00
if ( params . kv_overrides . empty ()) {
mparams . kv_overrides = NULL ;
} else {
GGML_ASSERT ( params . kv_overrides . back (). key [ 0 ] == 0 && "KV overrides not terminated with empty key" );
mparams . kv_overrides = params . kv_overrides . data ();
}
2023-05-02 22:39:51 +02:00
2023-09-28 21:42:38 +02:00
return mparams ;
}
2023-12-07 13:03:17 +02:00
static ggml_type kv_cache_type_from_str ( const std :: string & s ) {
if ( s == "f16" ) {
return GGML_TYPE_F16 ;
}
if ( s == "q8_0" ) {
return GGML_TYPE_Q8_0 ;
}
if ( s == "q4_0" ) {
return GGML_TYPE_Q4_0 ;
}
if ( s == "q4_1" ) {
return GGML_TYPE_Q4_1 ;
}
if ( s == "q5_0" ) {
return GGML_TYPE_Q5_0 ;
}
if ( s == "q5_1" ) {
return GGML_TYPE_Q5_1 ;
}
throw std :: runtime_error ( "Invalid cache type: " + s );
}
2023-09-28 21:42:38 +02:00
struct llama_context_params llama_context_params_from_gpt_params ( const gpt_params & params ) {
auto cparams = llama_context_default_params ();
2023-11-01 18:04:33 -04:00
cparams . n_ctx = params . n_ctx ;
cparams . n_batch = params . n_batch ;
cparams . n_threads = params . n_threads ;
cparams . n_threads_batch = params . n_threads_batch == - 1 ? params . n_threads : params . n_threads_batch ;
cparams . mul_mat_q = params . mul_mat_q ;
cparams . seed = params . seed ;
cparams . logits_all = params . logits_all ;
cparams . embedding = params . embedding ;
cparams . rope_scaling_type = params . rope_scaling_type ;
cparams . rope_freq_base = params . rope_freq_base ;
cparams . rope_freq_scale = params . rope_freq_scale ;
cparams . yarn_ext_factor = params . yarn_ext_factor ;
cparams . yarn_attn_factor = params . yarn_attn_factor ;
cparams . yarn_beta_fast = params . yarn_beta_fast ;
cparams . yarn_beta_slow = params . yarn_beta_slow ;
cparams . yarn_orig_ctx = params . yarn_orig_ctx ;
2023-12-07 13:03:17 +02:00
cparams . offload_kqv = ! params . no_kv_offload ;
cparams . type_k = kv_cache_type_from_str ( params . cache_type_k );
cparams . type_v = kv_cache_type_from_str ( params . cache_type_v );
2023-09-28 21:42:38 +02:00
return cparams ;
2023-07-12 00:18:43 +08:00
}
2023-10-18 16:21:57 +03:00
void llama_batch_clear ( struct llama_batch & batch ) {
batch . n_tokens = 0 ;
}
void llama_batch_add (
struct llama_batch & batch ,
llama_token id ,
llama_pos pos ,
const std :: vector < llama_seq_id > & seq_ids ,
bool logits ) {
batch . token [ batch . n_tokens ] = id ;
2023-11-19 08:52:57 -08:00
batch . pos [ batch . n_tokens ] = pos ;
2023-10-18 16:21:57 +03:00
batch . n_seq_id [ batch . n_tokens ] = seq_ids . size ();
for ( size_t i = 0 ; i < seq_ids . size (); ++ i ) {
batch . seq_id [ batch . n_tokens ][ i ] = seq_ids [ i ];
}
batch . logits [ batch . n_tokens ] = logits ;
batch . n_tokens ++ ;
}
2023-08-21 23:07:43 +03:00
std :: tuple < struct llama_model * , struct llama_context *> llama_init_from_gpt_params ( gpt_params & params ) {
2023-09-28 21:42:38 +02:00
auto mparams = llama_model_params_from_gpt_params ( params );
2023-07-12 00:18:43 +08:00
2023-09-28 21:42:38 +02:00
llama_model * model = llama_load_model_from_file ( params . model . c_str (), mparams );
2023-06-24 11:47:58 +03:00
if ( model == NULL ) {
2023-05-02 22:39:51 +02:00
fprintf ( stderr , "%s: error: failed to load model '%s' \n " , __func__ , params . model . c_str ());
2023-06-24 11:47:58 +03:00
return std :: make_tuple ( nullptr , nullptr );
}
2023-09-28 21:42:38 +02:00
auto cparams = llama_context_params_from_gpt_params ( params );
llama_context * lctx = llama_new_context_with_model ( model , cparams );
2023-06-24 11:47:58 +03:00
if ( lctx == NULL ) {
fprintf ( stderr , "%s: error: failed to create context with model '%s' \n " , __func__ , params . model . c_str ());
llama_free_model ( model );
return std :: make_tuple ( nullptr , nullptr );
2023-05-02 22:39:51 +02:00
}
2023-09-28 20:40:11 +02:00
for ( unsigned int i = 0 ; i < params . lora_adapter . size (); ++ i ) {
const std :: string & lora_adapter = std :: get < 0 > ( params . lora_adapter [ i ]);
float lora_scale = std :: get < 1 > ( params . lora_adapter [ i ]);
2023-06-24 11:47:58 +03:00
int err = llama_model_apply_lora_from_file ( model ,
2023-09-28 20:40:11 +02:00
lora_adapter . c_str (),
lora_scale ,
(( i > 0 ) || params . lora_base . empty ())
? NULL
: params . lora_base . c_str (),
2023-05-02 22:39:51 +02:00
params . n_threads );
if ( err != 0 ) {
fprintf ( stderr , "%s: error: failed to apply lora adapter \n " , __func__ );
2023-06-24 11:47:58 +03:00
llama_free ( lctx );
llama_free_model ( model );
return std :: make_tuple ( nullptr , nullptr );
2023-05-02 22:39:51 +02:00
}
}
2023-08-21 23:07:43 +03:00
if ( params . ignore_eos ) {
2023-10-23 12:40:03 -07:00
params . sparams . logit_bias [ llama_token_eos ( model )] = - INFINITY ;
2023-08-21 23:07:43 +03:00
}
2023-09-03 13:42:56 +03:00
{
LOG ( "warming up the model with an empty run \n " );
2023-10-23 12:40:03 -07:00
std :: vector < llama_token > tmp = { llama_token_bos ( model ), llama_token_eos ( model ), };
2023-09-28 21:42:38 +02:00
llama_decode ( lctx , llama_batch_get_one ( tmp . data (), std :: min ( tmp . size (), ( size_t ) params . n_batch ), 0 , 0 ));
2023-10-29 11:31:40 -06:00
llama_kv_cache_clear ( lctx );
2023-09-03 13:42:56 +03:00
llama_reset_timings ( lctx );
}
2023-06-24 11:47:58 +03:00
return std :: make_tuple ( model , lctx );
2023-05-02 22:39:51 +02:00
}
2023-08-21 23:07:43 +03:00
//
// Vocab utils
//
std :: vector < llama_token > llama_tokenize (
2023-09-28 21:42:38 +02:00
const struct llama_context * ctx ,
const std :: string & text ,
2023-10-17 17:11:01 +02:00
bool add_bos ,
bool special ) {
return llama_tokenize ( llama_get_model ( ctx ), text , add_bos , special );
2023-09-28 21:42:38 +02:00
}
std :: vector < llama_token > llama_tokenize (
const struct llama_model * model ,
2023-08-21 23:07:43 +03:00
const std :: string & text ,
2023-10-17 17:11:01 +02:00
bool add_bos ,
bool special ) {
2023-08-21 23:07:43 +03:00
// upper limit for the number of tokens
int n_tokens = text . length () + add_bos ;
std :: vector < llama_token > result ( n_tokens );
2023-10-17 17:11:01 +02:00
n_tokens = llama_tokenize ( model , text . data (), text . length (), result . data (), result . size (), add_bos , special );
2023-08-21 23:07:43 +03:00
if ( n_tokens < 0 ) {
result . resize ( - n_tokens );
2023-10-17 17:11:01 +02:00
int check = llama_tokenize ( model , text . data (), text . length (), result . data (), result . size (), add_bos , special );
2023-08-21 23:07:43 +03:00
GGML_ASSERT ( check == - n_tokens );
} else {
result . resize ( n_tokens );
}
return result ;
}
2023-08-27 14:19:19 +03:00
std :: string llama_token_to_piece ( const struct llama_context * ctx , llama_token token ) {
2023-08-21 23:07:43 +03:00
std :: vector < char > result ( 8 , 0 );
2023-09-28 21:42:38 +02:00
const int n_tokens = llama_token_to_piece ( llama_get_model ( ctx ), token , result . data (), result . size ());
2023-08-21 23:07:43 +03:00
if ( n_tokens < 0 ) {
result . resize ( - n_tokens );
2023-09-28 21:42:38 +02:00
int check = llama_token_to_piece ( llama_get_model ( ctx ), token , result . data (), result . size ());
2023-08-21 23:07:43 +03:00
GGML_ASSERT ( check == - n_tokens );
} else {
result . resize ( n_tokens );
}
return std :: string ( result . data (), result . size ());
}
2023-08-27 14:19:19 +03:00
std :: string llama_detokenize_spm ( llama_context * ctx , const std :: vector < llama_token > & tokens ) {
2023-10-23 12:40:03 -07:00
const llama_token bos_id = llama_token_bos ( llama_get_model ( ctx ));
2023-08-27 14:19:19 +03:00
std :: string piece ;
std :: string result ;
for ( size_t i = 0 ; i < tokens . size (); ++ i ) {
piece = llama_token_to_piece ( ctx , tokens [ i ]);
// remove the leading space of the first non-BOS token
if ((( tokens [ 0 ] == bos_id && i == 1 ) || ( tokens [ 0 ] != bos_id && i == 0 )) && piece [ 0 ] == ' ' ) {
piece = piece . substr ( 1 );
}
result += piece ;
}
return result ;
}
std :: string llama_detokenize_bpe ( llama_context * ctx , const std :: vector < llama_token > & tokens ) {
std :: string piece ;
std :: string result ;
for ( size_t i = 0 ; i < tokens . size (); ++ i ) {
piece = llama_token_to_piece ( ctx , tokens [ i ]);
result += piece ;
}
2023-10-03 09:16:26 +02:00
// NOTE: the original tokenizer decodes bytes after collecting the pieces.
2023-08-27 14:19:19 +03:00
return result ;
}
2023-08-28 17:59:39 +02:00
2023-11-16 19:14:37 -07:00
bool llama_should_add_bos_token ( const llama_model * model ) {
const int add_bos = llama_add_bos_token ( model );
return add_bos != - 1 ? bool ( add_bos ) : ( llama_vocab_type ( model ) == LLAMA_VOCAB_TYPE_SPM );
}
2023-09-03 15:12:08 +03:00
//
// YAML utils
//
2023-08-28 17:59:39 +02:00
// returns true if successful, false otherwise
bool create_directory_with_parents ( const std :: string & path ) {
#ifdef _WIN32
std :: wstring_convert < std :: codecvt_utf8 < wchar_t >> converter ;
std :: wstring wpath = converter . from_bytes ( path );
// if the path already exists, check whether it's a directory
const DWORD attributes = GetFileAttributesW ( wpath . c_str ());
if (( attributes != INVALID_FILE_ATTRIBUTES ) && ( attributes & FILE_ATTRIBUTE_DIRECTORY )) {
return true ;
}
size_t pos_slash = 0 ;
// process path from front to back, procedurally creating directories
while (( pos_slash = path . find ( '\\' , pos_slash )) != std :: string :: npos ) {
const std :: wstring subpath = wpath . substr ( 0 , pos_slash );
const wchar_t * test = subpath . c_str ();
const bool success = CreateDirectoryW ( test , NULL );
if ( ! success ) {
const DWORD error = GetLastError ();
// if the path already exists, ensure that it's a directory
if ( error == ERROR_ALREADY_EXISTS ) {
const DWORD attributes = GetFileAttributesW ( subpath . c_str ());
if ( attributes == INVALID_FILE_ATTRIBUTES || ! ( attributes & FILE_ATTRIBUTE_DIRECTORY )) {
return false ;
}
} else {
return false ;
}
}
pos_slash += 1 ;
}
return true ;
#else
// if the path already exists, check whether it's a directory
struct stat info ;
if ( stat ( path . c_str (), & info ) == 0 ) {
return S_ISDIR ( info . st_mode );
}
size_t pos_slash = 1 ; // skip leading slashes for directory creation
// process path from front to back, procedurally creating directories
while (( pos_slash = path . find ( '/' , pos_slash )) != std :: string :: npos ) {
const std :: string subpath = path . substr ( 0 , pos_slash );
struct stat info ;
// if the path already exists, ensure that it's a directory
if ( stat ( subpath . c_str (), & info ) == 0 ) {
if ( ! S_ISDIR ( info . st_mode )) {
return false ;
}
} else {
// create parent directories
const int ret = mkdir ( subpath . c_str (), 0755 );
if ( ret != 0 ) {
return false ;
}
}
pos_slash += 1 ;
}
return true ;
#endif // _WIN32
}
void dump_vector_float_yaml ( FILE * stream , const char * prop_name , const std :: vector < float > & data ) {
if ( data . empty ()) {
fprintf ( stream , "%s: \n " , prop_name );
return ;
}
fprintf ( stream , "%s: [" , prop_name );
for ( size_t i = 0 ; i < data . size () - 1 ; ++ i ) {
fprintf ( stream , "%e, " , data [ i ]);
}
fprintf ( stream , "%e] \n " , data . back ());
}
void dump_vector_int_yaml ( FILE * stream , const char * prop_name , const std :: vector < int > & data ) {
if ( data . empty ()) {
fprintf ( stream , "%s: \n " , prop_name );
return ;
}
fprintf ( stream , "%s: [" , prop_name );
for ( size_t i = 0 ; i < data . size () - 1 ; ++ i ) {
fprintf ( stream , "%d, " , data [ i ]);
}
fprintf ( stream , "%d] \n " , data . back ());
}
void dump_string_yaml_multiline ( FILE * stream , const char * prop_name , const char * data ) {
std :: string data_str ( data == NULL ? "" : data );
if ( data_str . empty ()) {
fprintf ( stream , "%s: \n " , prop_name );
return ;
}
size_t pos_start = 0 ;
size_t pos_found = 0 ;
if ( ! data_str . empty () && ( std :: isspace ( data_str [ 0 ]) || std :: isspace ( data_str . back ()))) {
data_str = std :: regex_replace ( data_str , std :: regex ( " \n " ), " \\ n" );
data_str = std :: regex_replace ( data_str , std :: regex ( " \" " ), " \\\" " );
2023-11-17 16:24:07 +01:00
data_str = std :: regex_replace ( data_str , std :: regex ( R "( \\ [^n" ]) "), R" ( \$ & ) ");
2023-08-28 17:59:39 +02:00
data_str = " \" " + data_str + " \" " ;
fprintf ( stream , "%s: %s \n " , prop_name , data_str . c_str ());
return ;
}
if ( data_str . find ( '\n' ) == std :: string :: npos ) {
fprintf ( stream , "%s: %s \n " , prop_name , data_str . c_str ());
return ;
}
fprintf ( stream , "%s: | \n " , prop_name );
while (( pos_found = data_str . find ( '\n' , pos_start )) != std :: string :: npos ) {
fprintf ( stream , " %s \n " , data_str . substr ( pos_start , pos_found - pos_start ). c_str ());
pos_start = pos_found + 1 ;
}
}
std :: string get_sortable_timestamp () {
using clock = std :: chrono :: system_clock ;
const clock :: time_point current_time = clock :: now ();
const time_t as_time_t = clock :: to_time_t ( current_time );
char timestamp_no_ns [ 100 ];
std :: strftime ( timestamp_no_ns , 100 , "%Y_%m_%d-%H_%M_%S" , std :: localtime ( & as_time_t ));
const int64_t ns = std :: chrono :: duration_cast < std :: chrono :: nanoseconds > (
current_time . time_since_epoch () % 1000000000 ). count ();
2023-08-28 21:51:47 +02:00
char timestamp_ns [ 11 ];
snprintf ( timestamp_ns , 11 , "%09" PRId64 , ns );
2023-08-28 17:59:39 +02:00
return std :: string ( timestamp_no_ns ) + "." + std :: string ( timestamp_ns );
}
void dump_non_result_info_yaml ( FILE * stream , const gpt_params & params , const llama_context * lctx ,
const std :: string & timestamp , const std :: vector < int > & prompt_tokens , const char * model_desc ) {
2023-10-20 21:07:23 +03:00
const llama_sampling_params & sparams = params . sparams ;
2023-10-11 13:35:46 -06:00
2023-11-02 02:50:16 -04:00
fprintf ( stream , "build_commit: %s \n " , LLAMA_COMMIT );
fprintf ( stream , "build_number: %d \n " , LLAMA_BUILD_NUMBER );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "cpu_has_arm_fma: %s \n " , ggml_cpu_has_arm_fma () ? "true" : "false" );
fprintf ( stream , "cpu_has_avx: %s \n " , ggml_cpu_has_avx () ? "true" : "false" );
2023-12-30 15:07:48 +07:00
fprintf ( stream , "cpu_has_avx_vnni: %s \n " , ggml_cpu_has_avx_vnni () ? "true" : "false" );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "cpu_has_avx2: %s \n " , ggml_cpu_has_avx2 () ? "true" : "false" );
fprintf ( stream , "cpu_has_avx512: %s \n " , ggml_cpu_has_avx512 () ? "true" : "false" );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "cpu_has_avx512_vbmi: %s \n " , ggml_cpu_has_avx512_vbmi () ? "true" : "false" );
fprintf ( stream , "cpu_has_avx512_vnni: %s \n " , ggml_cpu_has_avx512_vnni () ? "true" : "false" );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "cpu_has_blas: %s \n " , ggml_cpu_has_blas () ? "true" : "false" );
fprintf ( stream , "cpu_has_cublas: %s \n " , ggml_cpu_has_cublas () ? "true" : "false" );
fprintf ( stream , "cpu_has_clblast: %s \n " , ggml_cpu_has_clblast () ? "true" : "false" );
fprintf ( stream , "cpu_has_fma: %s \n " , ggml_cpu_has_fma () ? "true" : "false" );
fprintf ( stream , "cpu_has_gpublas: %s \n " , ggml_cpu_has_gpublas () ? "true" : "false" );
fprintf ( stream , "cpu_has_neon: %s \n " , ggml_cpu_has_neon () ? "true" : "false" );
fprintf ( stream , "cpu_has_f16c: %s \n " , ggml_cpu_has_f16c () ? "true" : "false" );
fprintf ( stream , "cpu_has_fp16_va: %s \n " , ggml_cpu_has_fp16_va () ? "true" : "false" );
fprintf ( stream , "cpu_has_wasm_simd: %s \n " , ggml_cpu_has_wasm_simd () ? "true" : "false" );
fprintf ( stream , "cpu_has_blas: %s \n " , ggml_cpu_has_blas () ? "true" : "false" );
fprintf ( stream , "cpu_has_sse3: %s \n " , ggml_cpu_has_sse3 () ? "true" : "false" );
fprintf ( stream , "cpu_has_vsx: %s \n " , ggml_cpu_has_vsx () ? "true" : "false" );
2023-08-28 17:59:39 +02:00
#ifdef NDEBUG
fprintf ( stream , "debug: false \n " );
#else
fprintf ( stream , "debug: true \n " );
#endif // NDEBUG
fprintf ( stream , "model_desc: %s \n " , model_desc );
2023-09-28 21:42:38 +02:00
fprintf ( stream , "n_vocab: %d # output size of the final layer, 32001 for some models \n " , llama_n_vocab ( llama_get_model ( lctx )));
2023-08-28 17:59:39 +02:00
#ifdef __OPTIMIZE__
fprintf ( stream , "optimize: true \n " );
#else
fprintf ( stream , "optimize: false \n " );
#endif // __OPTIMIZE__
fprintf ( stream , "time: %s \n " , timestamp . c_str ());
fprintf ( stream , " \n " );
fprintf ( stream , "############### \n " );
fprintf ( stream , "# User Inputs # \n " );
fprintf ( stream , "############### \n " );
fprintf ( stream , " \n " );
fprintf ( stream , "alias: %s # default: unknown \n " , params . model_alias . c_str ());
fprintf ( stream , "batch_size: %d # default: 512 \n " , params . n_batch );
2023-10-11 13:35:46 -06:00
dump_string_yaml_multiline ( stream , "cfg_negative_prompt" , sparams . cfg_negative_prompt . c_str ());
fprintf ( stream , "cfg_scale: %f # default: 1.0 \n " , sparams . cfg_scale );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "chunks: %d # default: -1 (unlimited) \n " , params . n_chunks );
fprintf ( stream , "color: %s # default: false \n " , params . use_color ? "true" : "false" );
fprintf ( stream , "ctx_size: %d # default: 512 \n " , params . n_ctx );
fprintf ( stream , "escape: %s # default: false \n " , params . escape ? "true" : "false" );
fprintf ( stream , "file: # never logged, see prompt instead. Can still be specified for input. \n " );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "frequency_penalty: %f # default: 0.0 \n " , sparams . penalty_freq );
dump_string_yaml_multiline ( stream , "grammar" , sparams . grammar . c_str ());
2023-08-28 17:59:39 +02:00
fprintf ( stream , "grammar-file: # never logged, see grammar instead. Can still be specified for input. \n " );
fprintf ( stream , "hellaswag: %s # default: false \n " , params . hellaswag ? "true" : "false" );
2023-09-01 09:34:50 -04:00
fprintf ( stream , "hellaswag_tasks: %zu # default: 400 \n " , params . hellaswag_tasks );
2023-08-28 17:59:39 +02:00
2023-10-23 12:40:03 -07:00
const auto logit_bias_eos = sparams . logit_bias . find ( llama_token_eos ( llama_get_model ( lctx )));
2023-10-11 13:35:46 -06:00
const bool ignore_eos = logit_bias_eos != sparams . logit_bias . end () && logit_bias_eos -> second == - INFINITY ;
2023-08-28 17:59:39 +02:00
fprintf ( stream , "ignore_eos: %s # default: false \n " , ignore_eos ? "true" : "false" );
dump_string_yaml_multiline ( stream , "in_prefix" , params . input_prefix . c_str ());
fprintf ( stream , "in_prefix_bos: %s # default: false \n " , params . input_prefix_bos ? "true" : "false" );
dump_string_yaml_multiline ( stream , "in_suffix" , params . input_prefix . c_str ());
fprintf ( stream , "instruct: %s # default: false \n " , params . instruct ? "true" : "false" );
fprintf ( stream , "interactive: %s # default: false \n " , params . interactive ? "true" : "false" );
fprintf ( stream , "interactive_first: %s # default: false \n " , params . interactive_first ? "true" : "false" );
fprintf ( stream , "keep: %d # default: 0 \n " , params . n_keep );
fprintf ( stream , "logdir: %s # default: unset (no logging) \n " , params . logdir . c_str ());
fprintf ( stream , "logit_bias: \n " );
2023-10-11 13:35:46 -06:00
for ( std :: pair < llama_token , float > lb : sparams . logit_bias ) {
2023-08-28 17:59:39 +02:00
if ( ignore_eos && lb . first == logit_bias_eos -> first ) {
continue ;
}
fprintf ( stream , " %d: %f" , lb . first , lb . second );
}
2023-09-28 20:40:11 +02:00
fprintf ( stream , "lora: \n " );
for ( std :: tuple < std :: string , float > la : params . lora_adapter ) {
if ( std :: get < 1 > ( la ) != 1.0f ) {
continue ;
}
fprintf ( stream , " - %s \n " , std :: get < 0 > ( la ). c_str ());
}
fprintf ( stream , "lora_scaled: \n " );
for ( std :: tuple < std :: string , float > la : params . lora_adapter ) {
if ( std :: get < 1 > ( la ) == 1.0f ) {
continue ;
}
fprintf ( stream , " - %s: %f \n " , std :: get < 0 > ( la ). c_str (), std :: get < 1 > ( la ));
}
2023-08-28 17:59:39 +02:00
fprintf ( stream , "lora_base: %s \n " , params . lora_base . c_str ());
fprintf ( stream , "main_gpu: %d # default: 0 \n " , params . main_gpu );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "mirostat: %d # default: 0 (disabled) \n " , sparams . mirostat );
fprintf ( stream , "mirostat_ent: %f # default: 5.0 \n " , sparams . mirostat_tau );
fprintf ( stream , "mirostat_lr: %f # default: 0.1 \n " , sparams . mirostat_eta );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "mlock: %s # default: false \n " , params . use_mlock ? "true" : "false" );
fprintf ( stream , "model: %s # default: models/7B/ggml-model.bin \n " , params . model . c_str ());
2023-09-03 15:12:08 +03:00
fprintf ( stream , "model_draft: %s # default: \n " , params . model_draft . c_str ());
2023-08-28 17:59:39 +02:00
fprintf ( stream , "multiline_input: %s # default: false \n " , params . multiline_input ? "true" : "false" );
2023-09-04 22:26:24 +03:00
fprintf ( stream , "n_gpu_layers: %d # default: -1 \n " , params . n_gpu_layers );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "n_predict: %d # default: -1 (unlimited) \n " , params . n_predict );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "n_probs: %d # only used by server binary, default: 0 \n " , sparams . n_probs );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "no_mmap: %s # default: false \n " , ! params . use_mmap ? "true" : "false" );
fprintf ( stream , "no_mul_mat_q: %s # default: false \n " , ! params . mul_mat_q ? "true" : "false" );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "no_penalize_nl: %s # default: false \n " , ! sparams . penalize_nl ? "true" : "false" );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "numa: %s # default: false \n " , params . numa ? "true" : "false" );
fprintf ( stream , "ppl_output_type: %d # default: 0 \n " , params . ppl_output_type );
fprintf ( stream , "ppl_stride: %d # default: 0 \n " , params . ppl_stride );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "presence_penalty: %f # default: 0.0 \n " , sparams . penalty_present );
2023-08-28 17:59:39 +02:00
dump_string_yaml_multiline ( stream , "prompt" , params . prompt . c_str ());
fprintf ( stream , "prompt_cache: %s \n " , params . path_prompt_cache . c_str ());
fprintf ( stream , "prompt_cache_all: %s # default: false \n " , params . prompt_cache_all ? "true" : "false" );
fprintf ( stream , "prompt_cache_ro: %s # default: false \n " , params . prompt_cache_ro ? "true" : "false" );
dump_vector_int_yaml ( stream , "prompt_tokens" , prompt_tokens );
fprintf ( stream , "random_prompt: %s # default: false \n " , params . random_prompt ? "true" : "false" );
2023-10-20 21:07:23 +03:00
fprintf ( stream , "repeat_penalty: %f # default: 1.1 \n " , sparams . penalty_repeat );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "reverse_prompt: \n " );
for ( std :: string ap : params . antiprompt ) {
size_t pos = 0 ;
while (( pos = ap . find ( '\n' , pos )) != std :: string :: npos ) {
ap . replace ( pos , 1 , " \\ n" );
pos += 1 ;
}
fprintf ( stream , " - %s \n " , ap . c_str ());
}
fprintf ( stream , "rope_freq_base: %f # default: 10000.0 \n " , params . rope_freq_base );
fprintf ( stream , "rope_freq_scale: %f # default: 1.0 \n " , params . rope_freq_scale );
fprintf ( stream , "seed: %d # default: -1 (random seed) \n " , params . seed );
fprintf ( stream , "simple_io: %s # default: false \n " , params . simple_io ? "true" : "false" );
2023-09-28 19:04:36 +03:00
fprintf ( stream , "cont_batching: %s # default: false \n " , params . cont_batching ? "true" : "false" );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "temp: %f # default: 0.8 \n " , sparams . temp );
2023-08-28 17:59:39 +02:00
const std :: vector < float > tensor_split_vector ( params . tensor_split , params . tensor_split + LLAMA_MAX_DEVICES );
dump_vector_float_yaml ( stream , "tensor_split" , tensor_split_vector );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "tfs: %f # default: 1.0 \n " , sparams . tfs_z );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "threads: %d # default: %d \n " , params . n_threads , std :: thread :: hardware_concurrency ());
2023-10-11 13:35:46 -06:00
fprintf ( stream , "top_k: %d # default: 40 \n " , sparams . top_k );
fprintf ( stream , "top_p: %f # default: 0.95 \n " , sparams . top_p );
2023-10-31 14:44:49 -05:00
fprintf ( stream , "min_p: %f # default: 0.0 \n " , sparams . min_p );
2023-10-11 13:35:46 -06:00
fprintf ( stream , "typical_p: %f # default: 1.0 \n " , sparams . typical_p );
2023-08-28 17:59:39 +02:00
fprintf ( stream , "verbose_prompt: %s # default: false \n " , params . verbose_prompt ? "true" : "false" );
}
2023-11-23 19:07:56 +02:00
//
// KV cache utils
//
void dump_kv_cache_view ( const llama_kv_cache_view & view , int row_size ) {
static const char slot_chars [] = ".123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz+" ;
printf ( "=== Dumping KV cache. total cells %d, max sequences per cell %d, populated cells %d, total tokens in cache %d, largest empty slot=%d @ %d" ,
view . n_cells , view . n_max_seq , view . used_cells , view . token_count , view . max_contiguous , view . max_contiguous_idx );
llama_kv_cache_view_cell * c_curr = view . cells ;
llama_seq_id * cs_curr = view . cells_sequences ;
for ( int i = 0 ; i < view . n_cells ; i ++ , c_curr ++ , cs_curr += view . n_max_seq ) {
if ( i % row_size == 0 ) {
printf ( " \n %5d: " , i );
}
int seq_count = 0 ;
for ( int j = 0 ; j < view . n_max_seq ; j ++ ) {
if ( cs_curr [ j ] >= 0 ) { seq_count ++ ; }
}
putchar ( slot_chars [ std :: min ( sizeof ( slot_chars ) - 2 , size_t ( seq_count ))]);
}
printf ( " \n === Done dumping \n " );
}
void dump_kv_cache_view_seqs ( const llama_kv_cache_view & view , int row_size ) {
static const char slot_chars [] = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz" ;
printf ( "=== Dumping KV cache. total cells %d, max sequences per cell %d, populated cells %d, total tokens in cache %d, largest empty slot=%d @ %d \n " ,
view . n_cells , view . n_max_seq , view . used_cells , view . token_count , view . max_contiguous , view . max_contiguous_idx );
std :: unordered_map < llama_seq_id , size_t > seqs ;
llama_kv_cache_view_cell * c_curr = view . cells ;
llama_seq_id * cs_curr = view . cells_sequences ;
for ( int i = 0 ; i < view . n_cells ; i ++ , c_curr ++ , cs_curr += view . n_max_seq ) {
for ( int j = 0 ; j < view . n_max_seq ; j ++ ) {
if ( cs_curr [ j ] < 0 ) { continue ; }
if ( seqs . find ( cs_curr [ j ]) == seqs . end ()) {
if ( seqs . size () + 1 >= sizeof ( slot_chars )) { break ; }
seqs [ cs_curr [ j ]] = seqs . size ();
}
}
if ( seqs . size () + 1 >= sizeof ( slot_chars )) { break ; }
}
printf ( "=== Sequence legend: " );
for ( const auto & it : seqs ) {
printf ( "%zu=%d, " , it . second , it . first );
}
printf ( "'+'=other sequence ids" );
c_curr = view . cells ;
cs_curr = view . cells_sequences ;
for ( int i = 0 ; i < view . n_cells ; i ++ , c_curr ++ , cs_curr += view . n_max_seq ) {
if ( i % row_size == 0 ) {
printf ( " \n %5d: " , i );
}
for ( int j = 0 ; j < view . n_max_seq ; j ++ ) {
if ( cs_curr [ j ] >= 0 ) {
const auto & it = seqs . find ( cs_curr [ j ]);
putchar ( it != seqs . end () ? int ( slot_chars [ it -> second ]) : '+' );
} else {
putchar ( '.' );
}
}
putchar ( ' ' );
}
printf ( " \n === Done dumping \n " );
}