Merge branch 'dev' into feat/auto-exit
This commit is contained in:
+25
-14
@@ -986,13 +986,13 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
|
||||
params.enable_chat_template = false;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_MAIN, LLAMA_EXAMPLE_INFILL}));
|
||||
add_opt(llama_arg(
|
||||
{"--no-warmup"},
|
||||
"skip warming up the model with an empty run",
|
||||
[](gpt_params & params) {
|
||||
params.warmup = false;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_MAIN}));
|
||||
// add_opt(llama_arg(
|
||||
// {"--no-warmup"},
|
||||
// "skip warming up the model with an empty run",
|
||||
// [](gpt_params & params) {
|
||||
// params.warmup = false;
|
||||
// }
|
||||
// ).set_examples({LLAMA_EXAMPLE_MAIN}));
|
||||
add_opt(llama_arg(
|
||||
{"--spm-infill"},
|
||||
format(
|
||||
@@ -1317,6 +1317,12 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
|
||||
{"-ctk", "--cache-type-k"}, "TYPE",
|
||||
format("KV cache data type for K (default: %s)", params.cache_type_k.c_str()),
|
||||
[](gpt_params & params, const std::string & value) {
|
||||
|
||||
#ifdef GGML_USE_METAL
|
||||
LOG_WRN("The option -ctk or --cache-type-k is not supported on Metal, use default type\n");
|
||||
return;
|
||||
#endif
|
||||
|
||||
// TODO: get the type right here
|
||||
params.cache_type_k = value;
|
||||
}
|
||||
@@ -1325,6 +1331,11 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
|
||||
{"-ctv", "--cache-type-v"}, "TYPE",
|
||||
format("KV cache data type for V (default: %s)", params.cache_type_v.c_str()),
|
||||
[](gpt_params & params, const std::string & value) {
|
||||
#ifdef GGML_USE_METAL
|
||||
LOG_WRN("The option -ctv or --cache-type-v is not supported on Metal, use default type\n");
|
||||
return;
|
||||
#endif
|
||||
|
||||
// TODO: get the type right here
|
||||
params.cache_type_v = value;
|
||||
}
|
||||
@@ -1413,13 +1424,13 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
|
||||
params.defrag_thold = std::stof(value);
|
||||
}
|
||||
).set_env("LLAMA_ARG_DEFRAG_THOLD"));
|
||||
add_opt(llama_arg(
|
||||
{"-np", "--parallel"}, "N",
|
||||
format("number of parallel sequences to decode (default: %d)", params.n_parallel),
|
||||
[](gpt_params & params, int value) {
|
||||
params.n_parallel = value;
|
||||
}
|
||||
).set_env("LLAMA_ARG_N_PARALLEL"));
|
||||
// add_opt(llama_arg(
|
||||
// {"-np", "--parallel"}, "N",
|
||||
// format("number of parallel sequences to decode (default: %d)", params.n_parallel),
|
||||
// [](gpt_params & params, int value) {
|
||||
// params.n_parallel = value;
|
||||
// }
|
||||
// ).set_env("LLAMA_ARG_N_PARALLEL"));
|
||||
add_opt(llama_arg(
|
||||
{"-ns", "--sequences"}, "N",
|
||||
format("number of sequences to decode (default: %d)", params.n_sequences),
|
||||
|
||||
+28
-3
@@ -1582,6 +1582,12 @@ static bool tune_layer_allocation(
|
||||
//
|
||||
|
||||
struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
|
||||
|
||||
#if !(defined(GGML_USE_METAL) || defined(GGML_USE_CUDA))
|
||||
// reset n_gpu_layers to 0 if GPU is not used
|
||||
params.n_gpu_layers = 0;
|
||||
#endif
|
||||
|
||||
llama_init_result iparams;
|
||||
auto mparams = llama_model_params_from_gpt_params(params);
|
||||
|
||||
@@ -1637,10 +1643,16 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
|
||||
|
||||
if (n_world == 1) {
|
||||
uint32_t n_layers = llama_model_n_layers(model);
|
||||
// assign all layers to this device
|
||||
params.n_layer_window[0] = n_layers;
|
||||
cparams.n_layer_window[0] = n_layers;
|
||||
mparams.n_layer_window[0] = n_layers;
|
||||
llama_context_n_layer_window(lctx)[0] = n_layers;
|
||||
|
||||
#if defined(GGML_USE_METAL) || defined(GGML_USE_CUDA)
|
||||
params.n_gpu_layers = std::min((int32_t)n_layers, params.n_gpu_layers);
|
||||
#endif
|
||||
|
||||
} else {
|
||||
uint32_t n_layer_window[32] = {0}, n_gpu_layers[32] = {0};
|
||||
|
||||
@@ -1649,12 +1661,20 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
|
||||
|
||||
// broadcast startup args
|
||||
struct startup_args args;
|
||||
if (my_rank==0){
|
||||
if (my_rank == 0){
|
||||
args.should_profile = auto_schedule;
|
||||
args.n_ctx = params.n_ctx;
|
||||
}
|
||||
|
||||
llama_bcast_startup_args(lctx, my_rank, &args);
|
||||
|
||||
auto_schedule = args.should_profile;
|
||||
if (my_rank > 0) {
|
||||
// receive startup args
|
||||
auto_schedule = args.should_profile;
|
||||
params.n_ctx = args.n_ctx;
|
||||
cparams.n_ctx = args.n_ctx;
|
||||
}
|
||||
|
||||
// if n_world > 1 and need auto schdule, then prifile
|
||||
if (auto_schedule){
|
||||
// get device profile
|
||||
@@ -1751,6 +1771,11 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
|
||||
cparams.n_gpu_layers = n_gpu_layers[my_rank];
|
||||
mparams.n_gpu_layers = n_gpu_layers[my_rank];
|
||||
llama_model_set_n_gpu_layers(model, n_gpu_layers[my_rank]);
|
||||
} else { // -ngl is set
|
||||
params.n_gpu_layers = std::min(params.n_gpu_layers, (int32_t)n_layer_window[my_rank]);
|
||||
cparams.n_gpu_layers = params.n_gpu_layers;
|
||||
mparams.n_gpu_layers = params.n_gpu_layers;
|
||||
llama_model_set_n_gpu_layers(model, params.n_gpu_layers);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1820,7 +1845,7 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
|
||||
}
|
||||
|
||||
if (params.warmup) {
|
||||
LOG_WRN("%s: warming up the model with an empty run - please wait ... (--no-warmup to disable)\n", __func__);
|
||||
LOG_WRN("%s: warming up the model with an empty run - please wait ...\n", __func__);
|
||||
|
||||
const uint32_t my_rank = cparams.rank;
|
||||
std::vector<llama_token> tmp;
|
||||
|
||||
+1
-2
@@ -350,7 +350,6 @@ float device_inp_embd_delay(struct llama_model * model, enum ggml_type src0t, in
|
||||
return 0.0f;
|
||||
}
|
||||
|
||||
size_t QK_K = 0;
|
||||
switch (src0t) {
|
||||
case GGML_TYPE_F32: {
|
||||
matrix_B = malloc(embd_size * sizeof(float));
|
||||
@@ -914,7 +913,7 @@ ioengine=%s
|
||||
direct=1
|
||||
time_based=1
|
||||
runtime=1
|
||||
size=4G
|
||||
size=1G
|
||||
group_reporting=1
|
||||
iodepth=1
|
||||
|
||||
|
||||
+2
-1
@@ -313,7 +313,8 @@ struct disk_props {
|
||||
};
|
||||
|
||||
struct startup_args{
|
||||
bool should_profile;
|
||||
bool should_profile;
|
||||
uint32_t n_ctx;
|
||||
};
|
||||
|
||||
struct device_info {
|
||||
|
||||
Reference in New Issue
Block a user