Merge branch 'dev' into feat/auto-exit

This commit is contained in:
Li, Zonghang
2025-05-20 02:04:14 +08:00
committed by GitHub
10 changed files with 488 additions and 87 deletions
+25 -14
View File
@@ -986,13 +986,13 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
params.enable_chat_template = false;
}
).set_examples({LLAMA_EXAMPLE_MAIN, LLAMA_EXAMPLE_INFILL}));
add_opt(llama_arg(
{"--no-warmup"},
"skip warming up the model with an empty run",
[](gpt_params & params) {
params.warmup = false;
}
).set_examples({LLAMA_EXAMPLE_MAIN}));
// add_opt(llama_arg(
// {"--no-warmup"},
// "skip warming up the model with an empty run",
// [](gpt_params & params) {
// params.warmup = false;
// }
// ).set_examples({LLAMA_EXAMPLE_MAIN}));
add_opt(llama_arg(
{"--spm-infill"},
format(
@@ -1317,6 +1317,12 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
{"-ctk", "--cache-type-k"}, "TYPE",
format("KV cache data type for K (default: %s)", params.cache_type_k.c_str()),
[](gpt_params & params, const std::string & value) {
#ifdef GGML_USE_METAL
LOG_WRN("The option -ctk or --cache-type-k is not supported on Metal, use default type\n");
return;
#endif
// TODO: get the type right here
params.cache_type_k = value;
}
@@ -1325,6 +1331,11 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
{"-ctv", "--cache-type-v"}, "TYPE",
format("KV cache data type for V (default: %s)", params.cache_type_v.c_str()),
[](gpt_params & params, const std::string & value) {
#ifdef GGML_USE_METAL
LOG_WRN("The option -ctv or --cache-type-v is not supported on Metal, use default type\n");
return;
#endif
// TODO: get the type right here
params.cache_type_v = value;
}
@@ -1413,13 +1424,13 @@ gpt_params_context gpt_params_parser_init(gpt_params & params, llama_example ex,
params.defrag_thold = std::stof(value);
}
).set_env("LLAMA_ARG_DEFRAG_THOLD"));
add_opt(llama_arg(
{"-np", "--parallel"}, "N",
format("number of parallel sequences to decode (default: %d)", params.n_parallel),
[](gpt_params & params, int value) {
params.n_parallel = value;
}
).set_env("LLAMA_ARG_N_PARALLEL"));
// add_opt(llama_arg(
// {"-np", "--parallel"}, "N",
// format("number of parallel sequences to decode (default: %d)", params.n_parallel),
// [](gpt_params & params, int value) {
// params.n_parallel = value;
// }
// ).set_env("LLAMA_ARG_N_PARALLEL"));
add_opt(llama_arg(
{"-ns", "--sequences"}, "N",
format("number of sequences to decode (default: %d)", params.n_sequences),
+28 -3
View File
@@ -1582,6 +1582,12 @@ static bool tune_layer_allocation(
//
struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
#if !(defined(GGML_USE_METAL) || defined(GGML_USE_CUDA))
// reset n_gpu_layers to 0 if GPU is not used
params.n_gpu_layers = 0;
#endif
llama_init_result iparams;
auto mparams = llama_model_params_from_gpt_params(params);
@@ -1637,10 +1643,16 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
if (n_world == 1) {
uint32_t n_layers = llama_model_n_layers(model);
// assign all layers to this device
params.n_layer_window[0] = n_layers;
cparams.n_layer_window[0] = n_layers;
mparams.n_layer_window[0] = n_layers;
llama_context_n_layer_window(lctx)[0] = n_layers;
#if defined(GGML_USE_METAL) || defined(GGML_USE_CUDA)
params.n_gpu_layers = std::min((int32_t)n_layers, params.n_gpu_layers);
#endif
} else {
uint32_t n_layer_window[32] = {0}, n_gpu_layers[32] = {0};
@@ -1649,12 +1661,20 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
// broadcast startup args
struct startup_args args;
if (my_rank==0){
if (my_rank == 0){
args.should_profile = auto_schedule;
args.n_ctx = params.n_ctx;
}
llama_bcast_startup_args(lctx, my_rank, &args);
auto_schedule = args.should_profile;
if (my_rank > 0) {
// receive startup args
auto_schedule = args.should_profile;
params.n_ctx = args.n_ctx;
cparams.n_ctx = args.n_ctx;
}
// if n_world > 1 and need auto schdule, then prifile
if (auto_schedule){
// get device profile
@@ -1751,6 +1771,11 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
cparams.n_gpu_layers = n_gpu_layers[my_rank];
mparams.n_gpu_layers = n_gpu_layers[my_rank];
llama_model_set_n_gpu_layers(model, n_gpu_layers[my_rank]);
} else { // -ngl is set
params.n_gpu_layers = std::min(params.n_gpu_layers, (int32_t)n_layer_window[my_rank]);
cparams.n_gpu_layers = params.n_gpu_layers;
mparams.n_gpu_layers = params.n_gpu_layers;
llama_model_set_n_gpu_layers(model, params.n_gpu_layers);
}
}
@@ -1820,7 +1845,7 @@ struct llama_init_result llama_init_from_gpt_params(gpt_params & params) {
}
if (params.warmup) {
LOG_WRN("%s: warming up the model with an empty run - please wait ... (--no-warmup to disable)\n", __func__);
LOG_WRN("%s: warming up the model with an empty run - please wait ...\n", __func__);
const uint32_t my_rank = cparams.rank;
std::vector<llama_token> tmp;
+1 -2
View File
@@ -350,7 +350,6 @@ float device_inp_embd_delay(struct llama_model * model, enum ggml_type src0t, in
return 0.0f;
}
size_t QK_K = 0;
switch (src0t) {
case GGML_TYPE_F32: {
matrix_B = malloc(embd_size * sizeof(float));
@@ -914,7 +913,7 @@ ioengine=%s
direct=1
time_based=1
runtime=1
size=4G
size=1G
group_reporting=1
iodepth=1
+2 -1
View File
@@ -313,7 +313,8 @@ struct disk_props {
};
struct startup_args{
bool should_profile;
bool should_profile;
uint32_t n_ctx;
};
struct device_info {