Files
calculet-npu-research-archive/history/llama.cpp/llama.cpp-unmerged-runtime_replace.patch
T

2098 lines
82 KiB
Diff
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
diff --git a/CMakeLists.txt b/CMakeLists.txt
index 5969865f..4bf8b278 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -183,10 +183,6 @@ if (MINGW)
add_compile_definitions(_WIN32_WINNT=${GGML_WIN_VER})
endif()
-if (LLAMA_USE_CALRT)
- add_compile_definitions(USE_CALRT)
-endif()
-
#
# build the library
#
diff --git a/common/arg.cpp b/common/arg.cpp
index e3856047..8a9ba2f8 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1752,6 +1752,20 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
params.device_info = true;
}
));
+ add_opt(common_arg(
+ {"--cal-llm-profile"},
+ "enable cal-llm backend profiling logs",
+ [](common_params & params) {
+ params.cal_llm_profile = true;
+ }
+ ).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CAL_LLM_PROFILE"));
+ add_opt(common_arg(
+ {"--cal-llm-profile-dump"}, "FILE",
+ "write cal-llm backend profiling JSONL to file",
+ [](common_params & params, const std::string & value) {
+ params.cal_llm_profile_dump = value;
+ }
+ ).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_CAL_LLM_PROFILE_DUMP"));
add_opt(common_arg(
{"--mmproj-case"}, "FNAME",
"set calbin-mtmd path",
diff --git a/common/common.cpp b/common/common.cpp
index 898d4452..81c203e8 100644
--- a/common/common.cpp
+++ b/common/common.cpp
@@ -936,11 +936,6 @@ struct common_init_result common_init_from_params(common_params & params) {
return iparams;
}
-#ifdef USE_CALRT
- calrt_context * cal_ctx = init_calrt_context(params.calbin_path.c_str());
- calrt_set_llama_context(cal_ctx, lctx);
-#endif
-
if (params.ctx_shift && !llama_memory_can_shift(llama_get_memory(lctx))) {
LOG_WRN("%s: KV cache shifting is not supported for this context, disabling KV cache shifting\n", __func__);
params.ctx_shift = false;
@@ -952,7 +947,6 @@ struct common_init_result common_init_from_params(common_params & params) {
const auto cvec = common_control_vector_load(params.control_vectors);
if (cvec.n_embd == -1) {
- calrt_context_free(cal_ctx);
llama_free(lctx);
llama_model_free(model);
@@ -967,7 +961,6 @@ struct common_init_result common_init_from_params(common_params & params) {
params.control_vector_layer_start,
params.control_vector_layer_end);
if (err) {
- calrt_context_free(cal_ctx);
llama_free(lctx);
llama_model_free(model);
@@ -995,7 +988,6 @@ struct common_init_result common_init_from_params(common_params & params) {
}
if (!ok) {
- calrt_context_free(cal_ctx);
llama_free(lctx);
llama_model_free(model);
@@ -1009,7 +1001,6 @@ struct common_init_result common_init_from_params(common_params & params) {
lora.reset(llama_adapter_lora_init(model, la.path.c_str()));
if (lora == nullptr) {
LOG_ERR("%s: failed to apply lora adapter '%s'\n", __func__, la.path.c_str());
- calrt_context_free(cal_ctx);
llama_free(lctx);
llama_model_free(model);
return iparams;
@@ -1079,11 +1070,7 @@ struct common_init_result common_init_from_params(common_params & params) {
// }
// if (llama_model_has_encoder(model)) {
- // #ifdef USE_CALRT
- // //calrt_encode(cal_ctx, llama_batch_get_one(tmp.data(), tmp.size()));
- // #else
// llama_encode(lctx, llama_batch_get_one(tmp.data(), tmp.size()));
- // #endif
// llama_token decoder_start_token_id = llama_model_decoder_start_token(model);
// if (decoder_start_token_id == LLAMA_TOKEN_NULL) {
// decoder_start_token_id = bos;
@@ -1092,11 +1079,7 @@ struct common_init_result common_init_from_params(common_params & params) {
// tmp.push_back(decoder_start_token_id);
// }
// if (llama_model_has_decoder(model)) {
- // #ifdef USE_CALRT
- // calrt_decode(cal_ctx, llama_batch_get_one(tmp.data(), std::min(tmp.size(), (size_t) params.n_batch)));
- // #else
// llama_decode(lctx, llama_batch_get_one(tmp.data(), std::min(tmp.size(), (size_t) params.n_batch)));
- // #endif
// }
// llama_memory_clear(llama_get_memory(lctx), true);
// llama_synchronize(lctx);
@@ -1106,7 +1089,6 @@ struct common_init_result common_init_from_params(common_params & params) {
iparams.model.reset(model);
iparams.context.reset(lctx);
- iparams.cal_ctx.reset(cal_ctx);
return iparams;
}
diff --git a/common/common.h b/common/common.h
index 0e7456d9..6a80f4dd 100644
--- a/common/common.h
+++ b/common/common.h
@@ -323,9 +323,11 @@ struct common_params {
struct common_params_model model;
bool device_info = false;
+ bool cal_llm_profile = false;
std::string calbin_path = ""; // NOLINT
std::string calbin_mtmd_path = ""; // model alias // NOLINT
+ std::string cal_llm_profile_dump = ""; // NOLINT
std::string model_alias = ""; // model alias // NOLINT
std::string hf_token = ""; // HF token // NOLINT
std::string prompt = ""; // NOLINT
@@ -621,7 +623,6 @@ std::string fs_get_cache_file(const std::string & filename);
struct common_init_result {
llama_model_ptr model;
llama_context_ptr context;
- calrt_context_ptr cal_ctx;
std::vector<llama_adapter_lora_ptr> lora;
};
diff --git a/include/llama-cpp.h b/include/llama-cpp.h
index 0611aab8..8f636817 100644
--- a/include/llama-cpp.h
+++ b/include/llama-cpp.h
@@ -24,12 +24,7 @@ struct llama_adapter_lora_deleter {
void operator()(llama_adapter_lora * adapter) { llama_adapter_lora_free(adapter); }
};
-struct calrt_context_deleter {
- void operator()(calrt_context * context) {calrt_context_free(context);}
-};
-
typedef std::unique_ptr<llama_model, llama_model_deleter> llama_model_ptr;
typedef std::unique_ptr<llama_context, llama_context_deleter> llama_context_ptr;
typedef std::unique_ptr<llama_sampler, llama_sampler_deleter> llama_sampler_ptr;
typedef std::unique_ptr<llama_adapter_lora, llama_adapter_lora_deleter> llama_adapter_lora_ptr;
-typedef std::unique_ptr<calrt_context, calrt_context_deleter> calrt_context_ptr;
\ No newline at end of file
diff --git a/include/llama.h b/include/llama.h
index c3df1468..fa0c8dbb 100644
--- a/include/llama.h
+++ b/include/llama.h
@@ -56,9 +56,6 @@ extern "C" {
//
// TODO: show sample usage
//
-#ifdef USE_CALRT
- struct calrt_context;
-#endif
struct llama_vocab;
struct llama_model;
struct llama_context;
@@ -1407,15 +1404,6 @@ extern "C" {
ggml_opt_epoch_callback callback_train,
ggml_opt_epoch_callback callback_eval);
- LLAMA_API struct calrt_context * init_calrt_context(const char * cal_case_path);
- LLAMA_API void calrt_set_llama_context(struct calrt_context * ctx, struct llama_context * llama_ctx);
- LLAMA_API const struct llama_context * calrt_get_llama_context(const struct calrt_context * ctx);
- LLAMA_API uint32_t calrt_n_ctx(const struct calrt_context * ctx);
- LLAMA_API uint32_t calrt_n_batch(const struct calrt_context * ctx);
- LLAMA_API uint32_t calrt_n_ctx_per_seq(const struct calrt_context * ctx);
- LLAMA_API void calrt_context_free(struct calrt_context * ctx);
- LLAMA_API int calrt_decode(struct calrt_context * cal_ctx, struct llama_batch batch);
- //LLAMA_API int calrt_encode(struct calrt_context * cal_ctx, struct llama_batch batch);
#ifdef __cplusplus
}
#endif
diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt
index 7d42cd40..355c311f 100644
--- a/src/CMakeLists.txt
+++ b/src/CMakeLists.txt
@@ -35,8 +35,6 @@ add_library(llama
unicode-data.cpp
unicode.cpp
unicode.h
- llama-calrt.cpp
- llama-kv-cache-calrt-adapter.cpp
models/apertus.cpp
models/arcee.cpp
models/arctic.cpp
@@ -139,47 +137,6 @@ target_compile_features (llama PRIVATE cxx_std_17) # don't bump
target_link_libraries(llama PUBLIC ggml rt)
-if (LLAMA_USE_CALRT)
- if (USE_CALRT_SRC)
- list(APPEND CALRT_LIBRARIES
- calrt_objs
- elf
- )
- else()
- if (NOT DEFINED CALRT_INSTALL_DIR)
- set(CALRT_INSTALL_DIR /home/common_share/SMG/libcalrt)
- endif()
-
- find_library(CALRT_LIB
- NAMES hostrt calrt
- PATHS
- "${CALRT_INSTALL_DIR}/lib"
- "${CALRT_INSTALL_DIR}/lib64"
- NO_DEFAULT_PATH
- )
-
- add_library(calrt_prebuilt SHARED IMPORTED GLOBAL)
- set_target_properties(calrt_prebuilt PROPERTIES
- # IMPORTED_LOCATION "${CALRT_INSTALL_DIR}/lib/libhostrt.a"
- IMPORTED_LOCATION "${CALRT_LIB}"
- INTERFACE_INCLUDE_DIRECTORIES "${CALRT_INSTALL_DIR}/include;${CALRT_INSTALL_DIR}/include/calrt"
- INSTALL_RPATH_USE_LINK_PATH TRUE
- )
-
- list(APPEND CALRT_LIBRARIES
- calrt_prebuilt
- elf
- )
-
- message(STATUS "CALRT: CALRT_INSTALL_DIR = ${CALRT_INSTALL_DIR}")
- endif()
-
- target_link_libraries(llama PUBLIC ${CALRT_LIBRARIES})
-
- message(STATUS "CALRT: CALRT_LIBRARIES = ${CALRT_LIBRARIES}")
-
-endif()
-
if (BUILD_SHARED_LIBS)
set_target_properties(llama PROPERTIES POSITION_INDEPENDENT_CODE ON)
target_compile_definitions(llama PRIVATE LLAMA_BUILD)
diff --git a/src/llama-sampling.cpp b/src/llama-sampling.cpp
index cf11eebe..55d2e355 100644
--- a/src/llama-sampling.cpp
+++ b/src/llama-sampling.cpp
@@ -1,6 +1,5 @@
#include "llama-sampling.h"
-#include "llama-calrt.h"
#include "llama-impl.h"
#include "llama-vocab.h"
#include "llama-grammar.h"
diff --git a/src/llama.cpp b/src/llama.cpp
index a9cb2859..c84e5696 100644
--- a/src/llama.cpp
+++ b/src/llama.cpp
@@ -1,4 +1,3 @@
-#include "llama-calrt.h"
#include "llama-context.h"
#include "llama-impl.h"
@@ -143,11 +142,9 @@ static int llama_model_load(const std::string & fname, std::vector<std::string>
return 0;
}
-#ifndef USE_CALRT
if (!model.load_tensors(ml)) {
return -2;
}
-#endif
} catch (const std::exception & err) {
LLAMA_LOG_ERROR("%s: error loading model: %s\n", __func__, err.what());
return -1;
@@ -418,4 +415,3 @@ const char * llama_print_system_info(void) {
return s.c_str();
}
-
diff --git a/tools/mtmd/clip.cpp b/tools/mtmd/clip.cpp
index d119dd12..fb40051d 100644
--- a/tools/mtmd/clip.cpp
+++ b/tools/mtmd/clip.cpp
@@ -9,7 +9,6 @@
#include "ggml-alloc.h"
#include "ggml-backend.h"
#include "gguf.h"
-#include "src/llama-calrt.h"
#include <cassert>
#include <cmath>
@@ -2560,9 +2559,6 @@ struct clip_model_loader {
}
}
-#ifdef USE_CALRT
-
-#else
// tensors
{
for (int i = 0; i < n_tensors; ++i) {
@@ -2576,7 +2572,6 @@ struct clip_model_loader {
__func__, i, ggml_n_dims(cur), cur->name, tensor_size, offset, cur->ne[0], cur->ne[1], cur->ne[2], cur->ne[3], ggml_type_name(type));
}
}
-#endif
}
void load_hparams(clip_model & model, clip_modality modality) {
@@ -3219,9 +3214,6 @@ struct clip_model_loader {
default:
GGML_ASSERT(false && "unknown projector type");
}
-#ifdef USE_CALRT
-
-#else
// load data
{
std::vector<uint8_t> read_buf;
@@ -3257,7 +3249,6 @@ struct clip_model_loader {
LOG_DBG("%s: loaded %zu tensors from %s\n", __func__, tensors_to_load.size(), fname.c_str());
}
-#endif
}
@@ -3499,22 +3490,14 @@ struct clip_init_result clip_init(const char * fname, struct clip_context_params
ctx_vision = new clip_ctx(ctx_params);
loader.load_hparams(ctx_vision->model, CLIP_MODALITY_VISION);
loader.load_tensors(*ctx_vision); // load some config tensor
-#ifdef USE_CALRT
-
-#else
loader.warmup(*ctx_vision);
-#endif
}
if (loader.has_audio) {
ctx_audio = new clip_ctx(ctx_params);
loader.load_hparams(ctx_audio->model, CLIP_MODALITY_AUDIO);
loader.load_tensors(*ctx_audio);
-#ifdef USE_CALRT
-
-#else
loader.warmup(*ctx_audio);
-#endif
}
} catch (const std::exception & e) {
@@ -4672,317 +4655,6 @@ static std::vector<std::vector<float>> get_2d_sincos_pos_embed(int embed_dim, co
return pos_embed_2d;
}
-#ifdef USE_CALRT
-bool clip_image_encode(struct clip_ctx * clip_ctx, struct calrt_context * ctx, clip_image_f32 * img, float * vec) {
- clip_image_f32_batch imgs;
- clip_image_f32_ptr img_copy(clip_image_f32_init());
- *img_copy = *img;
- imgs.entries.push_back(std::move(img_copy));
-
- return clip_image_batch_encode(clip_ctx,ctx, &imgs, vec);
-}
-
-bool clip_image_batch_encode(struct clip_ctx * clip_ctx, calrt_context * ctx, const clip_image_f32_batch * imgs_c_ptr, float * vec) {
- const clip_image_f32_batch & imgs = *imgs_c_ptr;
- int batch_size = imgs.entries.size();
- if (batch_size != 1) {
- return false; // only support batch size of 1
- }
-
- // const int image_size_width = imgs.entries[0]->nx;
- // const int image_size_height = imgs.entries[0]->ny;
-
- // const int patch_size = hparams.patch_size;
- // const int num_patches = ((image_size_width / patch_size) * (image_size_height / patch_size));
- // const int n_pos = num_patches + (model.class_embedding ? 1 : 0);
- // const int pos_w = image_size_width / patch_size;
- // const int pos_h = image_size_height / patch_size;
-
- // const bool use_window_attn = hparams.n_wa_pattern > 0; // for qwen2.5vl
- // build the inference graph
-
- std::vector<float> inp_raw;
- projector_type proj = clip_ctx->proj_type();
- std::vector<float> OPENAI_CLIP_MEAN = {0.48145466f, 0.4578275f, 0.40821073f};
- std::vector<float> OPENAI_CLIP_STD = {0.26862954f, 0.26130258f, 0.27577711f};
- switch (proj) {
- case PROJECTOR_TYPE_MINICPMV:
- {
- OPENAI_CLIP_MEAN = {0.485, 0.456, 0.406};
- OPENAI_CLIP_STD = {0.229, 0.224, 0.225};
- }break;
- case PROJECTOR_TYPE_QWEN25VL:
- {
- OPENAI_CLIP_MEAN = {0.48145466f, 0.4578275f, 0.40821073f};
- OPENAI_CLIP_STD = {0.26862954f, 0.26130258f, 0.27577711f};
- }break;
- }
-
- // set input pixel values
- if (!imgs.is_audio) {
- size_t nelem = 0;
- for (const auto & img : imgs.entries) {
- nelem += img->nx * img->ny * 3;
- }
- inp_raw.resize(nelem);
-
- // layout of data (note: the channel dim is unrolled to better visualize the layout):
- //
- // ┌──W──┐
- // │ H │ channel = R
- // ├─────┤ │
- // │ H │ channel = G
- // ├─────┤ │
- // │ H │ channel = B
- // └─────┘ │
- // ──────┘ x B
-
- for (size_t i = 0; i < imgs.entries.size(); i++) {
- const int nx = imgs.entries[i]->nx;
- const int ny = imgs.entries[i]->ny;
- const int n = nx * ny;
-
- for (int b = 0; b < batch_size; b++) {
- float * batch_entry = inp_raw.data() + b * (3*n);
- //float * batch_entry_copy = inp_raw.data() + b * (3*n) + nelem / 2;
- for (int y = 0; y < ny; y++) {
- for (int x = 0; x < nx; x++) {
- size_t base_src = 3*(y * nx + x); // idx of the first channel
- size_t base_dst = y * nx + x; // idx of the first channel
- // -------------------------- 核心:缩放+标准化 --------------------------
- // 1. 读取原始RGB值(假设buf为0-255的整数,转为float
- float r = static_cast<float>(imgs.entries[b]->buf[base_src]); // R通道
- float g = static_cast<float>(imgs.entries[b]->buf[base_src + 1]); // G通道
- float b_val = static_cast<float>(imgs.entries[b]->buf[base_src + 2]); // B通道(避名)
-
- // 2. 缩放:×1/255(归一化到0-1
- r *= (1.0f / 255.0f);
- g *= (1.0f / 255.0f);
- b_val *= (1.0f / 255.0f);
-
- // 3. 标准化:(x - mean) / std(使用CLIP的均值和标准差)
- r = (r - OPENAI_CLIP_MEAN[0]) / OPENAI_CLIP_STD[0]; // R通道标准化
- g = (g - OPENAI_CLIP_MEAN[1]) / OPENAI_CLIP_STD[1]; // G通道标准化
- b_val = (b_val - OPENAI_CLIP_MEAN[2]) / OPENAI_CLIP_STD[2]; // B通道标准化
- // ----------------------------------------------------------------------
-
- // 存入batch_entry(通道顺序:R→0~n-1G→n~2n-1B→2n~3n-1,与原代码一致)
- batch_entry[base_dst] = r; // R通道
- batch_entry[1 * n + base_dst] = g; // G通道(1×n偏移)
- batch_entry[2 * n + base_dst] = b_val; // B通道(2×n偏移)
-
- // batch_entry[ base_dst] = imgs.entries[b]->buf[base_src ];
- // batch_entry[1*n + base_dst] = imgs.entries[b]->buf[base_src + 1];
- // batch_entry[2*n + base_dst] = imgs.entries[b]->buf[base_src + 2];
-
- }
- }
- }
- }
- } else {
- // audio input
- GGML_ASSERT(imgs.entries.size() == 1);
- const auto & mel_inp = imgs.entries[0];
- const int n_step = mel_inp->nx;
- const int n_mel = mel_inp->ny;
- inp_raw.resize(n_step * n_mel);
- std::memcpy(inp_raw.data(), mel_inp->buf.data(), n_step * n_mel * sizeof(float));
- }
-
- const int C = 3; // 通道数(RGB
- int H = 896; // 图像高度
- int W = 896; // 图像宽度
- const int num_patches_h = 14; // Height方向分块数
- const int num_patches_w = 14; // Width方向分块数
- int vision_len = 1; // Batch大小
- int num_views = 1; // 视角/分支数(如时间步、参考帧与当前帧)
-
- switch (proj) {
- case PROJECTOR_TYPE_MINICPMV:
- {
- H = 448;
- W = 448;
- vision_len = 1;
- num_views = 1;
- // 目标数组形状:[batch_size, num_views, C, num_patches_h, num_patches_w]
- std::vector<float> target(
- vision_len * num_views * C * H * W
- );
- image_to_patches_minicpm(inp_raw, target);
-
- size_t output_size = 36;
-
- ctx->encode((char*)target.data(), target.size(), (char*)vec, output_size);
-
- }break;
- case PROJECTOR_TYPE_QWEN25VL:
- {
- H = 896;
- W = 896;
- vision_len = 4096;
- num_views = 2;
- // 目标数组形状:[batch_size, num_views, C, num_patches_h, num_patches_w]
- std::vector<float> target(
- vision_len * num_views * C * H * W
- );
- image_to_patches_qwen(inp_raw, target);
-
- size_t output_size = 36;
-
- ctx->encode((char*)target.data(), target.size(), (char*)vec, output_size);
-
- }break;
- }
-
- return true;
-}
-
-void image_to_patches_qwen(std::vector<float> input, std::vector<float>& output){
- // --------------------- 2. 分块参数计算 ---------------------
- const int C = 3; // 通道数(RGB
- const int H = 896; // 图像高度
- const int W = 896; // 图像宽度
- const int num_patches_h = 14; // Height方向分块数
- const int num_patches_w = 14; // Width方向分块数
- const int patch_h = H / num_patches_h; // 单个Patch的高度(64
- const int patch_w = W / num_patches_w; // 单个Patch的宽度(64
-
- // --------------------- 3. 提取每个Patch的统计特征(示例:均值) ---------------------
- // patch_means[ph][pw][c] 存储第ph行、第pw列Patch的第c通道均值
- std::vector<std::vector<std::vector<float>>> patch_means(
- num_patches_h,
- std::vector<std::vector<float>>(num_patches_w, std::vector<float>(C))
- );
-
- for (int c = 0; c < C; ++c) {
- for (int ph = 0; ph < num_patches_h; ++ph) {
- for (int pw = 0; pw < num_patches_w; ++pw) {
- float sum = 0.0f;
- int count = 0;
- // 遍历Patch内所有像素
- for (int h = ph * patch_h; h < (ph + 1) * patch_h; ++h) {
- for (int w = pw * patch_w; w < (pw + 1) * patch_w; ++w) {
- int idx = c * H * W + h * W + w; // 计算inp_raw中像素的索引
- sum += input[idx];
- count++;
- }
- }
- patch_means[ph][pw][c] = sum / count; // 计算该Patch的通道均值
- }
- }
- }
-
- // --------------------- 4. 构建目标维度:4096×2×3×14×14 ---------------------
- const int vision_len = 4096; // Batch大小
- const int num_views = 2; // 视角/分支数(如时间步、参考帧与当前帧)
- // 目标数组形状:[batch_size, num_views, C, num_patches_h, num_patches_w]
- // std::vector<float> target(
- // vision_len * num_views * C * num_patches_h * num_patches_w
- // );
-
- // 填充目标数组:每个Batch元素、视角、通道、Patch位置均使用对应Patch的均值
- for (int n = 0; n < vision_len; ++n) {
- for (int m = 0; m < num_views; ++m) {
- for (int c = 0; c < C; ++c) {
- for (int ph = 0; ph < num_patches_h; ++ph) {
- for (int pw = 0; pw < num_patches_w; ++pw) {
- // 计算目标数组的一维索引
- int target_idx = n * num_views * C * num_patches_h * num_patches_w
- + m * C * num_patches_h * num_patches_w
- + c * num_patches_h * num_patches_w
- + ph * num_patches_w + pw;
- output[target_idx] = patch_means[ph][pw][c];
- }
- }
- }
- }
- }
-
- // --------------------- 5. 验证输出(可选) ---------------------
- // std::cout << "目标数组总元素数: " << target.size() << "(预期: "
- // << vision_len * num_views * C * num_patches_h * num_patches_w << "" << endl;
-
- // // 打印第一个Batch、第一个视角、第一个通道的所有Patch值(前几个)
- // std::cout << "\n第一个Batch、第一个视角、第一个通道的Patch均值(前20个):" << endl;
- // for (int ph = 0; ph < std::min(5, num_patches_h); ++ph) { // 只打印前5个Patch的行
- // for (int pw = 0; pw < std::min(5, num_patches_w); ++pw) { // 只打印前5个Patch的列
- // int idx = 0 * num_views * C * num_patches_h * num_patches_w
- // + 0 * C * num_patches_h * num_patches_w
- // + 0 * num_patches_h * num_patches_w
- // + ph * num_patches_w + pw;
- // cout << target[idx] << "\t";
- // }
- // cout << endl;
- // }
-}
-void image_to_patches_minicpm(std::vector<float> input, std::vector<float>& output){
- // --------------------- 2. 分块参数计算 ---------------------
- const int C = 3; // 通道数(RGB
- const int H = 448; // 图像高度
- const int W = 448; // 图像宽度
- const int num_patches_h = 14; // Height方向分块数
- const int num_patches_w = 14; // Width方向分块数
- const int patch_h = H / num_patches_h; // 单个Patch的高度(64
- const int patch_w = W / num_patches_w; // 单个Patch的宽度(64
-
- // --------------------- 3. 提取每个Patch的统计特征(示例:均值) ---------------------
- // patch_means[ph][pw][c] 存储第ph行、第pw列Patch的第c通道均值
- std::vector<std::vector<std::vector<float>>> patch_means(
- num_patches_h,
- std::vector<std::vector<float>>(num_patches_w, std::vector<float>(C))
- );
-
- for (int c = 0; c < C; ++c) {
- for (int ph = 0; ph < num_patches_h; ++ph) {
- for (int pw = 0; pw < num_patches_w; ++pw) {
- float sum = 0.0f;
- int count = 0;
- // 遍历Patch内所有像素
- for (int h = ph * patch_h; h < (ph + 1) * patch_h; ++h) {
- for (int w = pw * patch_w; w < (pw + 1) * patch_w; ++w) {
- int idx = c * H * W + h * W + w; // 计算inp_raw中像素的索引
- sum += input[idx];
- count++;
- }
- }
- patch_means[ph][pw][c] = sum / count; // 计算该Patch的通道均值
- }
- }
- }
-
- // --------------------- 4. 构建目标维度:3×448×448 ---------------------
- const int vision_len = 1; // Batch大小
- const int num_views = 1; // 视角/分支数(如时间步、参考帧与当前帧)
- // 目标数组形状:[C, H, W]
- for (int c = 0; c < C; ++c) {
- for (int ph = 0; ph < num_patches_h; ++ph) {
- for (int pw = 0; pw < num_patches_w; ++pw) {
- // 计算目标数组的一维索引
- int target_idx = c * H * W + ph * patch_h * H + pw * patch_w;
- output[target_idx] = patch_means[ph][pw][c];
- }
- }
- }
- // --------------------- 5. 验证输出(可选) ---------------------
- // std::cout << "目标数组总元素数: " << target.size() << "(预期: "
- // << vision_len * num_views * C * num_patches_h * num_patches_w << "" << endl;
-
- // // 打印第一个Batch、第一个视角、第一个通道的所有Patch值(前几个)
- // std::cout << "\n第一个Batch、第一个视角、第一个通道的Patch均值(前20个):" << endl;
- // for (int ph = 0; ph < std::min(5, num_patches_h); ++ph) { // 只打印前5个Patch的行
- // for (int pw = 0; pw < std::min(5, num_patches_w); ++pw) { // 只打印前5个Patch的列
- // int idx = 0 * num_views * C * num_patches_h * num_patches_w
- // + 0 * C * num_patches_h * num_patches_w
- // + 0 * num_patches_h * num_patches_w
- // + ph * num_patches_w + pw;
- // cout << target[idx] << "\t";
- // }
- // cout << endl;
- // }
-}
-
-#endif
-
bool clip_image_encode(struct clip_ctx * ctx, const int n_threads, clip_image_f32 * img, float * vec) {
clip_image_f32_batch imgs;
clip_image_f32_ptr img_copy(clip_image_f32_init());
diff --git a/tools/mtmd/clip.h b/tools/mtmd/clip.h
index 513fe4f3..2eaca33e 100644
--- a/tools/mtmd/clip.h
+++ b/tools/mtmd/clip.h
@@ -1,7 +1,6 @@
#pragma once
#include "ggml.h"
-#include "src/llama-calrt.h"
#include <stddef.h>
#include <stdint.h>
@@ -99,10 +98,6 @@ struct ggml_tensor * clip_get_newline_tensor(const struct clip_ctx * ctx);
bool clip_image_encode (struct clip_ctx * ctx, int n_threads, struct clip_image_f32 * img, float * vec);
bool clip_image_batch_encode(struct clip_ctx * ctx, int n_threads, const struct clip_image_f32_batch * imgs, float * vec);
-#ifdef USE_CALRT
-bool clip_image_encode(struct clip_ctx * clip_ctx, struct calrt_context * ctx, clip_image_f32 * img, float * vec);
-bool clip_image_batch_encode(struct clip_ctx * clip_ctx, calrt_context * ctx, const clip_image_f32_batch * imgs_c_ptr, float * vec);
-#endif
int clip_is_minicpmv(const struct clip_ctx * ctx);
bool clip_is_glm(const struct clip_ctx * ctx);
@@ -118,7 +113,3 @@ void clip_image_f32_batch_add_mel(struct clip_image_f32_batch * batch, int n_mel
bool clip_has_vision_encoder(const struct clip_ctx * ctx);
bool clip_has_audio_encoder(const struct clip_ctx * ctx);
bool clip_has_whisper_encoder(const struct clip_ctx * ctx);
-
-// patch
-void image_to_patches_qwen(std::vector<float> input, std::vector<float>& output);
-void image_to_patches_minicpm(std::vector<float> input, std::vector<float>& output);
diff --git a/tools/mtmd/mtmd-helper.cpp b/tools/mtmd/mtmd-helper.cpp
index 82b98d90..686f42f3 100644
--- a/tools/mtmd/mtmd-helper.cpp
+++ b/tools/mtmd/mtmd-helper.cpp
@@ -165,76 +165,6 @@ struct decode_embd_batch {
};
// Helper function for decoding an image whose embeddings have already been calculated
-#ifdef USE_CALRT
-int32_t mtmd_helper_decode_image_chunk(
- mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunk * chunk,
- float * encoded_embd,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- llama_pos * new_n_past) {
- auto chunk_type = mtmd_input_chunk_get_type(chunk);
- const char * name = chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE ? "image" : "audio";
- if (chunk_type == MTMD_INPUT_CHUNK_TYPE_TEXT) {
- LOG_ERR("failed to decode chunk: input chunk not of image/audio type\n");
- return -1;
- }
-
- const llama_model * model = llama_get_model(calrt_get_llama_context(lctx));
- int n_mmproj_embd = llama_model_n_embd(model);
- int n_pos_per_embd = mtmd_decode_use_mrope(ctx) ? 4 : 1;
-
- int32_t n_tokens = mtmd_input_chunk_get_n_tokens(chunk);
- int32_t i_batch = 0;
- int32_t n_img_batches = GGML_PAD(n_tokens, n_batch) / n_batch;
- decode_embd_batch batch_embd(encoded_embd, n_tokens, n_pos_per_embd, n_mmproj_embd);
-
- if (mtmd_decode_use_mrope(ctx)) {
- if (chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE) {
- const auto image_tokens = mtmd_input_chunk_get_tokens_image(chunk);
- if (!image_tokens) {
- LOG_ERR("failed to decode chunk: image tokens are null\n");
- return -1;
- }
- const int nx = mtmd_image_tokens_get_nx(image_tokens);
- const int ny = mtmd_image_tokens_get_ny(image_tokens);
- batch_embd.set_position_mrope_2d(n_past, nx, ny, seq_id);
- } else if (chunk_type == MTMD_INPUT_CHUNK_TYPE_AUDIO) {
- batch_embd.set_position_mrope_1d(n_past, seq_id);
- } else {
- GGML_ABORT("invalid chunk type for M-RoPE");
- }
- } else {
- batch_embd.set_position_normal(n_past, seq_id);
- }
-
- while (i_batch < n_img_batches) { // split into batches
- int pos_offset = i_batch*n_batch;
- int n_tokens_batch = std::min(n_batch, n_tokens - pos_offset);
- llama_batch batch_embd_view = batch_embd.get_view(pos_offset, n_tokens_batch);
-
- LOG_INF("decoding %s batch %d/%d, n_tokens_batch = %d\n", name, i_batch+1, n_img_batches, n_tokens_batch);
-
- int64_t t1 = ggml_time_ms();
- int32_t ret = calrt_decode(lctx, batch_embd_view);
- if (ret != 0) {
- LOG_ERR("failed to decode %s\n", name);
- return ret;
- }
-
- LOG_INF("%s decoded (batch %d/%d) in %" PRId64 " ms\n", name, i_batch+1, n_img_batches, ggml_time_ms() - t1);
-
- i_batch++;
- }
-
- n_past += mtmd_input_chunk_get_n_pos(chunk);
- *new_n_past = n_past;
-
- return 0;
-}
-#else
int32_t mtmd_helper_decode_image_chunk(
mtmd_context * ctx,
struct llama_context * lctx,
@@ -312,80 +242,7 @@ int32_t mtmd_helper_decode_image_chunk(
}
return 0;
}
-#endif
-#ifdef USE_CALRT
-int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunk * chunk,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- bool logits_last,
- llama_pos * new_n_past) {
- int32_t ret;
- llama_batch text_batch = llama_batch_init(n_batch, 0, 1);
- auto chunk_type = mtmd_input_chunk_get_type(chunk);
- if (chunk_type == MTMD_INPUT_CHUNK_TYPE_TEXT) {
- size_t n_tokens;
- const auto tokens = mtmd_input_chunk_get_tokens_text(chunk, &n_tokens);
- // LOG_INF("decoding text chunk, n_tokens = %zu\n", n_tokens);
- size_t i = 0;
- while (i < n_tokens) { // split into batches
- text_batch.n_tokens = 0; // clear the batch
- for (; i < n_tokens && text_batch.n_tokens < n_batch; i++) {
- int32_t j = text_batch.n_tokens;
- text_batch.token [j] = tokens[i];
- text_batch.pos [j] = n_past++;
- text_batch.n_seq_id[j] = 1;
- text_batch.seq_id [j][0] = seq_id;
- text_batch.logits [j] = false;
-
- text_batch.n_tokens++;
- }
- bool is_last_token = (i == n_tokens);
- if (logits_last && is_last_token) {
- text_batch.logits[text_batch.n_tokens - 1] = true;
- }
- ret = calrt_decode(lctx, text_batch);
- if (ret != 0) {
- LOG_ERR("failed to decode text\n");
- llama_batch_free(text_batch);
- return ret;
- }
- *new_n_past += text_batch.n_tokens;
- }
-
- } else if (chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE || chunk_type == MTMD_INPUT_CHUNK_TYPE_AUDIO) {
- const char * name = chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE ? "image" : "audio";
- int64_t t0 = ggml_time_ms();
-
- LOG_INF("encoding %s slice...\n", name);
-
- ret = mtmd_encode_chunk(ctx, chunk);
- if (ret != 0) {
- LOG_ERR("failed to encode %s slice\n", name);
- llama_batch_free(text_batch);
- return ret;
- }
-
- LOG_INF("%s slice encoded in %" PRId64 " ms\n", name, ggml_time_ms() - t0);
-
- float * embd = mtmd_get_output_embd(ctx);
- ret = mtmd_helper_decode_image_chunk(ctx, lctx, chunk, embd, n_past, seq_id, n_batch, new_n_past);
- if (ret != 0) {
- LOG_ERR("failed to decode %s\n", name);
- llama_batch_free(text_batch);
- return ret;
- }
- } else {
- GGML_ABORT("chunk type not supported");
- }
-
- llama_batch_free(text_batch);
- return 0;
-}
-#else
int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
struct llama_context * lctx,
const mtmd_input_chunk * chunk,
@@ -457,37 +314,7 @@ int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
llama_batch_free(text_batch);
return 0;
}
-#endif
-#ifdef USE_CALRT
-int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunks * chunks,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- bool logits_last,
- llama_pos * new_n_past) {
- size_t n_chunks = mtmd_input_chunks_size(chunks);
- if (n_chunks == 0) {
- LOG_ERR("no chunks to eval\n");
- return 0;
- }
-
- for (size_t i = 0; i < n_chunks; i++) {
- bool chunk_logits_last = (i == n_chunks - 1) && logits_last;
- auto chunk = mtmd_input_chunks_get(chunks, i);
-
- int32_t res = mtmd_helper_eval_chunk_single(ctx, lctx, chunk, n_past, seq_id, n_batch, chunk_logits_last, &n_past);
- if (res != 0) {
- LOG_ERR("failed to eval chunk %zu\n", i);
- return res;
- }
- *new_n_past = n_past;
- }
- return 0;
-}
-#else
int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
struct llama_context * lctx,
const mtmd_input_chunks * chunks,
@@ -516,7 +343,6 @@ int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
return 0;
}
-#endif
namespace audio_helpers {
diff --git a/tools/mtmd/mtmd-helper.h b/tools/mtmd/mtmd-helper.h
index cc179ee9..49e6be4f 100644
--- a/tools/mtmd/mtmd-helper.h
+++ b/tools/mtmd/mtmd-helper.h
@@ -48,16 +48,6 @@ MTMD_API llama_pos mtmd_helper_get_n_pos(const mtmd_input_chunks * chunks);
// if any of the mtmd_encode() or llama_decode() calls return non-zero, stop and forward the error
// otherwise, returns 0 on success
// this function is NOT thread-safe
-#ifdef USE_CALRT
-MTMD_API int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunks * chunks,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- bool logits_last,
- llama_pos * new_n_past);
-#else
MTMD_API int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
struct llama_context * lctx,
const mtmd_input_chunks * chunks,
@@ -66,21 +56,10 @@ MTMD_API int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,
int32_t n_batch,
bool logits_last,
llama_pos * new_n_past);
-#endif
// works like mtmd_helper_eval_chunks(), but only for a single chunk
// this function is NOT thread-safe
-#ifdef USE_CALRT
-MTMD_API int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunk * chunk,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- bool logits_last,
- llama_pos * new_n_past);
-#else
MTMD_API int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
struct llama_context * lctx,
const mtmd_input_chunk * chunk,
@@ -89,21 +68,10 @@ MTMD_API int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,
int32_t n_batch,
bool logits_last,
llama_pos * new_n_past);
-#endif
// helper function to decode an image whose embeddings have already been calculated
// this helper will handle batching and pre/post decoding setup (for ex. gemma 3 requires non-causal attention)
// ret 0 on success, -1 on chunk not being a valid image chunk, 1 on decode failure
-#ifdef USE_CALRT
-MTMD_API int32_t mtmd_helper_decode_image_chunk(mtmd_context * ctx,
- struct calrt_context * lctx,
- const mtmd_input_chunk * chunk,
- float * encoded_embd,
- llama_pos n_past,
- llama_seq_id seq_id,
- int32_t n_batch,
- llama_pos * new_n_past);
-#else
MTMD_API int32_t mtmd_helper_decode_image_chunk(mtmd_context * ctx,
struct llama_context * lctx,
const mtmd_input_chunk * chunk,
@@ -112,7 +80,6 @@ MTMD_API int32_t mtmd_helper_decode_image_chunk(mtmd_context * ctx,
llama_seq_id seq_id,
int32_t n_batch,
llama_pos * new_n_past);
-#endif
#ifdef __cplusplus
} // extern "C"
diff --git a/tools/mtmd/mtmd.cpp b/tools/mtmd/mtmd.cpp
index a2b78cde..c865a65b 100644
--- a/tools/mtmd/mtmd.cpp
+++ b/tools/mtmd/mtmd.cpp
@@ -4,7 +4,6 @@
#include "mtmd-audio.h"
#include "llama.h"
-#include "src/llama-calrt.h"
// fix problem with std::min and std::max
#if defined(_WIN32)
@@ -116,7 +115,6 @@ mtmd_context_params mtmd_context_params_default() {
}
struct mtmd_context {
- struct calrt_context * cal_ctx_v;
struct clip_ctx * ctx_v; // vision
struct clip_ctx * ctx_a; // audio
const struct llama_model * text_model;
@@ -356,11 +354,6 @@ struct mtmd_context {
~mtmd_context() {
clip_free(ctx_a);
clip_free(ctx_v);
- calrt_context_free(cal_ctx_v);
- }
-
- void set_calrt_context(struct calrt_context * ctx) {
- cal_ctx_v = ctx;
}
private:
@@ -813,27 +806,6 @@ int32_t mtmd_encode(mtmd_context * ctx, const mtmd_image_tokens * image_tokens)
}
int n_mmproj_embd = clip_n_mmproj_embd(ctx_clip);
ctx->image_embd_v.resize(image_tokens->n_tokens() * n_mmproj_embd);
-#ifdef USE_CALRT
- calrt_context * cal_ctx = ctx->cal_ctx_v;
- if (!cal_ctx) {
- LOG_ERR("%s: this API does not support non-vision input, please use mtmd_encode_chunk instead\n", __func__);
- return 1;
- }
- bool ok = false;
-
- if (clip_is_minicpmv(ctx->ctx_v)) {
- const auto & entries = image_tokens->batch_f32.entries;
- for (size_t i = 0; i < entries.size(); i++) {
- int n_tokens_per_image = clip_n_output_tokens(ctx_clip, entries[i].get());
- ok = clip_image_encode(ctx->ctx_v,
- cal_ctx,
- entries[i].get(),
- ctx->image_embd_v.data() + i*n_mmproj_embd*n_tokens_per_image);
- }
- } else {
- ok = clip_image_batch_encode(ctx->ctx_v, cal_ctx, &image_tokens->batch_f32, ctx->image_embd_v.data());
- }
-#else
bool ok = false;
if (clip_is_llava(ctx_clip)
@@ -856,7 +828,6 @@ int32_t mtmd_encode(mtmd_context * ctx, const mtmd_image_tokens * image_tokens)
&image_tokens->batch_f32,
ctx->image_embd_v.data());
}
-#endif
return ok ? 0 : 1;
}
@@ -1124,9 +1095,3 @@ mtmd_input_chunks * mtmd_test_create_input_chunks() {
return chunks;
}
-
-#ifdef USE_CALRT
-MTMD_API void mtmd_set_calrt_ctx(mtmd_context * ctx, calrt_context * cal_ctx) {
- ctx->set_calrt_context(cal_ctx);
-}
-#endif
diff --git a/tools/mtmd/mtmd.h b/tools/mtmd/mtmd.h
index cf05e49d..775fba62 100644
--- a/tools/mtmd/mtmd.h
+++ b/tools/mtmd/mtmd.h
@@ -220,10 +220,6 @@ MTMD_API float * mtmd_get_output_embd(mtmd_context * ctx);
// test function, to be used in test-mtmd-c-api.c
MTMD_API mtmd_input_chunks * mtmd_test_create_input_chunks(void);
-#ifdef USE_CALRT
-MTMD_API void mtmd_set_calrt_ctx(mtmd_context * ctx, struct calrt_context * cal_ctx);
-#endif
-
#ifdef __cplusplus
} // extern "C"
#endif
diff --git a/tools/server/CMakeLists.txt b/tools/server/CMakeLists.txt
index 111faf8c..b0ec000a 100644
--- a/tools/server/CMakeLists.txt
+++ b/tools/server/CMakeLists.txt
@@ -2,6 +2,52 @@ set(TARGET llama-server)
include_directories(${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_BINARY_DIR})
+option(LLAMA_SERVER_USE_CAL_LLM "Use cal-llm EngineCore through the C API for llama-server generation." OFF)
+
+if (LLAMA_SERVER_USE_CAL_LLM)
+ set(CAL_LLM_ROOT "" CACHE PATH "Path to the installed cal-llm C++ SDK prefix.")
+ set(CALRT_LIBRARY "" CACHE FILEPATH "Path to the CalRT shared library used by cal-llm.")
+ if (NOT CAL_LLM_ROOT)
+ message(FATAL_ERROR "LLAMA_SERVER_USE_CAL_LLM requires CAL_LLM_ROOT.")
+ endif()
+ if (NOT CALRT_LIBRARY)
+ message(FATAL_ERROR "LLAMA_SERVER_USE_CAL_LLM requires CALRT_LIBRARY.")
+ endif()
+ if (NOT IS_ABSOLUTE "${CALRT_LIBRARY}")
+ message(FATAL_ERROR "CALRT_LIBRARY must be an absolute path: ${CALRT_LIBRARY}")
+ endif()
+ if (NOT EXISTS "${CALRT_LIBRARY}" OR IS_DIRECTORY "${CALRT_LIBRARY}")
+ message(FATAL_ERROR "CALRT_LIBRARY does not point to an existing shared library file: ${CALRT_LIBRARY}")
+ endif()
+ find_path(CAL_LLM_INCLUDE_DIR
+ NAMES c_api.h
+ PATHS "${CAL_LLM_ROOT}/include"
+ NO_DEFAULT_PATH
+ )
+ find_library(CAL_LLM_LIBRARY
+ NAMES cal_llm
+ PATHS "${CAL_LLM_ROOT}/lib64" "${CAL_LLM_ROOT}/lib"
+ NO_DEFAULT_PATH
+ )
+ if (NOT CAL_LLM_INCLUDE_DIR)
+ message(FATAL_ERROR "CAL_LLM_ROOT does not contain include/c_api.h: ${CAL_LLM_ROOT}")
+ endif()
+ if (NOT CAL_LLM_LIBRARY)
+ message(FATAL_ERROR "CAL_LLM_ROOT does not contain lib64/libcal_llm.so or lib/libcal_llm.so: ${CAL_LLM_ROOT}")
+ endif()
+ add_library(cal_llm::c_api SHARED IMPORTED GLOBAL)
+ set_target_properties(cal_llm::c_api PROPERTIES
+ IMPORTED_LOCATION "${CAL_LLM_LIBRARY}"
+ INTERFACE_INCLUDE_DIRECTORIES "${CAL_LLM_INCLUDE_DIR}"
+ )
+ add_library(calrt::calrt SHARED IMPORTED GLOBAL)
+ set_target_properties(calrt::calrt PROPERTIES
+ IMPORTED_LOCATION "${CALRT_LIBRARY}"
+ )
+ get_filename_component(CAL_LLM_LIBRARY_DIR "${CAL_LLM_LIBRARY}" DIRECTORY)
+ get_filename_component(CALRT_LIBRARY_DIR "${CALRT_LIBRARY}" DIRECTORY)
+endif()
+
if (MINGW)
# fix: https://github.com/ggml-org/llama.cpp/actions/runs/9651004652/job/26617901362?pr=8006
add_compile_definitions(_WIN32_WINNT=${GGML_WIN_VER})
@@ -37,6 +83,24 @@ target_include_directories(${TARGET} PRIVATE ../mtmd ../../)
target_include_directories(${TARGET} PRIVATE ${CMAKE_SOURCE_DIR})
target_link_libraries(${TARGET} PRIVATE common mtmd ${CMAKE_THREAD_LIBS_INIT})
+if (LLAMA_SERVER_USE_CAL_LLM)
+ target_compile_definitions(${TARGET} PRIVATE LLAMA_SERVER_USE_CAL_LLM)
+ if (CMAKE_SYSTEM_NAME STREQUAL "Linux")
+ target_link_libraries(${TARGET} PRIVATE
+ cal_llm::c_api
+ "-Wl,--push-state,--no-as-needed"
+ calrt::calrt
+ "-Wl,--pop-state"
+ )
+ else()
+ target_link_libraries(${TARGET} PRIVATE cal_llm::c_api calrt::calrt)
+ endif()
+ set_property(TARGET ${TARGET} APPEND PROPERTY BUILD_RPATH "${CAL_LLM_LIBRARY_DIR}")
+ set_property(TARGET ${TARGET} APPEND PROPERTY BUILD_RPATH "${CALRT_LIBRARY_DIR}")
+ set_property(TARGET ${TARGET} APPEND PROPERTY INSTALL_RPATH "${CAL_LLM_LIBRARY_DIR}")
+ set_property(TARGET ${TARGET} APPEND PROPERTY INSTALL_RPATH "${CALRT_LIBRARY_DIR}")
+endif()
+
if (WIN32)
TARGET_LINK_LIBRARIES(${TARGET} PRIVATE ws2_32)
endif()
diff --git a/tools/server/README.md b/tools/server/README.md
index c16d0bd6..bd21916d 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -151,6 +151,8 @@ The project is under active development, and we are [looking for feedback and co
| Argument | Explanation |
| -------- | ----------- |
+| `--cal-llm-profile` | enable cal-llm backend profiling logs<br/>(env: LLAMA_ARG_CAL_LLM_PROFILE) |
+| `--cal-llm-profile-dump FILE` | write cal-llm backend profiling JSONL to file<br/>(env: LLAMA_ARG_CAL_LLM_PROFILE_DUMP) |
| `--swa-checkpoints N` | max number of SWA checkpoints per slot to create (default: 3)<br/>[(more info)](https://github.com/ggml-org/llama.cpp/pull/15293)<br/>(env: LLAMA_ARG_SWA_CHECKPOINTS) |
| `--no-context-shift` | disables context shift on infinite text generation (default: enabled)<br/>(env: LLAMA_ARG_NO_CONTEXT_SHIFT) |
| `--context-shift` | enables context shift on infinite text generation (default: disabled)<br/>(env: LLAMA_ARG_CONTEXT_SHIFT) |
diff --git a/tools/server/bench/README.md b/tools/server/bench/README.md
index 9549795e..a5766f77 100644
--- a/tools/server/bench/README.md
+++ b/tools/server/bench/README.md
@@ -95,6 +95,9 @@ The `bench.py` script does several steps:
- run k6 script
- extract metrics from prometheus
+Pass `--cal-llm-profile` to `bench.py` to start `llama-server` with cal-llm backend profiling enabled.
+Pass `--cal-llm-profile-dump FILE` to write cal-llm backend profiling JSONL.
+
It aims to be used in the CI, but you can run it manually:
```shell
diff --git a/tools/server/bench/bench.py b/tools/server/bench/bench.py
index 0c57a2df..eb41839a 100644
--- a/tools/server/bench/bench.py
+++ b/tools/server/bench/bench.py
@@ -47,6 +47,10 @@ def main(args_in: list[str] | None = None) -> None:
parser.add_argument("--ubatch-size", type=int, help="physical maximum batch size", required=True)
parser.add_argument("--scenario", type=str, help="Scenario to run", required=True)
parser.add_argument("--duration", type=str, help="Bench scenario", required=True)
+ parser.add_argument("--cal-llm-profile", action="store_true",
+ help="Enable cal-llm backend profiling logs when starting llama-server")
+ parser.add_argument("--cal-llm-profile-dump", type=str,
+ help="Write cal-llm backend profiling JSONL to the given file")
args = parser.parse_args(args_in)
@@ -277,6 +281,10 @@ def start_server_background(args):
server_args.append('--cont-batching')
server_args.append('--metrics')
server_args.append('--flash-attn')
+ if args.cal_llm_profile:
+ server_args.append('--cal-llm-profile')
+ if args.cal_llm_profile_dump:
+ server_args.extend(['--cal-llm-profile-dump', args.cal_llm_profile_dump])
args = [str(arg) for arg in [server_path, *server_args]]
print(f"bench: starting server with: {' '.join(args)}")
pkwargs = {
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index 6a2a7433..5ffc5b53 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -11,9 +11,12 @@
#include "sampling.h"
#include "speculative.h"
#include "mtmd.h"
-#include "src/llama-calrt.h"
#include <cstdint>
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+#include "c_api.h"
+#endif
+
// mime type for sending response
#define MIMETYPE_JSON "application/json; charset=utf-8"
@@ -27,6 +30,7 @@
#include <cstddef>
#include <cinttypes>
#include <deque>
+#include <limits>
#include <memory>
#include <mutex>
#include <signal.h>
@@ -2398,6 +2402,220 @@ struct server_response {
}
};
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+struct server_cal_llm_generated_token {
+ std::string request_id;
+ llama_token token_id = LLAMA_TOKEN_NULL;
+ bool finished = false;
+ bool has_finish_reason = false;
+ int32_t finish_reason = -1;
+};
+
+struct server_cal_llm_step_result {
+ int32_t status = CALRT_STATUS_INTERNAL_ERROR;
+ std::vector<server_cal_llm_generated_token> generated_tokens;
+};
+
+static const char * server_cal_llm_status_name(int32_t status) {
+ switch (status) {
+ case CALRT_STATUS_OK: return "OK";
+ case CALRT_STATUS_ACCEPTED: return "ACCEPTED";
+ case CALRT_STATUS_INVALID_REQUEST: return "INVALID_REQUEST";
+ case CALRT_STATUS_DUPLICATE_REQUEST:return "DUPLICATE_REQUEST";
+ case CALRT_STATUS_MISSING_REQUEST: return "MISSING_REQUEST";
+ case CALRT_STATUS_NO_READY_WORK: return "NO_READY_WORK";
+ case CALRT_STATUS_CAPACITY_BLOCKED: return "CAPACITY_BLOCKED";
+ case CALRT_STATUS_INTERNAL_ERROR: return "INTERNAL_ERROR";
+ case CALRT_STATUS_EXECUTION_FAILED: return "EXECUTION_FAILED";
+ default: return "UNKNOWN";
+ }
+}
+
+static bool server_cal_llm_is_fatal_step_status(int32_t status) {
+ switch (status) {
+ case CALRT_STATUS_INVALID_REQUEST:
+ case CALRT_STATUS_DUPLICATE_REQUEST:
+ case CALRT_STATUS_MISSING_REQUEST:
+ case CALRT_STATUS_INTERNAL_ERROR:
+ case CALRT_STATUS_EXECUTION_FAILED:
+ return true;
+ default:
+ return false;
+ }
+}
+
+class server_cal_llm_backend {
+public:
+ ~server_cal_llm_backend() {
+ calrt_engine_destroy(engine);
+ }
+
+ bool init(const common_params & params) {
+ CalrtEngineConfig config = {};
+ config.first_sequence_id = 1;
+ config.calbin_path = params.calbin_path.empty() ? nullptr : params.calbin_path.c_str();
+ const bool cal_llm_profile_enabled = params.cal_llm_profile || !params.cal_llm_profile_dump.empty();
+ config.profiling_enabled = cal_llm_profile_enabled ? 1 : 0;
+ config.profiling_dump_path = nullptr;
+ if (!params.cal_llm_profile_dump.empty()) {
+ config.profiling_dump_path = params.cal_llm_profile_dump.c_str();
+ }
+
+ engine = calrt_engine_create(&config);
+ return engine != nullptr;
+ }
+
+ bool add_request(
+ const server_task & task,
+ const llama_vocab * vocab,
+ std::string & error) {
+ if (engine == nullptr) {
+ error = "cal-llm EngineCore is not initialized";
+ return false;
+ }
+
+ const llama_tokens & text_tokens = task.tokens.get_text_tokens();
+ if (text_tokens.empty()) {
+ error = "cal-llm requires a non-empty prompt";
+ return false;
+ }
+
+ std::vector<uint32_t> input_tokens;
+ input_tokens.reserve(text_tokens.size());
+ for (llama_token token : text_tokens) {
+ if (token < 0) {
+ error = "cal-llm text generation does not support non-text prompt chunks";
+ return false;
+ }
+ input_tokens.push_back(static_cast<uint32_t>(token));
+ }
+
+ uint32_t max_generated_tokens = 0;
+ if (task.params.n_predict > 0) {
+ max_generated_tokens = static_cast<uint32_t>(task.params.n_predict);
+ } else if (task.params.n_predict == 0) {
+ error = "cal-llm server backend does not support prompt-only evaluation";
+ return false;
+ }
+
+ std::vector<uint32_t> stop_token_ids;
+ if (!task.params.sampling.ignore_eos) {
+ const llama_token eos_token = llama_vocab_eos(vocab);
+ if (eos_token != LLAMA_TOKEN_NULL && eos_token >= 0) {
+ stop_token_ids.push_back(static_cast<uint32_t>(eos_token));
+ }
+ }
+
+ CalrtSamplingConfig sampling_config = {};
+ sampling_config.temperature = task.params.sampling.temp;
+ sampling_config.top_k = task.params.sampling.top_k > 0
+ ? static_cast<uint32_t>(task.params.sampling.top_k)
+ : 0;
+ sampling_config.top_p = task.params.sampling.top_p;
+ sampling_config.seed = task.params.sampling.seed;
+ sampling_config.max_generated_tokens = max_generated_tokens;
+ sampling_config.min_generated_tokens = 0;
+ sampling_config.stop_token_ids = stop_token_ids.empty() ? nullptr : stop_token_ids.data();
+ sampling_config.stop_token_id_count = static_cast<uint32_t>(stop_token_ids.size());
+
+ const std::string request_id = request_id_from_task_id(task.id);
+ CalrtRequest request = {};
+ request.request_id = request_id.c_str();
+ request.input_tokens = input_tokens.data();
+ request.input_token_count = static_cast<uint32_t>(input_tokens.size());
+ request.sampling_config = sampling_config;
+
+ const int32_t status = calrt_engine_add_request(engine, &request);
+ if (status != CALRT_STATUS_ACCEPTED && status != CALRT_STATUS_OK) {
+ error = std::string("cal-llm failed to add request: ") + server_cal_llm_status_name(status);
+ return false;
+ }
+
+ return true;
+ }
+
+ int32_t finish_request(int task_id) {
+ if (engine == nullptr) {
+ return CALRT_STATUS_INTERNAL_ERROR;
+ }
+ const std::string request_id = request_id_from_task_id(task_id);
+ return calrt_engine_finish_request(engine, request_id.c_str());
+ }
+
+ int32_t abort_request(int task_id) {
+ if (engine == nullptr) {
+ return CALRT_STATUS_INTERNAL_ERROR;
+ }
+ const std::string request_id = request_id_from_task_id(task_id);
+ return calrt_engine_abort_request(engine, request_id.c_str());
+ }
+
+ server_cal_llm_step_result step() {
+ server_cal_llm_step_result result;
+ if (engine == nullptr) {
+ result.status = CALRT_STATUS_INTERNAL_ERROR;
+ return result;
+ }
+
+ CalrtStepOutput * output = calrt_engine_step(engine);
+ if (output == nullptr) {
+ result.status = CALRT_STATUS_INTERNAL_ERROR;
+ return result;
+ }
+
+ result.status = output->status;
+ result.generated_tokens.reserve(output->generated_token_count);
+ for (size_t token_index = 0; token_index < output->generated_token_count; ++token_index) {
+ const CalrtGeneratedToken & c_token = output->generated_tokens[token_index];
+ server_cal_llm_generated_token token;
+ token.request_id = c_token.request_id == nullptr ? "" : c_token.request_id;
+ token.token_id = static_cast<llama_token>(c_token.token_id);
+ token.finished = c_token.finished != 0;
+ token.has_finish_reason = c_token.has_finish_reason != 0;
+ token.finish_reason = c_token.finish_reason;
+ result.generated_tokens.push_back(std::move(token));
+ }
+
+ calrt_engine_free_step_output(output);
+ return result;
+ }
+
+ static std::string request_id_from_task_id(int task_id) {
+ return std::to_string(task_id);
+ }
+
+private:
+ CalrtEngine * engine = nullptr;
+};
+
+static common_init_result common_init_vocab_only_from_params(common_params & params) {
+ common_init_result init_result;
+
+ auto mparams = common_model_params_to_llama(params);
+ mparams.vocab_only = true;
+
+ llama_model * model = llama_model_load_from_file(params.model.path.c_str(), mparams);
+ if (model == nullptr) {
+ SRV_ERR("failed to load vocab-only model, '%s'\n", params.model.path.c_str());
+ return init_result;
+ }
+
+ auto cparams = common_context_params_to_llama(params);
+
+ llama_context * ctx = llama_init_from_model(model, cparams);
+ if (ctx == nullptr) {
+ SRV_ERR("failed to create vocab-only context with model, '%s'\n", params.model.path.c_str());
+ llama_model_free(model);
+ return init_result;
+ }
+
+ init_result.model.reset(model);
+ init_result.context.reset(ctx);
+
+ return init_result;
+}
+#endif
+
struct server_context {
common_params params_base;
@@ -2406,9 +2624,6 @@ struct server_context {
common_init_result llama_init_dft;
llama_model * model = nullptr;
-#ifdef USE_CALRT
- calrt_context * cal_ctx = nullptr;
-#endif
llama_context * ctx = nullptr;
// multimodal
@@ -2440,6 +2655,10 @@ struct server_context {
server_metrics metrics;
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ server_cal_llm_backend cal_llm;
+#endif
+
// Necessary similarity of prompt for slot selection
float slot_prompt_similarity = 0.0f;
@@ -2471,13 +2690,31 @@ struct server_context {
params_base = params;
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ if (!params_base.speculative.model.path.empty() || !params_base.speculative.model.hf_repo.empty()) {
+ SRV_ERR("%s\n", "cal-llm server backend does not support speculative decoding");
+ return false;
+ }
+ if (!params_base.mmproj.path.empty() || !params_base.mmproj.hf_repo.empty()) {
+ SRV_ERR("%s\n", "cal-llm server backend does not support multimodal models");
+ return false;
+ }
+ if (!params_base.lora_adapters.empty()) {
+ SRV_ERR("%s\n", "cal-llm server backend does not support startup LoRA adapters");
+ return false;
+ }
+ if (!params_base.control_vectors.empty()) {
+ SRV_ERR("%s\n", "cal-llm server backend does not support control vectors");
+ return false;
+ }
+
+ llama_init = common_init_vocab_only_from_params(params_base);
+#else
llama_init = common_init_from_params(params_base);
+#endif
model = llama_init.model.get();
ctx = llama_init.context.get();
-#ifdef USE_CALRT
- cal_ctx = llama_init.cal_ctx.get();
-#endif
if (model == nullptr) {
SRV_ERR("failed to load model, '%s'\n", params_base.model.path.c_str());
@@ -2486,11 +2723,7 @@ struct server_context {
vocab = llama_model_get_vocab(model);
-#ifdef USE_CALRT
- n_ctx = calrt_n_ctx(cal_ctx);
-#else
n_ctx = llama_n_ctx(ctx);
-#endif
add_bos_token = llama_vocab_get_add_bos(vocab);
@@ -2554,9 +2787,6 @@ struct server_context {
mparams.image_min_tokens = params_base.image_min_tokens;
mparams.image_max_tokens = params_base.image_max_tokens;
mctx = mtmd_init_from_file(mmproj_path.c_str(), model, mparams);
-#ifdef USE_CALRT
- mtmd_set_calrt_ctx(mctx, cal_ctx);
-#endif
if (mctx == nullptr) {
SRV_ERR("failed to load multimodal model, '%s'\n", mmproj_path.c_str());
return false;
@@ -2594,9 +2824,13 @@ struct server_context {
return true;
}
- void init() {
+ bool init() {
SRV_INF("initializing slots, n_slots = %d\n", params_base.n_parallel);
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ int n_ctx_slot = std::numeric_limits<int>::max();
+ n_ctx = 0;
+#else
const int n_ctx_train = llama_model_n_ctx_train(model);
int n_ctx_slot = llama_n_ctx_seq(ctx);
@@ -2604,6 +2838,7 @@ struct server_context {
SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", n_ctx_slot, n_ctx_train);
n_ctx_slot = n_ctx_train;
}
+#endif
for (int i = 0; i < params_base.n_parallel; i++) {
server_slot slot;
@@ -2620,13 +2855,13 @@ struct server_context {
slot.ctx_dft = llama_init_from_model(model_dft, cparams_dft);
if (slot.ctx_dft == nullptr) {
SRV_ERR("%s", "failed to create draft context\n");
- return;
+ return false;
}
slot.spec = common_speculative_init(slot.ctx, slot.ctx_dft);
if (slot.spec == nullptr) {
SRV_ERR("%s", "failed to create speculator\n");
- return;
+ return false;
}
for (auto & pair : params_base.speculative.replacements) {
common_speculative_add_replacement_tgt_dft(slot.spec, pair.first.c_str(), pair.second.c_str());
@@ -2656,10 +2891,21 @@ struct server_context {
// the update_slots() logic will always submit a maximum of n_batch or n_parallel tokens
// note that n_batch can be > n_ctx (e.g. for non-causal attention models such as BERT where the KV cache is not used)
{
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ const int32_t n_batch = std::max<int32_t>(1, params_base.n_parallel);
+#else
const int32_t n_batch = llama_n_batch(ctx);
+#endif
batch = llama_batch_init(std::max(n_batch, params_base.n_parallel), 0, 1);
}
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ if (!cal_llm.init(params_base)) {
+ SRV_ERR("%s\n", "failed to initialize cal-llm EngineCore");
+ return false;
+ }
+#endif
+
metrics.init();
if (params_base.cache_ram_mib != 0) {
@@ -2692,6 +2938,8 @@ struct server_context {
/* allow_audio */ mctx ? mtmd_support_audio (mctx) : false,
/* enable_thinking */ enable_thinking,
};
+
+ return true;
}
server_slot * get_slot_by_id(int id) {
@@ -2838,7 +3086,88 @@ struct server_context {
return res;
}
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ bool validate_cal_llm_task(const server_task & task, std::string & error) const {
+ if (task.type != SERVER_TASK_TYPE_COMPLETION) {
+ error = "cal-llm server backend supports text completion requests only";
+ return false;
+ }
+
+ if (mctx != nullptr || task.tokens.has_mtmd) {
+ error = "cal-llm server backend does not support multimodal requests";
+ return false;
+ }
+
+ if (!task.params.lora.empty()) {
+ error = "cal-llm server backend does not support LoRA requests";
+ return false;
+ }
+
+ if (!task.params.sampling.grammar.empty() || task.params.sampling.grammar_lazy ||
+ !task.params.sampling.grammar_triggers.empty()) {
+ error = "cal-llm server backend does not support grammar or JSON schema constraints";
+ return false;
+ }
+
+ if (!task.params.sampling.logit_bias.empty()) {
+ error = "cal-llm server backend does not support logit_bias";
+ return false;
+ }
+
+ if (task.params.sampling.n_probs > 0 || task.params.post_sampling_probs) {
+ error = "cal-llm server backend does not support logprobs";
+ return false;
+ }
+
+ return true;
+ }
+
+ bool launch_slot_with_task_cal_llm(server_slot & slot, server_task && task) {
+ slot.reset();
+
+ std::string error;
+ if (!validate_cal_llm_task(task, error)) {
+ send_error(task, error, ERROR_TYPE_NOT_SUPPORTED);
+ return false;
+ }
+
+ if (!task.tokens.validate(ctx)) {
+ send_error(task, "Prompt contains invalid tokens", ERROR_TYPE_INVALID_REQUEST);
+ return false;
+ }
+
+ if (!cal_llm.add_request(task, vocab, error)) {
+ send_error(task, error, ERROR_TYPE_INVALID_REQUEST);
+ return false;
+ }
+
+ slot.prompt.tokens.clear();
+ slot.prompt.tokens.insert(task.tokens.get_text_tokens());
+ slot.n_prompt_tokens_cache = 0;
+ slot.n_prompt_tokens_processed = task.n_tokens();
+ slot.n_decoded = 0;
+ slot.n_remaining = task.params.n_predict;
+ slot.t_start_process_prompt = ggml_time_us();
+ slot.t_start_generation = 0;
+
+ if (slot.smpl != nullptr) {
+ common_sampler_free(slot.smpl);
+ slot.smpl = nullptr;
+ }
+
+ slot.task = std::make_unique<server_task>(std::move(task));
+ slot.state = SLOT_STATE_GENERATING;
+
+ SLT_INF(slot, "%s", "processing task with cal-llm\n");
+ return true;
+ }
+#endif
+
bool launch_slot_with_task(server_slot & slot, server_task && task) {
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ return launch_slot_with_task_cal_llm(slot, std::move(task));
+#else
+
slot.reset();
if (!are_lora_equal(task.params.lora, slot.lora)) {
@@ -2935,6 +3264,7 @@ struct server_context {
SLT_INF(slot, "%s", "processing task\n");
return true;
+#endif
}
void kv_cache_clear() {
@@ -3071,14 +3401,9 @@ struct server_context {
SLT_DBG(slot, "%s", "stopped by EOS\n");
}
-// Limit generated tokens to max_seq_len
-#ifdef USE_CALRT
- const auto n_ctx_per_seq = calrt_n_ctx_per_seq(cal_ctx);
- //const auto n_ctx_per_seq = 4096;
-#else
- const auto n_ctx_per_seq = 4096;
-#endif
-
+ // Limit generated tokens to max_seq_len.
+ const auto n_ctx_per_seq = slot.n_ctx;
+
if (slot.task->params.n_predict < 1 && slot.n_prompt_tokens_processed + slot.n_prompt_tokens_cache + slot.n_decoded >= static_cast<int32_t>(n_ctx_per_seq)) {
slot.truncated = true;
slot.stop = STOP_TYPE_LIMIT;
@@ -3486,6 +3811,13 @@ struct server_context {
// release slot linked with the task id
for (auto & slot : slots) {
if (slot.task && slot.task->id == task.id_target) {
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ const int32_t status = cal_llm.abort_request(task.id_target);
+ if (status != CALRT_STATUS_OK && status != CALRT_STATUS_MISSING_REQUEST) {
+ SRV_WRN("cal-llm failed to abort request %d: %s\n",
+ task.id_target, server_cal_llm_status_name(status));
+ }
+#endif
slot.release();
break;
}
@@ -3545,6 +3877,10 @@ struct server_context {
} break;
case SERVER_TASK_TYPE_SLOT_SAVE:
{
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ send_error(task, "cal-llm server backend does not support slots save", ERROR_TYPE_NOT_SUPPORTED);
+ break;
+#endif
if (!check_no_mtmd(task.id)) {
break;
}
@@ -3586,6 +3922,10 @@ struct server_context {
} break;
case SERVER_TASK_TYPE_SLOT_RESTORE:
{
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ send_error(task, "cal-llm server backend does not support slots restore", ERROR_TYPE_NOT_SUPPORTED);
+ break;
+#endif
if (!check_no_mtmd(task.id)) break;
int id_slot = task.slot_action.slot_id;
server_slot * slot = get_slot_by_id(id_slot);
@@ -3633,6 +3973,10 @@ struct server_context {
} break;
case SERVER_TASK_TYPE_SLOT_ERASE:
{
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ send_error(task, "cal-llm server backend does not support slots erase", ERROR_TYPE_NOT_SUPPORTED);
+ break;
+#endif
if (!check_no_mtmd(task.id)) {
break;
}
@@ -3662,6 +4006,10 @@ struct server_context {
} break;
case SERVER_TASK_TYPE_SET_LORA:
{
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ send_error(task, "cal-llm server backend does not support LoRA", ERROR_TYPE_NOT_SUPPORTED);
+ break;
+#endif
params_base.lora_adapters = std::move(task.set_lora);
auto res = std::make_unique<server_task_result_apply_lora>();
res->id = task.id;
@@ -3671,8 +4019,161 @@ struct server_context {
}
}
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ server_slot * get_slot_by_cal_llm_request_id(const std::string & request_id) {
+ for (server_slot & slot : slots) {
+ if (!slot.task) {
+ continue;
+ }
+ if (server_cal_llm_backend::request_id_from_task_id(slot.task->id) == request_id) {
+ return &slot;
+ }
+ }
+ return nullptr;
+ }
+
+ void finish_cal_llm_slot(server_slot & slot) {
+ const int32_t status = cal_llm.finish_request(slot.task->id);
+ if (status != CALRT_STATUS_OK && status != CALRT_STATUS_MISSING_REQUEST) {
+ SLT_WRN(slot, "cal-llm failed to finish request: %s\n", server_cal_llm_status_name(status));
+ }
+
+ if (params_base.device_info) {
+ slot.device_info = true;
+ }
+ slot.print_timings();
+ send_final_response(slot);
+ metrics.on_prediction(slot);
+ slot.release();
+ }
+
+ void apply_cal_llm_finish_reason(server_slot & slot, const server_cal_llm_generated_token & token) {
+ if (!token.finished || slot.stop != STOP_TYPE_NONE) {
+ return;
+ }
+
+ if (!token.has_finish_reason) {
+ slot.stop = STOP_TYPE_EOS;
+ slot.has_next_token = false;
+ return;
+ }
+
+ switch (token.finish_reason) {
+ case CALRT_FINISH_REASON_EOS_TOKEN:
+ case CALRT_FINISH_REASON_STOP_STRING:
+ case CALRT_FINISH_REASON_EXPLICIT_FINISH:
+ slot.stop = STOP_TYPE_EOS;
+ slot.has_next_token = false;
+ break;
+ case CALRT_FINISH_REASON_MAX_GENERATED_TOKENS:
+ case CALRT_FINISH_REASON_MAX_SEQUENCE_LENGTH:
+ slot.stop = STOP_TYPE_LIMIT;
+ slot.truncated = token.finish_reason == CALRT_FINISH_REASON_MAX_SEQUENCE_LENGTH;
+ slot.has_next_token = false;
+ break;
+ case CALRT_FINISH_REASON_EXPLICIT_ABORT:
+ slot.stop = STOP_TYPE_LIMIT;
+ slot.has_next_token = false;
+ break;
+ default:
+ slot.stop = STOP_TYPE_LIMIT;
+ slot.has_next_token = false;
+ break;
+ }
+ }
+
+ void fail_active_cal_llm_slots(const std::string & error) {
+ for (server_slot & slot : slots) {
+ if (!slot.is_processing()) {
+ continue;
+ }
+
+ send_error(slot, error);
+ cal_llm.abort_request(slot.task->id);
+ slot.release();
+ }
+ }
+
+ void update_slots_cal_llm() {
+ bool all_idle = true;
+ for (server_slot & slot : slots) {
+ if (slot.is_processing()) {
+ all_idle = false;
+ break;
+ }
+ }
+
+ if (all_idle) {
+ SRV_INF("%s", "all slots are idle\n");
+ return;
+ }
+
+ {
+ SRV_DBG("%s", "posting NEXT_RESPONSE\n");
+
+ server_task task(SERVER_TASK_TYPE_NEXT_RESPONSE);
+ task.id = queue_tasks.get_new_id();
+ queue_tasks.post(std::move(task));
+ }
+
+ server_cal_llm_step_result step_result = cal_llm.step();
+ if (step_result.status == CALRT_STATUS_NO_READY_WORK ||
+ step_result.status == CALRT_STATUS_CAPACITY_BLOCKED) {
+ return;
+ }
+
+ if (server_cal_llm_is_fatal_step_status(step_result.status) ||
+ (step_result.status != CALRT_STATUS_OK && step_result.status != CALRT_STATUS_ACCEPTED)) {
+ fail_active_cal_llm_slots(
+ std::string("cal-llm step failed: ") + server_cal_llm_status_name(step_result.status));
+ return;
+ }
+
+ auto accept_special_token = [&](server_slot & slot, llama_token token) {
+ return params_base.special ||
+ slot.task->params.sampling.preserved_tokens.find(token) != slot.task->params.sampling.preserved_tokens.end();
+ };
+
+ for (const server_cal_llm_generated_token & token : step_result.generated_tokens) {
+ server_slot * slot = get_slot_by_cal_llm_request_id(token.request_id);
+ if (slot == nullptr || !slot->is_processing()) {
+ SRV_WRN("cal-llm returned token for unknown request '%s'\n", token.request_id.c_str());
+ continue;
+ }
+
+ const int64_t t_current = ggml_time_us();
+ slot->n_decoded += 1;
+ if (slot->n_decoded == 1) {
+ slot->t_start_generation = t_current;
+ slot->t_prompt_processing = (slot->t_start_generation - slot->t_start_process_prompt) / 1e3;
+ metrics.on_prompt_eval(*slot);
+ }
+ slot->t_token_generation = std::max<int64_t>(1, t_current - slot->t_start_generation) / 1e3;
+
+ completion_token_output result;
+ result.tok = token.token_id;
+ result.text_to_send = common_token_to_piece(ctx, result.tok, accept_special_token(*slot, result.tok));
+ result.prob = 1.0f;
+
+ const bool continue_generating = process_token(result, *slot);
+ apply_cal_llm_finish_reason(*slot, token);
+
+ if (!continue_generating || token.finished || !slot->has_next_token) {
+ finish_cal_llm_slot(*slot);
+ continue;
+ }
+
+ slot->prompt.tokens.push_back(result.tok);
+ }
+ }
+#endif
+
uint8_t is_end = 0;
void update_slots() {
+#if defined(LLAMA_SERVER_USE_CAL_LLM)
+ update_slots_cal_llm();
+#else
+
const int64_t t_prep_start = ggml_time_us();
// check if all slots are idle
{
@@ -4100,11 +4601,7 @@ struct server_context {
if (slot.prompt.n_tokens() < slot.task->n_tokens() && input_tokens[slot.prompt.n_tokens()] == LLAMA_TOKEN_NULL) {
// process the image
size_t n_tokens_out = 0;
-#ifdef USE_CALRT
- int32_t res = input_tokens.process_chunk(cal_ctx, mctx, slot.prompt.n_tokens(), slot.prompt.tokens.pos_next(),slot.id, n_tokens_out);
-#else
int32_t res = input_tokens.process_chunk(ctx, mctx, slot.prompt.n_tokens(), slot.prompt.tokens.pos_next(), slot.id, n_tokens_out);
-#endif
if (res != 0) {
SLT_ERR(slot, "failed to process image, res = %d\n", res);
send_error(slot, "failed to process image", ERROR_TYPE_SERVER);
@@ -4299,14 +4796,9 @@ struct server_context {
}
const int64_t t_dec_start = ggml_time_us();
-#ifdef USE_CALRT
- const int ret = calrt_decode(cal_ctx, batch_view);
-#else
const int ret = llama_decode(ctx, batch_view);
-#endif
const int64_t t_dec_end = ggml_time_us();
const double t_dec_ms = (t_dec_end - t_dec_start) / 1e3;
- //printf("t_dec_ms: %10.2f\n", t_dec_ms);
if (ret == 0) {
for (auto & slot : slots) {
@@ -4314,19 +4806,7 @@ struct server_context {
// slot.i_batch 记录了该 slot 在全局 batch 中的索引
// 当前 batch_view 处理的范围是 [i, i + n_tokens)
if (slot.i_batch >= i && slot.i_batch < (i + n_tokens)) {
- //slot.t_inference_accumulated += t_dec_ms;
- slot.t_inference_accumulated += cal_ctx->get_infer_time(slot.id);//infer time
- slot.t_inference_incopy += cal_ctx->get_incopy_time(slot.id);//copy in time
- slot.t_inference_outcopy += cal_ctx->get_outcopy_time(slot.id);//copy out time
-
- //prefill infer
- slot.t_prefill_inference_accumulated = cal_ctx->get_prefillinfer_time(slot.id);//infer time
- slot.t_prefill_inference_incopy = cal_ctx->get_prefillincopy_time(slot.id);//copy in time
- slot.t_prefill_inference_outcopy = cal_ctx->get_prefilloutcopy_time(slot.id);//copy out time
-
- slot.t_unknown += cal_ctx->get_unknown_time(slot.id);
- slot.t_unknown_pre = cal_ctx->get_unknown_pre_time(slot.id);
-
+ slot.t_inference_accumulated += t_dec_ms;
}
}
}
@@ -4594,6 +5074,7 @@ struct server_context {
}
SRV_DBG("%s", "run slots completed\n");
+#endif
}
json model_meta() const {
@@ -5207,9 +5688,10 @@ int run_server(int argc, char ** argv) {
// Everything else, including multimodal completions.
inputs = tokenize_input_prompts(ctx_server.vocab, ctx_server.mctx, prompt, true, true);
}
- const size_t n_ctx_slot = ctx_server.slots.front().n_ctx;
tasks.reserve(inputs.size());
for (size_t i = 0; i < inputs.size(); i++) {
+#if !defined(LLAMA_SERVER_USE_CAL_LLM)
+ const size_t n_ctx_slot = ctx_server.slots.front().n_ctx;
auto n_prompt_tokens = inputs[i].size();
if (n_prompt_tokens >= n_ctx_slot) {
json error_data = format_error_response("the request exceeds the available context size, try increasing it", ERROR_TYPE_EXCEED_CONTEXT_SIZE);
@@ -5218,6 +5700,7 @@ int run_server(int argc, char ** argv) {
res_error(res, error_data);
return;
}
+#endif
server_task task = server_task(type);
task.id = ctx_server.queue_tasks.get_new_id();
@@ -5912,7 +6395,12 @@ int run_server(int argc, char ** argv) {
return 1;
}
- ctx_server.init();
+ if (!ctx_server.init()) {
+ clean_up();
+ t.join();
+ LOG_ERR("%s: exiting due to server initialization error\n", __func__);
+ return 1;
+ }
state.store(SERVER_STATE_READY);
LOG_INF("%s: model loaded\n", __func__);
@@ -5961,4 +6449,4 @@ int run_server(int argc, char ** argv) {
llama_memory_breakdown_print(ctx_server.ctx);
return 0;
-}
\ No newline at end of file
+}
diff --git a/tools/server/utils.hpp b/tools/server/utils.hpp
index ea688d33..958fc128 100644
--- a/tools/server/utils.hpp
+++ b/tools/server/utils.hpp
@@ -1366,38 +1366,6 @@ public:
}
// encode and decode the image chunk
-#ifdef USE_CALRT
- int32_t process_chunk(
- calrt_context * ctx,
- mtmd_context * mctx,
- size_t idx,
- llama_pos pos,
- int32_t seq_id,
- size_t & n_tokens_out) const {
- const auto & chunk = find_chunk(idx);
- const char * name = mtmd_input_chunk_get_type(chunk.get()) == MTMD_INPUT_CHUNK_TYPE_IMAGE
- ? "image" : "audio";
- SRV_INF("processing %s...\n", name);
- int32_t n_batch = calrt_n_batch(ctx);
- int64_t t0 = ggml_time_ms();
- llama_pos new_n_past; // unused for now
- int32_t result = mtmd_helper_eval_chunk_single(mctx, ctx,
- chunk.get(),
- pos,
- seq_id,
- n_batch,
- true, // logits last
- &new_n_past);
- SRV_INF("%s processed in %" PRId64 " ms\n", name, ggml_time_ms() - t0);
- if (result != 0) {
- LOG_ERR("mtmd_helper_eval failed with status %d", result);
- n_tokens_out = 0;
- return result;
- }
- n_tokens_out = mtmd_input_chunk_get_n_tokens(chunk.get());
- return 0;
- }
-#else
int32_t process_chunk(
llama_context * ctx,
mtmd_context * mctx,
@@ -1428,7 +1396,6 @@ public:
n_tokens_out = mtmd_input_chunk_get_n_tokens(chunk.get());
return 0;
}
-#endif
};
// Computes FNV-1a hash of the data