diff --git a/src/CMakeLists.txt b/src/CMakeLists.txt index 48910cc0..8b27c2ed 100644 --- a/src/CMakeLists.txt +++ b/src/CMakeLists.txt @@ -500,6 +500,7 @@ list(APPEND FLM_ENGINE_LINK_LIBS qwen3vl_flash) list(APPEND FLM_ENGINE_LINK_LIBS qwen3_5vl_npu + minicpm_v_4_7_npu qwen3_5_omni_npu qwen3_6_moe_npu qwen3_8mtp_npu diff --git a/src/CMakePresets.json b/src/CMakePresets.json index a35c897b..3ba154a8 100644 --- a/src/CMakePresets.json +++ b/src/CMakePresets.json @@ -5,7 +5,7 @@ "name": "common-default", "hidden": true, "cacheVariables": { - "FLM_VERSION": "1.0.7", + "FLM_VERSION": "1.0.8", "NPU_VERSION": "32.0.203.304" } }, diff --git a/src/common/AutoModel/automodel.cpp b/src/common/AutoModel/automodel.cpp index 404433d3..765b170f 100644 --- a/src/common/AutoModel/automodel.cpp +++ b/src/common/AutoModel/automodel.cpp @@ -282,7 +282,8 @@ void AutoModel::_shared_load_backend(std::string model_path, json model_info, this->is_model_loaded = false; throw; } - header_print("FLM", "Backend: " << id << " (from " << source << ")"); + // Hide the log now. + // header_print("FLM", "Backend: " << id << " (from " << source << ")"); } void AutoModel::_shared_after_inference_failure(bool poisoned) { diff --git a/src/common/AutoModel/builtin_backends.cpp b/src/common/AutoModel/builtin_backends.cpp index c25d32f5..6f2b7708 100644 --- a/src/common/AutoModel/builtin_backends.cpp +++ b/src/common/AutoModel/builtin_backends.cpp @@ -42,6 +42,7 @@ void register_builtin_backends(BackendRegistry& registry) { RegisterFlm(registry, "qwen3vl"); RegisterFlm(registry, "qwen3vl-flash"); RegisterFlm(registry, "qwen3.5"); + RegisterFlm(registry, "minicpm-v-4.7"); RegisterFlm(registry, "qwen3.6-moe"); RegisterFlm(registry, "qwen3.8-mtp"); RegisterFlm(registry, "gemma3"); diff --git a/src/common/AutoModel/modeling_minicpm_v.cpp b/src/common/AutoModel/modeling_minicpm_v.cpp new file mode 100644 index 00000000..f1edf215 --- /dev/null +++ b/src/common/AutoModel/modeling_minicpm_v.cpp @@ -0,0 +1,757 @@ +/// \file modeling_minicpm_v.cpp +/// \brief MiniCPM_V class +/// \author FastFlowLM Team +/// \date 2026-10-02 +/// \version 0.9.28 +/// \note This is a source file for the MiniCPM_V class + + +#include "AutoModel/modeling_minicpm_v.hpp" + +#include +#include + + +/************ MiniCPM_V family **************/ +MiniCPM_V::MiniCPM_V(flm_rt::device* npu_device_inst) : AutoModel(npu_device_inst, "MiniCPM_V") {} + +void MiniCPM_V::load_model(std::string model_path, json model_info, int default_context_length, bool enable_preemption, const std::string& backend) { + this->_shared_load_backend(model_path, model_info, default_context_length, enable_preemption, backend); + this->setup_tokenizer(model_path); + this->sampler.reset(); + + // The shipped template renders tools in the same XML format as Qwen3.5. + this->enable_tool = true; + + // MiniCPM-V ships no sampling defaults of its own; its text tower is + // Qwen3.5-0.8B's, so start from the Qwen3.5 non-thinking recommendation. + sampler_config config; + config.top_k = 20; + config.top_p = 0.8; + config.min_p = 0.0; + config.temperature = 0.7; + config.rep_penalty = 1.0; + config.freq_penalty = 1.0; + config.pre_penalty = 1.5f; + + this->set_sampler(config); + for (size_t i = 0; i < PROFILER_TYPE_NUM; i++) { + this->profiler_list[i].reset(); + } +} + +void MiniCPM_V::setup_tokenizer(std::string model_path) { + auto tokenizer_config = this->_shared_setup_tokenizer(model_path); +} + +std::string MiniCPM_V::apply_chat_template(nlohmann::ordered_json& messages, nlohmann::ordered_json tools) { + minja::chat_template_inputs inputs; + inputs.add_generation_prompt = true; + inputs.messages = messages; + inputs.extra_context = this->extra_context; + inputs.extra_context["enable_thinking"] = this->enable_think; + if (!tools.empty() && this->enable_tool) + inputs.tools = tools; + return this->chat_tmpl->apply(inputs); +} + +void MiniCPM_V::fail_inference() { + const bool poisoned = this->backend_ && this->backend_->poisoned(); + this->_shared_after_inference_failure(poisoned); + throw ModelRequestError(500, true, poisoned + ? "Inference failed; unload/reload is required because the model is poisoned" + : "Inference failed; the current conversation was cleared"); +} + +bool MiniCPM_V::insert(chat_meta_info_t& meta_info, lm_uniform_input_t& input, std::function is_cancelled) { + this->_shared_guard_poisoned(); + constexpr int image_pad = 248056; + // preprocess + this->profiler_list[TKOEN_ENCODE_TIME].start(); + std::string templated_text; + if (input.messages.empty() && input.prompt.empty()) { + header_print("WARNING", "No messages or prompt provided"); + return false; + } + + if (!input.audios.empty()) { + header_print("WARNING", "MiniCPM-V has no audio input, ignoring " << input.audios.size() << " audio input(s)"); + input.audios.clear(); + input.audio_payload_types.clear(); + } + + // is_vlm is set from config.json's vision_model_weight; without it the engine + // has no vision tower, so images are dropped here rather than failing there. + const bool vlm = this->lm_config && this->lm_config->get("is_vlm", false); + + // One layout per picture, in prompt order; the payload holds their VIEWS. + minicpm_v_image_payload_t image_payload; + image_payload.num_images = 0; + std::vector layouts; + auto add_image = [&](const std::string& source, bool base64) { + image_data_t chw; + if (!this->load_image(source, base64, chw)) { + header_print("ERROR", "Skipping image that failed to load"); + return false; + } + layouts.push_back(this->preprocess_image(chw, image_payload)); + image_reader_.recycle(chw); + return true; + }; + + if (!input.messages.empty()) { // already a formated messages, usually from REST API + nlohmann::ordered_json messages = nlohmann::ordered_json::array(); + for (auto& message : input.messages) { + message.erase("audios"); + if (!message.contains("images")) { + messages.push_back(message); + continue; + } + if (!vlm) { + header_print("WARNING", "this MiniCPM-V package has no vision tower, ignoring " << message["images"].size() << " image(s)"); + message.erase("images"); + messages.push_back(message); + continue; + } + // Decode/preprocess up front so an invalid image never gets a + // placeholder in the template below. + nlohmann::ordered_json content = nlohmann::ordered_json::array(); + for (const auto& img : message["images"]) { + if (add_image(img.get(), true)) { + content.push_back({ {"type", "image"} }); + } + } + content.push_back({ {"type", "text"}, {"text", message["content"]} }); + messages.push_back({ {"role", message["role"]}, {"content", content} }); + } + templated_text = this->apply_chat_template(messages, input.tools); + } + else if (!input.prompt.empty()) { // a pure text, usually from the cli + nlohmann::ordered_json messages; + nlohmann::ordered_json content = nlohmann::ordered_json::array(); + if (!input.images.empty() && !vlm) { + header_print("WARNING", "this MiniCPM-V package has no vision tower, ignoring " << input.images.size() << " image(s)"); + } + else { + for (const auto& path : input.images) { + if (add_image(path, false)) { + content.push_back({ {"type", "image"} }); + } + } + } + content.push_back({ {"type", "text"}, {"text", input.prompt} }); + messages.push_back({ {"role", "user"}, {"content", content} }); + templated_text = this->apply_chat_template(messages); + } + input.images.clear(); + input.image_payload_types.clear(); + + // Expand each picture's single <|image_pad|> to the processor's placeholder. + std::vector tokens_init = this->tokenizer->encode(templated_text); + std::vector tokens; + { + size_t extra = 0; + for (const auto& l : layouts) extra += (size_t)l.tokens + 16 + 3 * (size_t)l.views; + tokens.reserve(tokens_init.size() + extra); + size_t picture = 0; + for (int id : tokens_init) { + if (id != image_pad) { + tokens.push_back(id); + continue; + } + if (picture >= layouts.size()) { + header_print("ERROR", "the prompt has more <|image_pad|> than images"); + return false; + } + const std::vector ph = this->image_placeholder(layouts[picture], (int)picture); + tokens.insert(tokens.end(), ph.begin(), ph.end()); + picture++; + } + if (picture != layouts.size()) { + header_print("ERROR", "the prompt has " << picture << " <|image_pad|> for " << layouts.size() << " images"); + return false; + } + } + + this->profiler_list[TKOEN_ENCODE_TIME].stop(tokens.size()); + + if (meta_info.restore_allowed) { + const int restore_idx = this->lm_engine->restore(); + if (restore_idx >= 0) { + this->total_tokens = restore_idx; + this->token_history = checkpoint_his; // restore the token history to be consistent with the restored KV cache, which is crucial for correct functioning of _shared_insert's prefix-matching logic + } + } + + // The generation prompt ends in "\n" (2 tokens) or, with thinking + // off, "\n\n\n\n" (4 tokens). Leave them out of the prompt so + // the checkpoint lands right after "<|im_start|>assistant\n"; decode() + // feeds them back. + tokens.resize(tokens.size() - (this->enable_think ? 2 : 4)); + + // ---------------------------------------------------------------------- + // Prompt-cache aware image alignment. _shared_insert erases the prefix of + // `tokens` that matches token_history -- only when it matches ALL of + // token_history, else it clears the context and skips nothing. The payload + // must lose exactly the pictures that erased prefix fully covers (all of + // their views) so it lines up with the image rows that are prefilled. + // + // This is _shared_insert's rule on _shared_insert's inputs: token_history + // AFTER the restore above and `tokens` AFTER the think trim. Computing it + // against checkpoint_his instead (Qwen3_5VL's way) disagrees in the CLI, + // where token_history also holds the previous answer: a repeated question + // then "hits" here, its image is dropped, _shared_insert clears and + // re-prefills the whole prompt, and the engine is handed image rows with + // no pixels. + // ---------------------------------------------------------------------- + size_t prefix_skip_count = 0; + { + const size_t idx = this->token_history.size(); + while (prefix_skip_count < idx && prefix_skip_count < tokens.size() && + tokens[prefix_skip_count] == this->token_history[prefix_skip_count]) { + prefix_skip_count++; + } + if (prefix_skip_count != idx) prefix_skip_count = 0; + + if (prefix_skip_count > 0 && !layouts.empty()) { + int skipped_rows = 0; + for (size_t i = 0; i < prefix_skip_count; i++) skipped_rows += (tokens[i] == image_pad); + size_t drop_pictures = 0, drop_views = 0, drop_values = 0; + int consumed = 0; + for (const auto& l : layouts) { + if (consumed + l.tokens > skipped_rows) break; + consumed += l.tokens; + drop_pictures++; + drop_views += (size_t)l.views; + drop_values += l.values; + } + if (drop_pictures > 0) { + image_payload.images.erase(image_payload.images.begin(), image_payload.images.begin() + drop_views); + image_payload._data__processed.erase(image_payload._data__processed.begin(), + image_payload._data__processed.begin() + drop_values); + image_payload.num_images = (unsigned)image_payload.images.size(); + header_print("FLM", "Prompt-cache hit: dropped " << drop_pictures << " cached image(s) from payload"); + } + } + } + + // The payload rides on the first prefill chunk only, so that chunk must hold + // every image GROUP whole -- through the / that close it, + // which the canvas m-rope positions too -- not just up to the last pad row. + int first_len_run = 0; + { + int last = -1; + for (int i = (int)prefix_skip_count; i < (int)tokens.size(); i++) { + if (tokens[i] == image_pad) last = i; + } + if (last >= 0) { + auto structural = [](int t) { + return t == 248078 || t == 248079 || t == 248088 || t == 248089; + }; + while (last + 1 < (int)tokens.size() && structural(tokens[last + 1])) last++; + first_len_run = last + 1 - (int)prefix_skip_count; + } + } + const bool has_images = !image_payload.images.empty(); + + // FLM_MINICPM_V_DUMP=: the payload and the prompt exactly as the engine + // gets them, for scoring the preprocessing against the HF processor + // (pixels.bf16 raw, views.txt "h w" per view, tokens.txt one id per line). + if (const char* dump = std::getenv("FLM_MINICPM_V_DUMP"); dump && *dump) { + const std::string dir(dump); + std::ofstream(dir + "/pixels.bf16", std::ios::binary) + .write(reinterpret_cast(image_payload._data__processed.data()), + image_payload._data__processed.size() * sizeof(bf16)); + std::ofstream vf(dir + "/views.txt"); + for (const auto& v : image_payload.images) vf << v.grid_h << " " << v.grid_w << "\n"; + std::ofstream tf(dir + "/tokens.txt"); + for (int t : tokens) tf << t << "\n"; + header_print("FLM", "dumped the image payload and prompt to " << dir); + } + + bool success = false; + try { + success = this->_shared_insert(meta_info, tokens, is_cancelled, has_images ? &image_payload : nullptr, + has_images ? first_len_run : 0, input.requested_max_new_tokens); + } catch (const ModelRequestError&) { + throw; + } catch (...) { + this->fail_inference(); + } + + checkpoint_his = token_history; + this->lm_engine->checkpoint(); + return success; +} + +std::string MiniCPM_V::generate(chat_meta_info_t& meta_info, int length_limit, std::ostream& os, std::function is_cancelled) { + this->_shared_guard_poisoned(); + try { + return this->decode(meta_info, length_limit, os, std::move(is_cancelled)); + } catch (const ModelRequestError&) { + throw; + } catch (...) { + this->fail_inference(); + } +} + +std::string MiniCPM_V::decode(chat_meta_info_t& meta_info, int length_limit, std::ostream& os, std::function is_cancelled) { + std::string result; + assert(this->last_token != -1); + stop_reason_t reason = EOT_DETECTED; + + this->profiler_list[DECODING_TIME].reset(); + this->profiler_list[TKOEN_DECODE_TIME].reset(); + // Some backends refuse to decode past a limit of their own, below MAX_L. + const uint32_t decode_cap = this->decode_cap(); + + // The tail of the generation prompt that insert() held back. + const std::vector think_prefix = this->enable_think + ? std::vector{ think_start_id, 198 } // \n + : std::vector{ think_start_id, 271, think_end_id, 271 }; // \n\n\n\n + if (this->total_tokens + think_prefix.size() >= decode_cap) { + header_print("WARNING", "Max length reached, stopping generation..."); + meta_info.stop_reason = MAX_LENGTH_REACHED; + return result; + } + // The first prefix forward is left out of DECODING_TIME, as Qwen3_5VL does, + // so `flm bench` measures decode speed the same way as for qwen3.5:0.8b. + buffer y; + for (size_t i = 0; i < think_prefix.size(); i++) { + this->token_history.push_back(think_prefix[i]); + if (i > 0) this->profiler_list[DECODING_TIME].start(); + y = this->lm_engine->forward(think_prefix[i]); + if (i > 0) this->profiler_list[DECODING_TIME].stop(1); + this->total_tokens++; + } + if (this->enable_think) { + // Echo the opening tag so the stream parser enters reasoning mode. + for (int id : think_prefix) { + std::string token_str = this->tokenizer->run_time_decoder(id); + result += token_str; + os << token_str << std::flush; + } + } + // First real token: the model would open a tool call right here, so it + // needs the mask too. + this->_apply_tool_choice_mask(y, meta_info); + int sampled_token = this->sampler->sample(y); + + while (true) { + this->profiler_list[TKOEN_DECODE_TIME].start(); + if (this->is_normal_token(sampled_token)){ // filter out special tokens + std::string token_str = this->tokenizer->run_time_decoder(sampled_token); + os << token_str << std::flush; + result += token_str; + } + this->profiler_list[TKOEN_DECODE_TIME].stop(1); + this->token_history.push_back(sampled_token); + meta_info.generated_tokens++; + + if (this->is_eos(sampled_token)){ + // Keep the kv cache aligned with token_history for a following turn. + if (this->forward_on_eos && + (!this->backend_ || this->backend_->forwards_past_eos()) && + this->total_tokens < decode_cap) { + this->lm_engine->forward(sampled_token); + this->total_tokens++; + } + break; + } + if ((length_limit > 0) && (meta_info.generated_tokens >= length_limit)){ + reason = MAX_LENGTH_REACHED; + break; + } + if (this->total_tokens >= decode_cap){ + header_print("WARNING", "Max length reached, stopping generation..."); + reason = MAX_LENGTH_REACHED; + break; + } + if (is_cancelled()) { + reason = CANCEL_DETECTED; + // reset stream content + buffer_.clear(); + current_mode_ = StreamEventType::CONTENT; + tool_name_.clear(); + is_in_tool_block_ = false; + break; + } + + this->profiler_list[DECODING_TIME].start(); + y = this->lm_engine->forward(sampled_token); + this->profiler_list[DECODING_TIME].stop(1); + this->total_tokens++; + + this->profiler_list[SAMPLING_TIME].start(); + this->_apply_tool_choice_mask(y, meta_info); + sampled_token = this->sampler->sample(y); + this->profiler_list[SAMPLING_TIME].stop(1); + } + meta_info.decoding_duration = (uint64_t)(time_utils::cast_to_us(this->profiler_list[DECODING_TIME].get_total_time()).first) * 1e3; + meta_info.stop_reason = reason; + + std::cout << std::endl; + if (this->log_raw_output) { + header_print("FLM", "Model RAW Output: \n" + result); + } + return result; +} + +std::string MiniCPM_V::generate_with_prompt(chat_meta_info_t& meta_info, lm_uniform_input_t& input, int length_limit, std::ostream& os) { + if (!this->insert(meta_info, input)) { + return ""; + } + // Not _shared_generate: insert() held back the think prefix, which only + // decode() knows to feed. + return this->generate(meta_info, length_limit, os); +} + +// Non-stream +NonStreamResult MiniCPM_V::parse_nstream_content(const std::string response_text) { + NonStreamResult result; + + const std::string think_start_tag = ""; + const std::string think_end_tag = ""; + std::string start_tag = ""; + std::string end_tag = ""; + std::string func_end_tag = ""; + std::string func_open = " fallback + size_t func_end_pos = answer_text.find(func_end_tag, block_content_start); + if (func_end_pos != std::string::npos) { + block_end = func_end_pos + func_end_tag.length(); + } else { + block_end = answer_text.length(); + } + search_from = block_end; + } + + std::string block = answer_text.substr(block_content_start, block_end - block_content_start); + + std::string tool_name; + size_t func_start = block.find(func_open); + if (func_start != std::string::npos) { + func_start += func_open.length(); + size_t func_name_end = block.find(">", func_start); + if (func_name_end != std::string::npos) { + tool_name = block.substr(func_start, func_name_end - func_start); + } + } + + nlohmann::json args = nlohmann::json::object(); + size_t pos = 0; + + while (true) { + size_t param_start = block.find(param_open, pos); + if (param_start == std::string::npos) break; + + param_start += param_open.length(); + size_t param_name_end = block.find(">", param_start); + if (param_name_end == std::string::npos) break; + + std::string param_name = block.substr(param_start, param_name_end - param_start); + size_t value_start = param_name_end + 1; + size_t value_end = block.find(param_close, value_start); + + size_t next_param_pos = block.find(param_open, value_start); + size_t func_boundary_pos = block.find(func_end_tag, value_start); + + auto use_earlier_boundary = [&value_end](size_t boundary_pos) { + if (boundary_pos != std::string::npos && (value_end == std::string::npos || boundary_pos < value_end)) { + value_end = boundary_pos; + } + }; + + use_earlier_boundary(next_param_pos); + use_earlier_boundary(func_boundary_pos); + + if (value_end == std::string::npos) { + value_end = block.length(); + } + + std::string param_value = trim_tool_value(block.substr(value_start, value_end - value_start)); + + try { + args[param_name] = nlohmann::json::parse(param_value); + } + catch (...) { + args[param_name] = param_value; + } + + pos = value_end; + if (block.compare(value_end, param_close.length(), param_close) == 0) { + pos += param_close.length(); + } + } + + result.tool_calls_list.emplace_back(tool_name, args.dump()); + } + + if (result.tool_calls_list.empty()) { + result.content = trim_tool_value(answer_text); + } else { + // Populate legacy single-tool fields from the first call for backward compatibility + result.tool_name = result.tool_calls_list[0].first; + result.tool_args = result.tool_calls_list[0].second; + // Extract content before the first + size_t first_tool = answer_text.find(start_tag); + if (first_tool != std::string::npos && first_tool > 0) { + result.content = trim_tool_value(answer_text.substr(0, first_tool)); + } + } + + return result; +} + +// Stream +StreamResult MiniCPM_V::parse_stream_content(const std::string content) { + return parse_stream_content_impl(content, false); +} + +StreamResult MiniCPM_V::parse_stream_content_final(const std::string content) { + return parse_stream_content_impl(content, true); +} + +StreamResult MiniCPM_V::parse_stream_content_impl(const std::string content, bool is_final) { + const std::string MARKER_THINK_START = ""; + const std::string MARKER_THINK_END = ""; + const std::string MARKER_TOOL_START = ""; + const std::string MARKER_TOOL_END = ""; + const std::string MARKER_FUNC_END = ""; + + + StreamResult result; + buffer_ += content; + + while (true) { + if (!is_in_tool_block_) { + size_t stray_end_pos = buffer_.find(MARKER_TOOL_END); + if (stray_end_pos != std::string::npos) { + buffer_.erase(stray_end_pos, MARKER_TOOL_END.length()); + } + } + + if (!is_in_tool_block_) { + size_t tool_start_pos = buffer_.find(MARKER_TOOL_START); + if (tool_start_pos != std::string::npos) { + if (tool_start_pos > 0) { + result.content = buffer_.substr(0, tool_start_pos); + result.type = current_mode_; + buffer_ = buffer_.substr(tool_start_pos); + return result; + } + + is_in_tool_block_ = true; + buffer_ = buffer_.substr(MARKER_TOOL_START.length()); + result.type = StreamEventType::WAITING; + return result; + } + } + + // tool calling process + if (is_in_tool_block_) { + size_t tool_end_pos = buffer_.find(MARKER_TOOL_END); + size_t func_end_pos = buffer_.find(MARKER_FUNC_END); + + if (tool_end_pos != std::string::npos || func_end_pos != std::string::npos || (is_final && !buffer_.empty())) { + size_t actual_end_pos = buffer_.size(); + size_t skip_length = 0; + + if (tool_end_pos != std::string::npos) { + actual_end_pos = tool_end_pos; + skip_length = MARKER_TOOL_END.length(); + } + else if (func_end_pos != std::string::npos) { + actual_end_pos = func_end_pos; + skip_length = MARKER_FUNC_END.length(); + } + + std::string block = buffer_.substr(0, actual_end_pos + skip_length); + buffer_ = buffer_.substr(actual_end_pos + skip_length); + is_in_tool_block_ = false; + + try { + result.type = StreamEventType::TOOL_DONE; + result.tool_id = "call_" + std::to_string(std::time(nullptr)); + + // parse function name + std::string func_open = "", func_start); + if (func_end != std::string::npos) { + result.tool_name = block.substr(func_start, func_end - func_start); + } + } + + // parse parameters + nlohmann::json args = nlohmann::json::object(); + std::string param_open = "", p_start); + if (p_name_end == std::string::npos) break; + std::string param_name = block.substr(p_start, p_name_end - p_start); + + size_t val_start = p_name_end + 1; + if (val_start < block.size() && block[val_start] == '\n') val_start++; + + size_t param_close_pos = block.find(param_close, val_start); + size_t val_end = param_close_pos; + + size_t next_param_pos = block.find(param_open, val_start); + size_t func_boundary_pos = block.find(MARKER_FUNC_END, val_start); + size_t tool_boundary_pos = block.find(MARKER_TOOL_END, val_start); + + auto use_earlier_boundary = [&val_end](size_t boundary_pos) { + if (boundary_pos != std::string::npos && (val_end == std::string::npos || boundary_pos < val_end)) { + val_end = boundary_pos; + } + }; + + use_earlier_boundary(next_param_pos); + use_earlier_boundary(func_boundary_pos); + use_earlier_boundary(tool_boundary_pos); + + if (val_end == std::string::npos && is_final) { + val_end = block.size(); + } + if (val_end == std::string::npos) break; + + std::string param_value = block.substr(val_start, val_end - val_start); + + // Enhanced trim: handle multiple newlines or spaces that the model may generate after a parameter + while(!param_value.empty() && (param_value.back() == '\n' || param_value.back() == '\r' || param_value.back() == ' ')) { + param_value.pop_back(); + } + + try { + // Try to parse as native JSON type (Integer, Float, Boolean, Array, Object) + args[param_name] = nlohmann::json::parse(param_value); + } + catch (...) { + args[param_name] = param_value; + } + + search_pos = param_close_pos != std::string::npos && val_end == param_close_pos + ? val_end + param_close.length() + : val_end; + } + result.tool_args_str = args.dump(); + return result; + } + catch (...) { + result.type = StreamEventType::CONTENT; + result.content = "[Error parsing tool call]"; + return result; + } + } + else { + result.type = StreamEventType::WAITING; + return result; + } + } + + if (current_mode_ == StreamEventType::CONTENT) { + size_t think_start_pos = buffer_.find(MARKER_THINK_START); + if (think_start_pos != std::string::npos) { + if (think_start_pos > 0) { + result.content = buffer_.substr(0, think_start_pos); + result.type = StreamEventType::CONTENT; + buffer_ = buffer_.substr(think_start_pos); + return result; + } + buffer_ = buffer_.substr(MARKER_THINK_START.length()); + current_mode_ = StreamEventType::REASONING; + continue; + } + } + else if (current_mode_ == StreamEventType::REASONING) { + size_t think_end_pos = buffer_.find(MARKER_THINK_END); + if (think_end_pos != std::string::npos) { + if (think_end_pos > 0) { + result.content = buffer_.substr(0, think_end_pos); + result.type = StreamEventType::REASONING; + buffer_ = buffer_.substr(think_end_pos); + return result; + } + buffer_ = buffer_.substr(MARKER_THINK_END.length()); + current_mode_ = StreamEventType::CONTENT; + continue; + } + } + + if (!buffer_.empty()) { + size_t last_lt = buffer_.rfind('<'); + // If '<' appears at the end (possibly an incomplete or tag) + if (last_lt != std::string::npos && (buffer_.length() - last_lt) <= 15) { + if (last_lt > 0) { + // Only output the content before '<' + result.content = buffer_.substr(0, last_lt); + result.type = current_mode_; + buffer_ = buffer_.substr(last_lt); + return result; + } else { + // If '<' is the first character in the buffer, directly wait for the next chunk + result.type = StreamEventType::WAITING; + return result; + } + } + + result.content = buffer_; + result.type = current_mode_; + buffer_.clear(); + return result; + } + + break; + } + + result.type = current_mode_; + return result; +} diff --git a/src/common/AutoModel/modeling_minicpm_v_image.cpp b/src/common/AutoModel/modeling_minicpm_v_image.cpp new file mode 100644 index 00000000..503c15d8 --- /dev/null +++ b/src/common/AutoModel/modeling_minicpm_v_image.cpp @@ -0,0 +1,317 @@ +/// \file modeling_minicpm_v_image.cpp +/// \brief MiniCPM_V image preprocessing: MiniCPMV4ImageProcessorPil in C++. +/// \author FastFlowLM Team +/// \date 2026-10-05 +/// \note Port of the checkpoint's own image_processing_pil_minicpmv4.py and of +/// the placeholder expansion in processing_minicpmv4.py, rule for rule. +/// Every number here (the 448 scale, the factor 4 in ensure_divide, the +/// slice-grid search) changes how many views and tokens an image produces, +/// and the engine checks that the prompt's image rows match its views +/// exactly -- a wrong rule fails loudly there rather than quietly here. +/// +/// Layout handed to the engine (minicpm_v_4_7_npu.cpp, _prefill_with_mm): one +/// minicpm_v_image_t per VIEW, and `_data__processed` holding every view's +/// patches back to back -- raster order within a view, each patch flattened +/// (c, kh, kw) as 3*14*14 bf16. That is the NaViT `[3, 14, T*14]` tensor the +/// HF processor emits, transposed per patch. + +#include "AutoModel/modeling_minicpm_v.hpp" + +#include +#include +#include + +namespace { + +// ---- the processor's arithmetic, as Python computes it ------------------------ +// Python's round() is half-to-even and int() truncates; nearbyint and a cast +// are those two exactly. + +/// `max(round(length / divisor) * divisor, divisor)` +int ensure_divide(double length, int divisor) { + return std::max((int)std::nearbyint(length / divisor) * divisor, divisor); +} + +/// `find_best_resize()`: (height, width) in pixels, multiples of patch*4 -- two +/// successive 2x2 merges. The inputs are floats on the refine path. +void find_best_resize(double height, double width, int scale, int patch, bool allow_upscale, + int& best_h, int& best_w) { + if (height * width > (double)scale * scale || allow_upscale) { + const double ar = width / height; + const int h = (int)(scale / std::sqrt(ar)); + const int w = (int)(h * ar); + height = h; + width = w; + } + best_w = ensure_divide(width, patch * 4); + best_h = ensure_divide(height, patch * 4); +} + +/// `get_sliced_grid()`: (rows, cols) of slices, or false for no slicing. +bool sliced_grid(int h, int w, int max_slices, int scale, int& rows, int& cols) { + const double log_ratio = std::log((double)w / h); + const double ratio = (double)w * h / ((double)scale * scale); + const int multiple = std::min((int)std::ceil(ratio), max_slices); + if (multiple <= 1) return false; + + rows = cols = 1; + double min_error = std::numeric_limits::infinity(); + for (int num_slices : {multiple - 1, multiple, multiple + 1}) { + if (num_slices == 1 || num_slices > max_slices) continue; + for (int r = 1; r <= num_slices; r++) { + if (num_slices % r) continue; + const int c = num_slices / r; + const double error = std::fabs(log_ratio - std::log((double)c / r)); + if (error < min_error) { rows = r; cols = c; min_error = error; } + else if (error == min_error && r > rows) { rows = r; cols = c; } + } + } + return true; +} + +// ---- Pillow's bicubic resample, bit for bit ------------------------------------ +// +// The HF processor resizes the uint8 image with PIL (Image.resize, BICUBIC; the +// output is exactly PIL's, checked). The shared AVX-512 bicubic +// (imgproc::avx512::resize_bicubic_antialias_rgb_planar_avx512) is close but not +// the same function -- it disagrees with PIL by up to 29 uint8 levels at +// high-contrast pixels (relL2 0.013 on the reference photo), and SigLIP +// amplifies that ~9x: image_embeds relL2 0.116 through HF's own fp32 tower, +// more than every device rounding in the tower put together (0.080). So this +// tower gets Pillow's own algorithm (libImaging/Resample.c, ImagingResample): +// double-precision coefficients normalised per output pixel, quantised to +// 22-bit fixed point, a horizontal pass into a CLAMPED uint8 image, then a +// vertical pass. + +constexpr int PIL_PRECISION_BITS = 32 - 8 - 2; + +double pil_bicubic(double x) { + constexpr double a = -0.5; + if (x < 0.0) x = -x; + if (x < 1.0) return ((a + 2.0) * x - (a + 3.0)) * x * x + 1; + if (x < 2.0) return (((x - 5) * x + 8) * x - 4) * a; + return 0.0; +} + +/// precompute_coeffs() + normalize_coeffs_8bpc(): per output pixel, the first +/// input index, the tap count, and ksize fixed-point weights. +int pil_coeffs(int in_size, int out_size, std::vector& bounds, std::vector& kk) { + const double scale = (double)in_size / out_size; + const double filterscale = std::max(scale, 1.0); + const double support = 2.0 * filterscale; // bicubic support 2 + const int ksize = (int)std::ceil(support) * 2 + 1; + bounds.assign((size_t)out_size * 2, 0); + kk.assign((size_t)out_size * ksize, 0); + std::vector k((size_t)ksize); + for (int xx = 0; xx < out_size; xx++) { + const double center = (xx + 0.5) * scale; + const double ss = 1.0 / filterscale; + int xmin = (int)(center - support + 0.5); + if (xmin < 0) xmin = 0; + int xmax = (int)(center + support + 0.5); + if (xmax > in_size) xmax = in_size; + xmax -= xmin; + double ww = 0.0; + for (int x = 0; x < xmax; x++) { + const double w = pil_bicubic((x + xmin - center + 0.5) * ss); + k[(size_t)x] = w; + ww += w; + } + for (int x = 0; x < xmax; x++) { + double w = ww != 0.0 ? k[(size_t)x] / ww : k[(size_t)x]; + kk[(size_t)xx * ksize + x] = w < 0 ? (int32_t)(-0.5 + w * (1 << PIL_PRECISION_BITS)) + : (int32_t)(0.5 + w * (1 << PIL_PRECISION_BITS)); + } + bounds[(size_t)xx * 2] = xmin; + bounds[(size_t)xx * 2 + 1] = xmax; + } + return ksize; +} + +inline uint8_t pil_clip8(int32_t v) { + v >>= PIL_PRECISION_BITS; // arithmetic: floor, as Pillow's lookup table indexes + return (uint8_t)(v < 0 ? 0 : v > 255 ? 255 : v); +} + +/// Planar uint8 [3][h][w] -> [3][oh][ow], identical to PIL's Image.resize(BICUBIC). +std::vector pil_resize_bicubic(const uint8_t* src, int w, int h, int ow, int oh) { + std::vector bh, bv; + std::vector kh, kv; + const int ksh = pil_coeffs(w, ow, bh, kh); + const int ksv = pil_coeffs(h, oh, bv, kv); + const bool need_h = ow != w, need_v = oh != h; + + // Only the input rows the vertical pass reads, as ImagingResample does. + const int y_first = bv[0]; + const int y_last = bv[(size_t)oh * 2 - 2] + bv[(size_t)oh * 2 - 1]; + + std::vector out((size_t)3 * oh * ow); + for (int c = 0; c < 3; c++) { + const uint8_t* in = src + (size_t)c * h * w; + // horizontal pass, into a clamped uint8 temporary + const int th = need_h ? (y_last - y_first) : h; + const int y0 = need_h ? y_first : 0; + std::vector tmp; + const uint8_t* mid = in; + int mid_w = w; + if (need_h) { + tmp.resize((size_t)th * ow); +#pragma omp parallel for schedule(static) + for (int y = 0; y < th; y++) { + const uint8_t* row = in + (size_t)(y + y0) * w; + for (int xx = 0; xx < ow; xx++) { + const int xmin = bh[(size_t)xx * 2], xmax = bh[(size_t)xx * 2 + 1]; + const int32_t* k = &kh[(size_t)xx * ksh]; + int32_t ss = 1 << (PIL_PRECISION_BITS - 1); + for (int x = 0; x < xmax; x++) ss += (int32_t)row[x + xmin] * k[x]; + tmp[(size_t)y * ow + xx] = pil_clip8(ss); + } + } + mid = tmp.data(); + mid_w = ow; + } + // vertical pass; bounds shifted by the rows the horizontal pass skipped + uint8_t* o = out.data() + (size_t)c * oh * ow; + if (need_v) { + const int shift = need_h ? y_first : 0; +#pragma omp parallel for schedule(static) + for (int yy = 0; yy < oh; yy++) { + const int ymin = bv[(size_t)yy * 2] - shift, ymax = bv[(size_t)yy * 2 + 1]; + const int32_t* k = &kv[(size_t)yy * ksv]; + for (int xx = 0; xx < mid_w; xx++) { + int32_t ss = 1 << (PIL_PRECISION_BITS - 1); + for (int y = 0; y < ymax; y++) ss += (int32_t)mid[(size_t)(y + ymin) * mid_w + xx] * k[y]; + o[(size_t)yy * ow + xx] = pil_clip8(ss); + } + } + } else { + std::copy(mid, mid + (size_t)oh * ow, o); + } + } + return out; +} + +/// Append one view's patches -- the region (y0, x0, vh, vw) of a normalised +/// planar [3][H][W] image -- as raster rows of (c, kh, kw). +void patchify(const float* img, int H, int W, int y0, int x0, int vh, int vw, int p, + std::vector& out) { + const int gh = vh / p, gw = vw / p; + const size_t row = (size_t)3 * p * p; + const size_t base = out.size(); + out.resize(base + (size_t)gh * gw * row); +#pragma omp parallel for schedule(static) + for (int t = 0; t < gh * gw; t++) { + const int pr = t / gw, pc = t % gw; + bf16* dst = out.data() + base + (size_t)t * row; + for (int c = 0; c < 3; c++) + for (int kh = 0; kh < p; kh++) { + const float* src = img + ((size_t)c * H + (y0 + pr * p + kh)) * W + x0 + pc * p; + for (int kw = 0; kw < p; kw++) *dst++ = (bf16)src[kw]; + } + } +} + +} // namespace + +bool MiniCPM_V::load_image(const std::string& source, bool base64, image_data_t& chw) { + image_data_t decoded; + const bool ok = base64 ? image_reader_.load_image_base64(source, decoded) + : image_reader_.load_image(source, decoded); + if (!ok) return false; + const bool reordered = image_reader_.reorder_hwc_to_chw(decoded, chw); + image_reader_.recycle(decoded); + return reordered && chw.width > 0 && chw.height > 0; +} + +MiniCPM_V::image_layout MiniCPM_V::preprocess_image(const image_data_t& chw, + minicpm_v_image_payload_t& payload) { + auto* eng = reinterpret_cast(this->lm_engine); + const int patch = (int)eng->MINICPM_V_PATCH_SIZE; // 14 + const int max_slices = (int)eng->MINICPM_V_MAX_SLICE_NUMS; // 9 + const int divisor = (int)eng->MINICPM_V_DOWNSAMPLE_FACTOR; // 16 (or 4) + constexpr int scale = 448; // scale_resolution: the processor's default; not in any config + const float rescale = eng->MINICPM_V_VISION_RESCALE_FACTOR; + const float mean = eng->MINICPM_V_VISION_RESCALE_IMAGE_MEAN; + const float stdv = eng->MINICPM_V_VISION_RESCALE_IMAGE_STD; + + const int H = chw.height, W = chw.width; + const uint8_t* src = chw.pixels.data(); + + image_layout l; + const bool sliced = sliced_grid(H, W, max_slices, scale, l.rows, l.cols); + if (!sliced) l.rows = l.cols = 0; + const size_t before = payload._data__processed.size(); + + // resize (Pillow's bicubic, exactly), rescale + normalise, return planar fp32 + std::vector norm; + auto resized = [&](int h, int w) { + std::vector r = pil_resize_bicubic(src, W, H, w, h); + norm.resize((size_t)3 * h * w); + imgproc::avx512::rescale_and_normalize_avx512(r.data(), norm.data(), w, h, 3, true, rescale, + true, mean, stdv); + }; + auto push_view = [&](int y0, int x0, int vh, int vw, int img_h, int img_w) { + patchify(norm.data(), img_h, img_w, y0, x0, vh, vw, patch, payload._data__processed); + minicpm_v_image_t v{}; + v.height = H; + v.width = W; + v.height_resized = vh; + v.width_resized = vw; + v.grid_h = vh / patch; + v.grid_w = vw / patch; + payload.images.push_back(v); + return v.grid_h * v.grid_w / divisor; + }; + + // the source view: upscaled only when the image is not sliced + int sh, sw; + find_best_resize(H, W, scale, patch, /*allow_upscale=*/!sliced, sh, sw); + resized(sh, sw); + l.src_tokens = push_view(0, 0, sh, sw, sh, sw); + l.views = 1; + l.tokens = l.src_tokens; + + if (sliced) { + // get_refine_size(): the whole image resized so it tiles exactly + const int rw0 = ensure_divide(W, l.cols), rh0 = ensure_divide(H, l.rows); + int bh, bw; + find_best_resize((double)rh0 / l.rows, (double)rw0 / l.cols, scale, patch, true, bh, bw); + const int rh = bh * l.rows, rw = bw * l.cols; + resized(rh, rw); + // divide_to_patches(): row-major tiles + for (int r = 0; r < l.rows; r++) + for (int c = 0; c < l.cols; c++) { + l.slice_tokens = push_view(r * bh, c * bw, bh, bw, rh, rw); + l.tokens += l.slice_tokens; + l.views++; + } + } + payload.num_images = (unsigned)payload.images.size(); + l.values = payload._data__processed.size() - before; + return l; +} + +std::vector MiniCPM_V::image_placeholder(const image_layout& l, int index) { + constexpr int image_pad = 248056; + constexpr int image_start = 248078, image_end = 248079; // + constexpr int slice_start = 248088, slice_end = 248089; // + constexpr int image_id_start = 248090, image_id_end = 248091; // + constexpr int newline = 198; + + std::vector t; + t.push_back(image_id_start); + for (int d : this->tokenizer->encode(std::to_string(index))) t.push_back(d); + t.push_back(image_id_end); + t.push_back(image_start); + t.insert(t.end(), (size_t)l.src_tokens, image_pad); + t.push_back(image_end); + for (int r = 0; r < l.rows; r++) { + if (r > 0) t.push_back(newline); // "\n".join over slice rows + for (int c = 0; c < l.cols; c++) { + t.push_back(slice_start); + t.insert(t.end(), (size_t)l.slice_tokens, image_pad); + t.push_back(slice_end); + } + } + return t; +} diff --git a/src/create_new_model.md b/src/create_new_model.md index 73c1dbea..d92e3351 100644 --- a/src/create_new_model.md +++ b/src/create_new_model.md @@ -119,16 +119,13 @@ Gemma4_12B::Gemma4_12B(flm_rt::device* npu_device_inst) ### 2b. `load_model` -Always in this order — the shared helper parses `config.json`, resolves `MAX_L`, -and sets `this->model_path` / `this->lm_config` before you can construct the engine: +`_shared_load_backend` parses `config.json`, resolves `MAX_L`, sets +`this->model_path` / `this->lm_config`, then builds the engine through the +backend registry (Step 4b) and points `this->lm_engine` at it -- the frontend no +longer constructs the engine or loads the Q4NX itself: ```cpp -this->_shared_load_model(model_path, model_info, default_context_length, enable_preemption); -this->q4nx = std::make_unique(this->model_path); -this->lm_engine = std::make_unique(*this->lm_config, this->npu.get(), this->MAX_L); -this->lm_engine->load_weights(*this->q4nx); -this->q4nx.reset(); // free the mmap'd weights immediately -this->lm_engine->clear_context(); +this->_shared_load_backend(model_path, model_info, default_context_length, enable_preemption, backend); this->setup_tokenizer(model_path); this->sampler.reset(); sampler_config config; // set the model's recommended defaults here @@ -231,6 +228,18 @@ case SupportedModelFamily::gemma4_12b: A mismatch between (3) and `model_list.json` shows up at runtime as "unsupported model family", not at compile time. +### Step 4b — Register the engine backend + +`common/AutoModel/builtin_backends.cpp` — one line, keyed by the same +`details.family` string: + +```cpp +RegisterFlm(registry, "gemma4-12b"); +``` + +Without it `load_model` throws at runtime naming the backends the family does +register (none). + --- ## Step 5 — Link the engine library @@ -389,6 +398,7 @@ Linux path. Starts with `-include ../common.mk` (which sets ```make SOURCES += test.cpp SOURCES += ../../common/AutoModel/automodel.cpp +SOURCES += ../../common/AutoModel/model_backend.cpp SOURCES += ../../common/AutoModel/modeling_gemma4_12b.cpp SOURCES += ../../common/tokenizer/tokenizer.cpp SOURCES += ../../common/modules/sampler.cpp @@ -399,6 +409,21 @@ LDFLAGS += -lgemma4_12b_npu The `test` target copies `model_list.json` into `BUILD_DIR` before running — the executable looks for it next to itself. +`automodel.cpp` reaches the engine through the backend registry, so the harness +must also compile `model_backend.cpp` and define `register_builtin_backends` +itself, registering only its own engine (`builtin_backends.cpp` would pull in +every engine library). See `test/minicpm_v_4_7_npu/test.cpp`: + +```cpp +#include "AutoModel/flm_backend.hpp" + +namespace flm::backend { +void register_builtin_backends(BackendRegistry& registry) { + registry.register_backend("gemma4-12b", kFlmBackendId, flm_factory()); +} +} // namespace flm::backend +``` + ### `test//activate.sh` (optional convenience) ```bash @@ -440,6 +465,7 @@ flm serve gemma4-12b:12b # then exercise /v1/chat/completions, streaming, an - [ ] `common/AutoModel/modeling_.cpp` - [ ] Engine include added to `include/AutoModel/automodel.hpp` - [ ] All four edits in `include/AutoModel/all_models.hpp` +- [ ] `RegisterFlm<…>` line in `common/AutoModel/builtin_backends.cpp` - [ ] Engine lib added to `target_link_libraries(flm PUBLIC …)` in `CMakeLists.txt` - [ ] `model_list.json` entry, `details.family` matches `modelFamilyMap` - [ ] `model_info.json` manifest (if `flm pull` support is wanted) diff --git a/src/include/AutoModel/all_models.hpp b/src/include/AutoModel/all_models.hpp index 0392fe8e..87a2972a 100644 --- a/src/include/AutoModel/all_models.hpp +++ b/src/include/AutoModel/all_models.hpp @@ -18,6 +18,7 @@ #include "modeling_qwen2vl.hpp" #include "modeling_qwen3vl.hpp" #include "modeling_qwen3_5vl.hpp" +#include "modeling_minicpm_v.hpp" #include "modeling_qwen3_5_omni.hpp" #include "modeling_qwen3_6_moe.hpp" #include "modeling_qwen3_8mtp.hpp" @@ -40,6 +41,7 @@ typedef enum { qwen3vl, qwen3vl_flash, qwen3_5, + minicpm_v, qwen3_5_omni, qwen3_6_moe, qwen3_8mtp, @@ -72,6 +74,7 @@ inline std::pair> get_auto_model(const s {"qwen3vl", SupportedModelFamily::qwen3vl}, {"qwen3vl-flash", SupportedModelFamily::qwen3vl_flash}, {"qwen3.5", SupportedModelFamily::qwen3_5}, + {"minicpm-v-4.7", SupportedModelFamily::minicpm_v}, {"qwen3.5-omni", SupportedModelFamily::qwen3_5_omni}, {"qwen3.6-moe", SupportedModelFamily::qwen3_6_moe}, {"qwen3.8-mtp", SupportedModelFamily::qwen3_8mtp}, @@ -154,6 +157,9 @@ inline std::pair> get_auto_model(const s case SupportedModelFamily::qwen3_5: auto_chat_engine = std::make_unique(npu_device_inst); break; + case SupportedModelFamily::minicpm_v: + auto_chat_engine = std::make_unique(npu_device_inst); + break; case SupportedModelFamily::qwen3_5_omni: auto_chat_engine = std::make_unique(npu_device_inst); break; diff --git a/src/include/AutoModel/automodel.hpp b/src/include/AutoModel/automodel.hpp index 32bb9ac7..8ba53524 100644 --- a/src/include/AutoModel/automodel.hpp +++ b/src/include/AutoModel/automodel.hpp @@ -32,6 +32,7 @@ #include "models/qwen3vl/flm/aie2p/qwen3vl_npu.hpp" #include "models/qwen3vl_flash/flm/aie2p/qwen3vl_flash.hpp" #include "models/qwen3_5vl/flm/aie2p/qwen3_5vl_npu.hpp" +#include "models/minicpm_v_4_7/flm/aie2p/minicpm_v_4_7_npu.hpp" #include "models/qwen3_6_moe/flm/aie2p/qwen3_6_moe_npu.hpp" #include "models/qwen3_8mtp/flm/aie2p/qwen3_8mtp_npu.hpp" #include "models/gemma/flm/aie2p/gemma_npu.hpp" diff --git a/src/include/AutoModel/modeling_minicpm_v.hpp b/src/include/AutoModel/modeling_minicpm_v.hpp new file mode 100644 index 00000000..34c6b61d --- /dev/null +++ b/src/include/AutoModel/modeling_minicpm_v.hpp @@ -0,0 +1,118 @@ +/// \file modeling_minicpm_v.hpp +/// \brief MiniCPM_V class +/// \author FastFlowLM Team +/// \date 2026-10-02 +/// \version 0.9.28 +/// \note This is a header file for the MiniCPM_V class +/// \note MiniCPM-V-4.7-1B's text tower is config-identical to Qwen3.5-0.8B and +/// shares its tokenizer ids, so the chat flow follows Qwen3_5VL. +/// \note Images (S7b): each picture is sliced the way MiniCPMV4ImageProcessorPil +/// slices it -- a source view plus up to max_slice_nums slices -- and each +/// VIEW becomes one minicpm_v_image_t in the payload. The prompt's single +/// <|image_pad|> per picture is expanded to the processor's placeholder +/// (Npad..pad....). See +/// modeling_minicpm_v_image.cpp. Audio is dropped with a warning; video is +/// not supported. + +#pragma once +#include "AutoModel/automodel.hpp" +#include "image/image_reader.hpp" +#include "image_process_utils/imageproc.hpp" +#include "image_process_utils/imageprocAVX512.hpp" + +#include + +/************ MiniCPM_V **************/ +class MiniCPM_V : public AutoModel { +private: + + bool enable_think = false; + bool enable_tool = false; + static constexpr int think_start_id = 248068; + static constexpr int think_end_id = 248069; + static constexpr int tool_start_token_id = 248058; + + void setup_tokenizer(std::string model_path); + + // ---- images -------------------------------------------------------------- + /// \brief How one picture was sliced: what the prompt placeholder needs. + struct image_layout { + int views = 0; ///< 1 + rows*cols + int rows = 0, cols = 0;///< slice grid, 0 x 0 when unsliced + int src_tokens = 0; ///< LLM rows of the source view + int slice_tokens = 0; ///< LLM rows of each slice + int tokens = 0; ///< all image_token_id rows of the picture + size_t values = 0; ///< bf16 values it adds to _data__processed + }; + ImageReader image_reader_; + /// \brief Decode to uint8 CHW; false on failure. + bool load_image(const std::string& source, bool base64, image_data_t& chw); + /// \brief Slice, resize, normalise and patchify one picture into `payload`. + image_layout preprocess_image(const image_data_t& chw, minicpm_v_image_payload_t& payload); + /// \brief The token sequence `<|image_pad|>` expands to for picture `index`. + std::vector image_placeholder(const image_layout& l, int index); + + /// \brief Clear the conversation and throw after a failed inference + [[noreturn]] void fail_inference(); + + std::string decode(chat_meta_info_t& meta_info, int length_limit, std::ostream& os, std::function is_cancelled); + +public: + MiniCPM_V(flm_rt::device* npu_device_inst); + + void load_model(std::string model_path, json model_inf, int default_context_length = -1, bool enable_preemption = false, const std::string& backend = "") override; + bool insert(chat_meta_info_t& meta_info, lm_uniform_input_t& input, std::function is_cancelled = [] { return false; }) override; + std::string generate(chat_meta_info_t& meta_info, int length_limit, std::ostream& os, std::function is_cancelled = [] { return false; }) override; + std::string generate_with_prompt(chat_meta_info_t& meta_info, lm_uniform_input_t& input, int length_limit, std::ostream& os = std::cout) override; + std::string apply_chat_template(nlohmann::ordered_json& messages, nlohmann::ordered_json tools = nlohmann::ordered_json::object()) override; + NonStreamResult parse_nstream_content(const std::string response_text) override; + StreamResult parse_stream_content(const std::string content) override; + StreamResult parse_stream_content_final(const std::string content) override; + +private: + StreamResult parse_stream_content_impl(const std::string content, bool is_final); + +public: + + /// \note MiniCPM-V opens every tool call with , so masking that + /// one token is enough to honour tool_choice=none. Gated on enable_tool + /// to match apply_chat_template, which only renders tools when it is set. + int get_tool_start_token_id() const override { + return this->enable_tool ? tool_start_token_id : -1; + } + + /// \brief Configure a parameter with type-erased value + /// \param parameter_name the name of the parameter + /// \param value the value to set (can be any type) + /// \return true if the parameter was configured successfully, false otherwise + bool configure_parameter(std::string parameter_name, const std::any& value) override{ + if (parameter_name == "enable_think") { + try { + this->enable_think = std::any_cast(value); + return true; + } catch (const std::bad_any_cast&) { + return false; + } + } + else if (parameter_name == "reasoning_effort") { + std::string reasoning_effort; + try { + reasoning_effort = std::any_cast(value); + if (reasoning_effort == "high" || reasoning_effort == "medium" || reasoning_effort == "low") + this->enable_think = true; + else if (reasoning_effort == "none") + this->enable_think = false; + else + header_print("WARNING", "Reasoning effort must be 'none', 'low', 'medium' or 'high'!"); + return true; + } catch (const std::bad_any_cast&) { + return false; + } + } + else if (parameter_name == "toggle_think") { + this->enable_think = !this->enable_think; + return true; + } + return AutoModel::configure_parameter(parameter_name, value); + } +}; diff --git a/src/include/models/minicpm_v_4_7/flm/aie2p/minicpm_v_4_7_npu.hpp b/src/include/models/minicpm_v_4_7/flm/aie2p/minicpm_v_4_7_npu.hpp new file mode 100644 index 00000000..a7b8ff16 --- /dev/null +++ b/src/include/models/minicpm_v_4_7/flm/aie2p/minicpm_v_4_7_npu.hpp @@ -0,0 +1,179 @@ +/// \file minicpm_v_4_7_npu.hpp +/// \brief minicpm_v_4_7_npu class -- MiniCPM-V-4.7-1B engine (text tower + SigLIP vision tower) +/// \author FastFlowLM Team +/// \date 2026-10-02 +/// \version 0.9.28 +/// +/// \note CONTRACT FILE. This is the engine ABI: S3 implements it, S6 (the AutoModel +/// wrapper) and S7 (vision) compile against it. Do not edit without the brain's +/// agreement -- see /scratch/michyu/minicpm-1b/CONTRACTS.md. +/// +/// MiniCPM-V-4.7-1B's text tower is config-identical to Qwen3.5-0.8B (hidden 1024, +/// 8/2 heads, head_dim 256, FFN 3584, 24 layers, 16x128 GatedDeltaNet heads, conv 4, +/// full_attention_interval 4, rope_theta 1e7, mrope_section [11,11,10]), so the text +/// path reuses the Qwen3.5-0.8B xclbins unchanged. The vision tower is SigLIP and +/// shares nothing with Qwen3.5-VL's ViT. +#pragma once +#include "lm_config.hpp" +#include "npu_utils/npu_utils.hpp" +#include "tensor_utils/q4_npu_eXpress.hpp" +#include "modules/embedding.hpp" +#include "modules/lm_head.hpp" +#include "modules/gemm.hpp" +#include "modules/dequant.hpp" +#include "tensor_2d.hpp" +#include "utils/utils.hpp" +#include "causal_lm.hpp" +#if USEAVX2 +#include // For AVX intrinsics +#endif + +/// \brief One decoded image, before and after preprocessing. +/// +/// MiniCPM slices a large image into at most `MINICPM_V_MAX_SLICE_NUMS` tiles plus one +/// global view; each tile becomes one of these. `grid_h`/`grid_w` are the patch-grid +/// dimensions fed to the vision tower as `target_sizes`, i.e. `height_resized / +/// MINICPM_V_PATCH_SIZE` and `width_resized / MINICPM_V_PATCH_SIZE`. +typedef struct { + int height; + int width; + int height_resized; // assigned by image preprocessing + int width_resized; + int grid_h; // patch rows == height_resized / patch_size + int grid_w; // patch cols == width_resized / patch_size + + bytes _data; +} minicpm_v_image_t; + +/// \brief The multimodal blob handed to prefill() as `void* payload`. +/// +/// `_data__processed` is the concatenated, normalised pixel buffer for every image in +/// `images`, laid out tile-major. The engine turns it into soft tokens and splices them +/// over the rows whose id equals `image_token_id` (248056) -- see §7 of +/// ADDING_A_NEW_MODEL.md: those rows must bypass any embedding scale. +typedef struct { + std::vector images; + std::vector _data__processed; + unsigned int num_images; +} minicpm_v_image_payload_t; + +class minicpm_v_4_7_npu : public causal_lm { +public: + /// \brief initialize the minicpm_v_4_7_npu + /// \param config the configuration + /// \param npu_instance the npu instance + minicpm_v_4_7_npu(LM_Config config, npu_xclbin_manager *npu_instance, int MAX_L = 4096); + ~minicpm_v_4_7_npu(); + + /// \brief forward the minicpm_v_4_7_npu -- decode one token against the cached KV + /// \param ids the token id + /// \return the logits + buffer forward(int ids) override; + + /// \brief prefill the whole prompt + /// \param ids the token ids + /// \param payload nullptr for text-only, else a minicpm_v_image_payload_t* + buffer prefill(std::vector& ids, void* payload = nullptr) override; + + /// \brief set the context length + /// \param L the context length + void set_context_length(int L) override; + + /// \brief load the weights + /// \param q4nx the q4nx + void load_weights(Q4NX& q4nx) override; + + /// \brief clear the conversation state + void clear_context() override; + + /// \brief get the k cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the k cache + buffer get_k_cache(int layer_idx, int idx) override; + + /// \brief get the v cache + /// \param layer_idx the layer index + /// \param idx the index + /// \return the v cache + buffer get_v_cache(int layer_idx, int idx) override; + + /// \brief grow the caches without dropping the conversation + /// \param MAX_L the max length + void update_max_length(uint32_t MAX_L) override; + + /// \brief get the current context length + /// \return the current context length + int get_current_context_length() override; + int checkpoint() override; + int restore() override; + + // ---------------------------------------------------------------------- + // Vision preprocessing parameters. + // + // These live in preprocessor_config.json and config.json's vision_config, NOT in + // the top-level config.json, so they are read through a + // preprocessor -> config -> literal fallback chain. A checkpoint packaged before + // preprocessor_config.json existed still preprocesses correctly. + // ---------------------------------------------------------------------- + unsigned int MINICPM_V_PATCH_SIZE; ///< vision_config.patch_size (14) + unsigned int MINICPM_V_IMAGE_SIZE; ///< vision_config.image_size (980) + unsigned int MINICPM_V_MAX_SLICE_NUMS; ///< preprocessor.max_slice_nums (9) + unsigned int MINICPM_V_INSERT_LAYER_ID; ///< config.insert_layer_id (6) + unsigned int MINICPM_V_SPATIAL_MERGE_SIZE; ///< 2x2 per merge stage (2) + unsigned int MINICPM_V_TEMPORAL_PATCH_SIZE; ///< always 1 -- MiniCPM has no temporal + ///< patching; video frames are independent + ///< images, unlike Qwen3.5-VL's 2. + unsigned int MINICPM_V_DOWNSAMPLE_FACTOR; ///< config.downsample_mode "16x" (16) + float MINICPM_V_VISION_RESCALE_FACTOR; ///< 1/255 + float MINICPM_V_VISION_RESCALE_IMAGE_MEAN; ///< preprocessor.image_mean[0] (0.5) + float MINICPM_V_VISION_RESCALE_IMAGE_STD; ///< preprocessor.image_std[0] (0.5) + + /// \brief Read the vision preprocessing parameters. + /// + /// \note Called by Impl's constructor. + /// + /// Source order follows the shipped-config convention (see + /// `Qwen3.5-0.8B-NPU2/config.json`): the packaged `config.json` is a FLATTENED + /// config whose `vision_config` block carries ENGINE-named keys + /// (`MINICPM_V_PATCH_SIZE`, ...), not the upstream HuggingFace names. Each field + /// falls back to the upstream HF name, then to the literal from the shipping + /// MiniCPM-V-4.7-1B checkpoint, so a checkpoint packaged either way works and + /// nothing throws. + inline void load_vision_preprocess_parameters(LM_Config& config) { + const nlohmann::json& vc = config.sub("vision_config"); + + MINICPM_V_PATCH_SIZE = + cfg_get(vc, "MINICPM_V_PATCH_SIZE", cfg_get(vc, "patch_size", 14u)); + MINICPM_V_IMAGE_SIZE = + cfg_get(vc, "MINICPM_V_IMAGE_SIZE", cfg_get(vc, "image_size", 980u)); + MINICPM_V_MAX_SLICE_NUMS = + cfg_get(vc, "MINICPM_V_MAX_SLICE_NUMS", + cfg_get(config._json_config, "max_slice_nums", 9u)); + MINICPM_V_INSERT_LAYER_ID = + cfg_get(vc, "MINICPM_V_INSERT_LAYER_ID", + cfg_get(config._json_config, "insert_layer_id", 6u)); + MINICPM_V_SPATIAL_MERGE_SIZE = cfg_get(vc, "MINICPM_V_SPATIAL_MERGE_SIZE", 2u); + MINICPM_V_TEMPORAL_PATCH_SIZE = cfg_get(vc, "MINICPM_V_TEMPORAL_PATCH_SIZE", 1u); + + // downsample_mode is a string ("16x" / "4x") upstream, so the packaged config + // carries the resolved integer instead. + MINICPM_V_DOWNSAMPLE_FACTOR = cfg_get(vc, "MINICPM_V_DOWNSAMPLE_FACTOR", 0u); + if (MINICPM_V_DOWNSAMPLE_FACTOR == 0u) { + const std::string mode = + cfg_get(config._json_config, "downsample_mode", std::string("16x")); + MINICPM_V_DOWNSAMPLE_FACTOR = (mode == "4x") ? 4u : 16u; + } + + MINICPM_V_VISION_RESCALE_FACTOR = + cfg_get(vc, "MINICPM_V_VISION_RESCALE_FACTOR", 1.0f / 255.0f); + MINICPM_V_VISION_RESCALE_IMAGE_MEAN = + cfg_get(vc, "MINICPM_V_VISION_RESCALE_IMAGE_MEAN", 0.5f); + MINICPM_V_VISION_RESCALE_IMAGE_STD = + cfg_get(vc, "MINICPM_V_VISION_RESCALE_IMAGE_STD", 0.5f); + } + +private: + struct Impl; + Impl* _impl; +}; diff --git a/src/inno/flm.iss b/src/inno/flm.iss index 75461775..fce74480 100644 --- a/src/inno/flm.iss +++ b/src/inno/flm.iss @@ -4,7 +4,7 @@ AppName=flm -AppVersion=1.0.7 +AppVersion=1.0.8 AppPublisher=FastFlowLM @@ -63,6 +63,7 @@ Source: "qwen3_npu.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3vl_npu.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3vl_flash.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3_5vl_npu.dll"; DestDir: "{app}"; Flags: ignoreversion +Source: "minicpm_v_4_7_npu.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3_5_omni_npu.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3_6_moe_npu.dll"; DestDir: "{app}"; Flags: ignoreversion Source: "qwen3_8mtp_npu.dll"; DestDir: "{app}"; Flags: ignoreversion diff --git a/src/lib/hrx/libminicpm_v_4_7_npu.so b/src/lib/hrx/libminicpm_v_4_7_npu.so new file mode 100644 index 00000000..165def4e Binary files /dev/null and b/src/lib/hrx/libminicpm_v_4_7_npu.so differ diff --git a/src/lib/hrx/minicpm_v_4_7_npu.dll b/src/lib/hrx/minicpm_v_4_7_npu.dll new file mode 100644 index 00000000..03721e99 Binary files /dev/null and b/src/lib/hrx/minicpm_v_4_7_npu.dll differ diff --git a/src/lib/hrx/minicpm_v_4_7_npu.lib b/src/lib/hrx/minicpm_v_4_7_npu.lib new file mode 100644 index 00000000..5e681cd0 Binary files /dev/null and b/src/lib/hrx/minicpm_v_4_7_npu.lib differ diff --git a/src/lib/xrt/libminicpm_v_4_7_npu.so b/src/lib/xrt/libminicpm_v_4_7_npu.so new file mode 100644 index 00000000..615a5e16 Binary files /dev/null and b/src/lib/xrt/libminicpm_v_4_7_npu.so differ diff --git a/src/lib/xrt/minicpm_v_4_7_npu.dll b/src/lib/xrt/minicpm_v_4_7_npu.dll new file mode 100644 index 00000000..a9b55dc8 Binary files /dev/null and b/src/lib/xrt/minicpm_v_4_7_npu.dll differ diff --git a/src/lib/xrt/minicpm_v_4_7_npu.lib b/src/lib/xrt/minicpm_v_4_7_npu.lib new file mode 100644 index 00000000..35e69229 Binary files /dev/null and b/src/lib/xrt/minicpm_v_4_7_npu.lib differ diff --git a/src/model_info.json b/src/model_info.json index 1d5e3670..8981d61b 100644 --- a/src/model_info.json +++ b/src/model_info.json @@ -1149,6 +1149,63 @@ "path": "vision_weight.q4nx" } ], + "minicpm-v-4.7:1b": + [ + { + "type": "file", + "oid": "0948ab105fab2a52f84bdc85c3a784221f8452d4", + "size": 7250, + "path": "chat_template.jinja" + }, + { + "type": "file", + "oid": "71dcc793b79ca4b539b00c4e7437e4848abeff72", + "size": 2880, + "path": "config.json" + }, + { + "type": "file", + "oid": "dcc79b367f8f858d9b1135a10f380c526051aeb4", + "size": 1068839584, + "lfs": { + "oid": "45d35fc84c21de51bdaab7990a84e52ec68b6b0200ab6b4a37d095a4b7830882", + "size": 1068839584, + "pointerSize": 135 + }, + "xetHash": "252b39e15006a1eb791df098e2edfe1f30857abf622e47465211841790b498b2", + "path": "model.q4nx" + }, + { + "type": "file", + "oid": "20e07d534707e3e879933e9f39dfb631ff1cbd9a", + "size": 19992481, + "lfs": { + "oid": "33861e37bb955af1e3f3061182b820f347eba2b9c2c1011c82794bf0d6e77b54", + "size": 19992481, + "pointerSize": 133 + }, + "xetHash": "f86bdcfc3aa736aa1f8a745e0eca3fc68513697f8e01be57a96f813e7dc32da2", + "path": "tokenizer.json" + }, + { + "type": "file", + "oid": "b2798b89454026d0968018f5b97e0cf825f6dd29", + "size": 9022, + "path": "tokenizer_config.json" + }, + { + "type": "file", + "oid": "a565f8ad42f6cabf681c45154c93bc5773992623", + "size": 1135147712, + "lfs": { + "oid": "3e1f4168ea50db218659de01787d7aca355fa5769e01c1af6bc454e6169f37bf", + "size": 1135147712, + "pointerSize": 135 + }, + "xetHash": "ffae0462aa68358b7f116dc700c8d92cc71dcc4ba3e0b6cd021f04e1d40d0c1b", + "path": "vision_weight.q4nx" + } + ], "lfm2:1.2b": [ { "type": "file", diff --git a/src/model_list.json b/src/model_list.json index 30320654..ee9ba48e 100644 --- a/src/model_list.json +++ b/src/model_list.json @@ -454,6 +454,41 @@ } }, + "minicpm-v-4.7": { + "1b": { + "supported_platforms": ["aie2p"], + "name": "MiniCPM-V-4.7-1B-NPU2", + "url": "https://huggingface.co/FastFlowLM/MiniCPM-V-4.7-1B-NPU2/resolve/main", + "file_url": "https://huggingface.co/api/models/FastFlowLM/MiniCPM-V-4.7-1B-NPU2/tree/main", + "ms_url": "https://modelscope.cn/models/amd/MiniCPM-V-4.7-1B-NPU2", + "size": 1000000000, + "flm_min_version": "1.0.8", + "files": [ + "config.json", + "model.q4nx", + "tokenizer.json", + "tokenizer_config.json", + "vision_weight.q4nx", + "chat_template.jinja" + ], + "vlm": true, + "default_context_length": 32768, + "max_prefill_len": 4096, + "details": { + "format": "NPU2", + "family": "minicpm-v-4.7", + "think": true, + "parameter_size": "1B", + "quantization_level": "Q4_K" + }, + "label": [ + "vision", + "reasoning" + ], + "footprint": 1.3 + } + }, + "qwen3.6-moe": { "35b-a3b": { "supported_platforms": ["aie2p"], diff --git a/src/test/minicpm_v_4_7_npu/CMakeLists.txt b/src/test/minicpm_v_4_7_npu/CMakeLists.txt new file mode 100644 index 00000000..90cd5338 --- /dev/null +++ b/src/test/minicpm_v_4_7_npu/CMakeLists.txt @@ -0,0 +1,27 @@ +cmake_minimum_required(VERSION 3.22) +project(minicpm_v_4_7_npu VERSION 1.0.0 LANGUAGES CXX) + +include(${CMAKE_CURRENT_LIST_DIR}/../CMakeLists.txt) +npu_test_setup() + +add_npu_test( + test_minicpm_v_4_7_npu + test/minicpm_v_4_7_npu + USE_AUTOMODEL + USE_TOKENIZER + USE_SAMPLER + SOURCES + "${CMAKE_SOURCE_DIR}/../../common/AutoModel/model_backend.cpp" + "${CMAKE_SOURCE_DIR}/../../common/AutoModel/modeling_minicpm_v.cpp" +) + +target_link_libraries(test_minicpm_v_4_7_npu PUBLIC + minicpm_v_4_7_npu + xrt_coreutil +) + +# Add test target +add_custom_target(test_minicpm_v_4_7_npu_target + DEPENDS test_minicpm_v_4_7_npu + COMMENT "Building test_minicpm_v_4_7_npu executable" +) diff --git a/src/test/minicpm_v_4_7_npu/Makefile b/src/test/minicpm_v_4_7_npu/Makefile new file mode 100644 index 00000000..dd6a66ab --- /dev/null +++ b/src/test/minicpm_v_4_7_npu/Makefile @@ -0,0 +1,98 @@ +# ============================================================================= +# MiniCPM-V NPU test Makefile +# ============================================================================= +# +# Usage: +# make - Build the test +# make test - Build and run minicpm-v-4.7:1b (prompt, prompt-cache follow-up, +# fresh prompt with thinking) +# make clean - Remove all built files +# +# ============================================================================= +-include ../common.mk + +LENGTH ?= 256 + +SOURCES += test.cpp +SOURCES += ../../common/AutoModel/automodel.cpp +SOURCES += ../../common/AutoModel/model_backend.cpp +SOURCES += ../../common/AutoModel/modeling_minicpm_v.cpp +# Image support. modeling_minicpm_v_image.cpp holds the slicing, the PIL-exact +# resize and image_placeholder(); image_reader.cpp decodes the file via ffmpeg; +# imageprocAVX512.cpp supplies rescale_and_normalize_avx512(), and pulls in +# imageproc.cpp for resize_bicubic_* at link time. +# Same file list as the other VLM harnesses. MiniCPM does NOT call the shared +# resize -- it ships a PIL-exact one in modeling_minicpm_v_image.cpp, because the +# shared bicubic differs from PIL by up to 29 levels and SigLIP amplifies that +# more than every device rounding combined +# (notes/tricks/2026-10-05-siglip-amplifies-resize-differences.md in FLM_IRON). +# imageproc.cpp is still linked: imageprocAVX512.cpp references it regardless. +SOURCES += ../../common/AutoModel/modeling_minicpm_v_image.cpp +SOURCES += ../../common/image/image_reader.cpp +SOURCES += ../../common/image_process_utils/imageproc.cpp +SOURCES += ../../common/image_process_utils/imageprocAVX512.cpp +SOURCES += ../../common/tokenizer/tokenizer.cpp +SOURCES += ../../common/modules/sampler.cpp + +HEADERS += ../../include/models/minicpm_v_4_7/flm/aie2p/minicpm_v_4_7_npu.hpp +HEADERS += ../../include/AutoModel/modeling_minicpm_v.hpp + + +ifeq ($(WSL), 0) + +LDFLAGS += -lminicpm_v_4_7_npu +# image_reader.cpp decodes through ffmpeg, same as every other VLM harness +LDFLAGS += -lavformat -lavcodec -lavutil -lswscale -lswresample + +CPP_SOURCES := $(filter %.cpp,$(SOURCES)) +OBJECTS := $(addprefix $(BUILD_DIR)/,$(notdir $(CPP_SOURCES:.cpp=.o))) +DEPS := $(OBJECTS:.o=.d) +VPATH := $(sort $(dir $(CPP_SOURCES))) + +all: $(BUILD_DIR)/test_minicpm_v_4_7_npu + +$(BUILD_DIR): + mkdir -p $(BUILD_DIR) + +$(BUILD_DIR)/%.o: %.cpp | $(BUILD_DIR) + $(CXX) $(CXX_FLAGS) -c $< -o $@ + +$(BUILD_DIR)/test_minicpm_v_4_7_npu: $(OBJECTS) + $(CXX) $(CXX_FLAGS) -o $@ $^ $(LDFLAGS) $(DEPENDENCY_LDFLAGS) + +clean: + rm -rf $(BUILD_DIR) + +test: $(BUILD_DIR)/test_minicpm_v_4_7_npu + cp ../../model_list.json $(BUILD_DIR)/model_list.json + cd $(BUILD_DIR) && ./test_minicpm_v_4_7_npu --model minicpm-v-4.7:1b --Length $(LENGTH) + +-include $(DEPS) +.PHONY: all clean test + +else + +# WSL build environment +# Use CMake to invoke the Visual Studio +PWSH := powershell.exe + +all: directories test + +directories: + mkdir -p $(BUILD_DIR) + +$(BUILD_DIR)/test_minicpm_v_4_7_npu.exe: $(SOURCES) + cd $(BUILD_DIR) && $(PWSH) -Command "cmake ../../../test/minicpm_v_4_7_npu" + cd $(BUILD_DIR) && $(PWSH) -Command "cmake --build . --config Release --target test_minicpm_v_4_7_npu" + +clean: + rm -rf $(BUILD_DIR) + + +test: directories $(BUILD_DIR)/test_minicpm_v_4_7_npu.exe + cp ../../model_list.json $(BUILD_DIR)/model_list.json + cd $(BUILD_DIR) && ${PWSH} -Command "\$$env:PATH = '..\..\..\lib;' + \$$env:PATH; .\test_minicpm_v_4_7_npu.exe -m minicpm-v-4.7:1b -l $(LENGTH)" + +.PHONY: all clean test directories + +endif diff --git a/src/test/minicpm_v_4_7_npu/activate.sh b/src/test/minicpm_v_4_7_npu/activate.sh new file mode 100755 index 00000000..74b2bfbb --- /dev/null +++ b/src/test/minicpm_v_4_7_npu/activate.sh @@ -0,0 +1,119 @@ +#!/usr/bin/env bash +# Set up the environment for the MiniCPM-V-4.7-1B test harness. +# +# source activate.sh # stage whatever engine is already built +# source activate.sh host # rebuild the all-host engine first (no NPU) +# source activate.sh npu # rebuild the all-NPU engine first (S4/S5) +# +# then: +# make test # or: make test LENGTH=64 +# # (the thinking turn generates 4x LENGTH) +# +# Why this is needed: the harness links -lminicpm_v_4_7_npu out of src/lib/xrt/, but +# that engine is built in the *other* repo (FastFlowLM_IRON) and is deliberately +# not committed here -- it is a build artifact whose host/NPU flavour changes. +# Without staging it you get: +# /usr/bin/ld: cannot find -lminicpm_v_4_7_npu: No such file or directory +# +# It also exports FLM_MODEL_PATH so the harness finds the model in /scratch +# rather than ~/.config/flm. model_list.json's "model_path" is "models", and +# model_root_path is FLM_MODEL_PATH/models, so point this at the PARENT of the +# model folders. + +# --- settings, override by exporting before sourcing ------------------------- +IRON_REPO="${IRON_REPO:-/scratch/michyu/Projects/FastFlowLM_IRON}" +FLM_MODEL_PATH="${FLM_MODEL_PATH:-/scratch/michyu}" +MODEL_NAME="${MODEL_NAME:-MiniCPM-V-4.7-1B-NPU2}" + +# Host build: every stage on the CPU. Correct but slow (~20 tok/s decode). +# The polarity is opt-OUT -- a plain `make` is all-NPU. +HOST_FLAGS="PREFILL_MM=cpu PREFILL_ATTN=cpu PREFILL_GDN=cpu PREFILL_CONV=cpu DECODE=cpu" + +# Tolerate `bash activate.sh` as well as `source activate.sh`; only the latter +# makes the exports stick, so say so rather than silently doing half the job. +_sourced=1 +[ "${BASH_SOURCE[0]}" = "${0}" ] && _sourced=0 + +_here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +_libdir="$(cd "$_here/../../lib/xrt" && pwd)" +_engdir="$IRON_REPO/FLM_DLL/detail/minicpm_v_4_7_npu" +_built="$IRON_REPO/FLM_DLL/build/lib/libminicpm_v_4_7_npu.so" + +_fail() { echo "[activate] ERROR: $*" >&2; return 1; } + +_main() { + [ -d "$_engdir" ] || { _fail "engine dir not found: $_engdir (set IRON_REPO)"; return 1; } + + case "${1:-}" in + host) echo "[activate] building all-host engine ..." + make -C "$_engdir" -j"$(nproc)" $HOST_FLAGS >/tmp/minicpm_v_activate_build.log 2>&1 \ + || { tail -25 /tmp/minicpm_v_activate_build.log; _fail "build failed"; return 1; } ;; + npu) echo "[activate] building all-NPU engine ..." + make -C "$_engdir" -j"$(nproc)" >/tmp/minicpm_v_activate_build.log 2>&1 \ + || { tail -25 /tmp/minicpm_v_activate_build.log; _fail "build failed"; return 1; } ;; + "") ;; + *) _fail "unknown argument '$1' (expected: host, npu, or nothing)"; return 1 ;; + esac + + [ -f "$_built" ] || { _fail "no engine built yet: $_built + run: source activate.sh host (all-CPU, works without an NPU) + or: source activate.sh npu (needs S4/S5 and the NPU box)"; return 1; } + + # Stale-.so check. build/ is gitignored and therefore per-worktree, so + # several copies of this file exist on disk. AGENTS.md section 8: a stale + # .so makes experiments lie. Compare against the newest engine source. + local newest + newest="$(find "$_engdir" -name '*.cpp' -o -name '*.hpp' | xargs ls -t 2>/dev/null | head -1)" + if [ -n "$newest" ] && [ "$newest" -nt "$_built" ]; then + echo "[activate] WARNING: engine is OLDER than $(basename "$newest")." + echo "[activate] rebuild with: source activate.sh host|npu" + fi + + install -m 755 "$_built" "$_libdir/libminicpm_v_4_7_npu.so" || { _fail "could not stage into $_libdir"; return 1; } + echo "[activate] staged $(basename "$_built") -> src/lib/xrt/ ($(date -r "$_built" '+%H:%M'))" + + # Report which flavour was staged, so a surprising tok/s is explained before + # it is chased. A -D define leaves no trace in the binary, so test for the + # "layer.xclbin" literal instead: its register_xclbin() call sits inside + # #if MINICPM_V_DECODE_ON_NPU, so the host build does not contain it. + # (Do not test lm_head.xclbin -- that string appears in both, from the + # shared detail/lm_head objects linked into every engine.) + if strings "$_libdir/libminicpm_v_4_7_npu.so" | grep -q 'layer\.xclbin'; then + echo "[activate] flavour: NPU (expect ~44-47 tok/s decode in make test)" + # The harness is built with -DDEV_BUILD, which hardwires LM_Config's + # exec_path to "../../../" relative to its cwd -- i.e. src/ -- so it reads + # src/xclbins// and IGNORES FLM_XCLBIN_PATH. Check that directory. + local xdir="$_here/../../xclbins/$MODEL_NAME" + local x missing="" + for x in layer lm_head mm attn conv GateDeltaNet_prefill; do + [ -f "$xdir/$x.xclbin" ] || missing="$missing $x.xclbin" + done + if [ -n "$missing" ]; then + echo "[activate] WARNING: missing in src/xclbins/$MODEL_NAME/:$missing" + echo "[activate] the text tower reuses Qwen3.5-0.8B's bitstreams unchanged:" + echo "[activate] mkdir -p $xdir && cp $_here/../../xclbins/Qwen3.5-0.8B-NPU2/{layer,lm_head,mm,attn,conv,GateDeltaNet_prefill}.xclbin $xdir/" + fi + else + echo "[activate] flavour: ALL-HOST -- decode will be ~17-20 tok/s, NOT the NPU number." + echo "[activate] for NPU numbers: source activate.sh npu" + fi + + local mdir="$FLM_MODEL_PATH/models/$MODEL_NAME" + [ -d "$mdir" ] || echo "[activate] WARNING: model folder not found: $mdir" + [ -f "$mdir/model.q4nx" ] || echo "[activate] WARNING: no model.q4nx in $mdir" + + export FLM_MODEL_PATH + echo "[activate] FLM_MODEL_PATH=$FLM_MODEL_PATH (resolves $FLM_MODEL_PATH/models/$MODEL_NAME)" + + if [ "$_sourced" = "0" ]; then + echo "[activate] NOTE: run with 'source activate.sh' -- executed directly, FLM_MODEL_PATH will not persist." + else + echo "[activate] ready: make test # add LENGTH=64 for a quicker run" + fi +} + +_main "${1:-}" +_activate_rc=$? +unset -f _main _fail +unset _here _libdir _engdir _built _sourced +return $_activate_rc 2>/dev/null || exit $_activate_rc diff --git a/src/test/minicpm_v_4_7_npu/test.cpp b/src/test/minicpm_v_4_7_npu/test.cpp new file mode 100644 index 00000000..e420a4d3 --- /dev/null +++ b/src/test/minicpm_v_4_7_npu/test.cpp @@ -0,0 +1,201 @@ +/// \file test.cpp +/// \brief standalone harness for MiniCPM_V +/// \note Four turns: a CLI-style prompt, a REST-style follow-up that restores +/// the prompt-cache checkpoint, then clear_context() and a fresh prompt +/// with thinking on, then an image. Each turn reports the token that +/// stopped it. +#include +#include +#include +#include +#include +#include "utils/utils.hpp" +#include "utils/vm_args.hpp" +#include "AutoModel/modeling_minicpm_v.hpp" +#include "AutoModel/flm_backend.hpp" +#include "model_list.hpp" + +flm_rt::device npu_device_global; + +/// \note builtin_backends.cpp names every engine and so links every prebuilt +/// library; this harness registers just the one it drives. +namespace flm::backend { +void register_builtin_backends(BackendRegistry& registry) { + registry.register_backend("minicpm-v-4.7", kFlmBackendId, flm_factory()); +} +} // namespace flm::backend + +/// \brief Name a picture the way a person would: "720p", "1080p", ... +/// \note The label keys off the SHORT side, which is what the p-numbers mean -- +/// 960x720 and 1280x720 are both 720p. It is a label, not a measurement: +/// anything that is not within 5% of a standard height is reported by its +/// pixel count instead of being rounded into the nearest bucket. +static std::string resolution_label(int w, int h) { + const int shortest = std::min(w, h); + static const int kStandard[] = {480, 576, 720, 1080, 1440, 2160, 4320}; + for (int s : kStandard) { + if (std::abs(shortest - s) * 20 <= s) return std::to_string(s) + "p"; + } + std::ostringstream mp; + mp.precision(1); + mp << std::fixed << (w * (double)h / 1e6) << " MP"; + return mp.str(); +} + +/// \brief Decode the image header and say how big it is. +/// \note MiniCPM's TTFT is driven by how many VIEWS the slicer makes, which is a +/// function of both the pixel count and the aspect ratio (get_sliced_grid +/// picks a grid, capped by max_slice_nums), so print both. The view count +/// itself is private to MiniCPM_V -- read it off the "vision:" line that +/// FLM_MINICPM_V_VISION_PROFILE=1 prints. +static void report_image(const std::string& path) { + ImageReader reader; + image_data_t decoded; + if (!reader.load_image(path, decoded)) { + header_print("WARNING", "could not decode " << path << " to report its size"); + return; + } + const int w = decoded.width, h = decoded.height; + reader.recycle(decoded); + std::ostringstream aspect; + aspect.precision(2); + aspect << std::fixed << (w / (double)h); + header_print("info", "image: " << w << "x" << h << " (" << resolution_label(w, h) + << ", aspect " << aspect.str() << ")"); +} + +static void report_turn(AutoModel& chat, const chat_meta_info_t& meta_info) { + std::vector history = chat.get_history().second; + std::cout << std::endl; + std::cout << "prompt_tokens: " << meta_info.prompt_tokens + << ", cached_prompt_tokens: " << meta_info.cached_prompt_tokens + << ", generated_tokens: " << meta_info.generated_tokens + << ", stop_reason: " << stop_reason_to_string(meta_info.stop_reason) << std::endl; + std::cout << "last token: " << (history.empty() ? -1 : history.back()) + << ", context length: " << chat.get_current_context_length() << std::endl; + std::cout << chat.show_profile() << std::endl; +} + +int main(int argc, char* argv[]) { + arg_utils::po::options_description desc("Allowed options"); + arg_utils::po::variables_map vm; + desc.add_options()("model,m", arg_utils::po::value()->default_value("minicpm-v-4.7:1b"), "Model tag"); + desc.add_options()("Length,l", arg_utils::po::value()->default_value(256), "Max generated tokens per turn"); + desc.add_options()("Preemption,p", arg_utils::po::value()->default_value(false), "Preemption"); + // Default is the repo's own test image, found relative to the executable the + // same way model_list.json is. Pass --image "" to skip turn 4 outright. + desc.add_options()("image,i", arg_utils::po::value()->default_value(""), + "Image for turn 4; \"\" skips it"); + arg_utils::po::store(arg_utils::po::parse_command_line(argc, argv, desc), vm); + + std::string tag = vm["model"].as(); + int length = vm["Length"].as(); + bool preemption = vm["Preemption"].as(); + std::cout << "Model: " << tag << std::endl; + std::string exe_dir = utils::get_executable_directory(); + std::string model_dir = utils::get_models_directory(); + std::string model_list_path = exe_dir + "/model_list.json"; + model_list model_list(model_list_path, model_dir); + + header_print("info", "Initializing chat model..."); + std::string model_path = model_list.get_model_path(tag); + nlohmann::json model_info = model_list.get_model_info(tag).second; + std::cout << "Model path: " << model_path << std::endl; + + std::unique_ptr chat = std::make_unique(&npu_device_global); + npu_device_global = flm_rt::device(0); + chat->load_model(model_path, model_info, -1, preemption); + header_print("info", "Model loaded"); + chat->set_topk(1); + chat->configure_parameter("enable_think", false); + + // Turn 1: a pure prompt, as `flm run` sends it. + lm_uniform_input_t input; + input.prompt = "My name is Ada and I live in Lisbon. In one sentence, what is Lisbon known for?"; + std::cout << "Prompt: " << input.prompt << std::endl << "Response: " << std::endl; + chat_meta_info_t meta_info; + chat->start_total_timer(); + chat->insert(meta_info, input); + std::string answer = chat->generate(meta_info, length, std::cout); + chat->stop_total_timer(); + report_turn(*chat, meta_info); + + // Turn 2: the whole conversation, as `flm serve` sends it on a prompt-cache + // hit. restore() rewinds to the end of turn 1's prompt, so only the + // assistant reply and the new user turn should be prefilled. + lm_uniform_input_t follow_up; + follow_up.messages = nlohmann::ordered_json::array(); + follow_up.messages.push_back({ {"role", "user"}, {"content", input.prompt} }); + follow_up.messages.push_back({ {"role", "assistant"}, {"content", answer} }); + follow_up.messages.push_back({ {"role", "user"}, {"content", "What is my name, and which city did I mention?"} }); + std::cout << "Prompt: " << follow_up.messages.back()["content"].get() << std::endl << "Response: " << std::endl; + chat_meta_info_t meta_info2; + meta_info2.restore_allowed = true; + chat->start_total_timer(); + chat->insert(meta_info2, follow_up); + chat->generate(meta_info2, length, std::cout); + chat->stop_total_timer(); + report_turn(*chat, meta_info2); + + // Turn 3: a fresh conversation with thinking on. + chat->clear_context(); + chat->configure_parameter("enable_think", true); + lm_uniform_input_t fresh; + fresh.prompt = "What is 17 * 23? Answer briefly."; + std::cout << "Prompt: " << fresh.prompt << std::endl << "Response: " << std::endl; + chat_meta_info_t meta_info3; + chat->start_total_timer(); + chat->insert(meta_info3, fresh); + chat->generate(meta_info3, 4 * length, std::cout); + chat->stop_total_timer(); + report_turn(*chat, meta_info3); + + // Turn 4: an image, as `flm run` sends it after /input. The vision tower only + // runs if the package has one -- config.json needs vision_model_weight, which + // is what sets is_vlm. Without it MiniCPM_V warns and drops the image, and + // this turn degenerates to a text question about nothing, so say so rather + // than let a 25-token prefill read as success. + std::string image = vm["image"].as(); + if (image == "") + image = (std::filesystem::path(exe_dir) / ".." / ".." / ".." / ".." + / ".github" / "script" / "assets" / "test_image1.jpg").lexically_normal().string(); + + if (image.empty()) { + header_print("info", "turn 4 (image) skipped: --image \"\""); + } else if (!std::filesystem::exists(image)) { + header_print("WARNING", "turn 4 (image) skipped: no such file: " << image); + } else { + chat->clear_context(); + chat->configure_parameter("enable_think", false); + lm_uniform_input_t with_image; + with_image.prompt = "Describe this image."; + with_image.images.push_back(image); + std::cout << "Image: " << image << std::endl; + report_image(image); + std::cout << "Prompt: " << with_image.prompt << std::endl << "Response: " << std::endl; + chat_meta_info_t meta_info4; + chat->start_total_timer(); + chat->insert(meta_info4, with_image); + chat->generate(meta_info4, length, std::cout); + chat->stop_total_timer(); + report_turn(*chat, meta_info4); + + // The tower turns the repo's 960x720 test image into 5 views / 5040 + // patches / 315 rows, so a real image prefill is ~330 tokens against ~15 + // for the question alone. Anything near the latter means the image never + // reached the tower -- a missing vision_model_weight, or a path the + // reader could not decode -- which is otherwise easy to miss, because the + // model still answers, just from the question alone. + constexpr int kMinImagePromptTokens = 100; + if (meta_info4.prompt_tokens < kMinImagePromptTokens) { + header_print("WARNING", "prompt_tokens=" << meta_info4.prompt_tokens + << " is too low for an image turn; the image was NOT consumed" + " (is_vlm false, or the file failed to decode)"); + } else { + header_print("info", "image consumed: " << meta_info4.prompt_tokens + << " prompt tokens (question alone would be ~15)"); + } + } + + return 0; +} diff --git a/src/wix/flm.wxs b/src/wix/flm.wxs index 6c5bb138..251ffd59 100644 --- a/src/wix/flm.wxs +++ b/src/wix/flm.wxs @@ -28,7 +28,7 @@ + + + diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/GateDeltaNet_prefill.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/GateDeltaNet_prefill.xclbin new file mode 100644 index 00000000..c0a69726 Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/GateDeltaNet_prefill.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/attn.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/attn.xclbin new file mode 100644 index 00000000..6bcd5ece Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/attn.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/conv.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/conv.xclbin new file mode 100644 index 00000000..7d70327b Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/conv.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/layer.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/layer.xclbin new file mode 100644 index 00000000..5aa3c61a Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/layer.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/lm_head.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/lm_head.xclbin new file mode 100644 index 00000000..27fde16a Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/lm_head.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/mm.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/mm.xclbin new file mode 100644 index 00000000..b5393ec6 Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/mm.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_attn.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_attn.xclbin new file mode 100644 index 00000000..0c369d79 Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_attn.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_a.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_a.xclbin new file mode 100644 index 00000000..45eae121 Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_a.xclbin differ diff --git a/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_b.xclbin b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_b.xclbin new file mode 100644 index 00000000..8dd2e10b Binary files /dev/null and b/src/xclbins/MiniCPM-V-4.7-1B-NPU2/vision_mm_b.xclbin differ